chore: normalize line endings (CRLF -> LF)
No content changes: git diff --ignore-all-space over these files is empty. The churn came from editing on Windows against a repo checked out with LF.
This commit is contained in:
1 parent
15566a6951
commit
caf8e98378
315 files changed
+86950
-86950
No files matched your search
+14
-14
@@ -1,14 +1,14 @@
|
||||
.git
|
||||
.github
|
||||
.venv
|
||||
.venv-api
|
||||
PaddleOCR-VL-1.6_Online_Demo/.venv
|
||||
**/__pycache__
|
||||
**/*.pyc
|
||||
.cache
|
||||
.python-version
|
||||
issues
|
||||
*.md
|
||||
pfm-web-app/node_modules
|
||||
pfm-web-app/.next
|
||||
|
||||
.git
|
||||
.github
|
||||
.venv
|
||||
.venv-api
|
||||
PaddleOCR-VL-1.6_Online_Demo/.venv
|
||||
**/__pycache__
|
||||
**/*.pyc
|
||||
.cache
|
||||
.python-version
|
||||
issues
|
||||
*.md
|
||||
pfm-web-app/node_modules
|
||||
pfm-web-app/.next
|
||||
|
||||
@@ -1,8 +1,8 @@
|
||||
# Port to serve the application on the host (routed via Nginx)
|
||||
APP_PORT=8000
|
||||
|
||||
# GPU index to allocate (e.g. 0, 1, or 0,1)
|
||||
CUDA_VISIBLE_DEVICES=1
|
||||
|
||||
# Secret used to sign/verify account login JWTs
|
||||
JWT_SECRET=change-me
|
||||
# Port to serve the application on the host (routed via Nginx)
|
||||
APP_PORT=8000
|
||||
|
||||
# GPU index to allocate (e.g. 0, 1, or 0,1)
|
||||
CUDA_VISIBLE_DEVICES=1
|
||||
|
||||
# Secret used to sign/verify account login JWTs
|
||||
JWT_SECRET=change-me
|
||||
+18
-18
@@ -1,18 +1,18 @@
|
||||
.venv/
|
||||
.venv-api/
|
||||
.env
|
||||
__pycache__/
|
||||
*.pyc
|
||||
.python-version
|
||||
.antigravitycli/
|
||||
PaddleOCR-VL-1.6_Online_Demo/examples/do-pfm/*.jpg
|
||||
pfm-web-app/public/do-pfm/*
|
||||
uploads/*
|
||||
pfm-web-app/public/produk-pfm/**/*.jpeg
|
||||
pfm-web-app/public/produk-pfm/yolo_dataset/*
|
||||
pfm-web-app/public/produk-pfm/runs/*
|
||||
pfm-web-app/public/produk-pfm/models/*
|
||||
*.pt
|
||||
test_img.jpeg
|
||||
screenshots/*.jpg
|
||||
pfm-web-app/public/test-images/*
|
||||
.venv/
|
||||
.venv-api/
|
||||
.env
|
||||
__pycache__/
|
||||
*.pyc
|
||||
.python-version
|
||||
.antigravitycli/
|
||||
PaddleOCR-VL-1.6_Online_Demo/examples/do-pfm/*.jpg
|
||||
pfm-web-app/public/do-pfm/*
|
||||
uploads/*
|
||||
pfm-web-app/public/produk-pfm/**/*.jpeg
|
||||
pfm-web-app/public/produk-pfm/yolo_dataset/*
|
||||
pfm-web-app/public/produk-pfm/runs/*
|
||||
pfm-web-app/public/produk-pfm/models/*
|
||||
*.pt
|
||||
test_img.jpeg
|
||||
screenshots/*.jpg
|
||||
pfm-web-app/public/test-images/*
|
||||
+18
-18
@@ -1,18 +1,18 @@
|
||||
[submodule "deepseek-ocr-2-demo-2026"]
|
||||
path = deepseek-ocr-2-demo-2026
|
||||
url = https://github.com/abdshomad/deepseek-ocr-2-demo-2026.git
|
||||
[submodule "LightOnOCR-2-1B-Demo-2026"]
|
||||
path = LightOnOCR-2-1B-Demo-2026
|
||||
url = https://github.com/abdshomad/LightOnOCR-2-1B-Demo-2026.git
|
||||
[submodule "nvidia-nemotron-ocr-v2-demo-2026"]
|
||||
path = nvidia-nemotron-ocr-v2-demo-2026
|
||||
url = https://github.com/abdshomad/nvidia-nemotron-ocr-v2-demo-2026.git
|
||||
[submodule "dots.ocr-demo-2026"]
|
||||
path = dots.ocr-demo-2026
|
||||
url = https://github.com/abdshomad/dots.ocr-demo-2026.git
|
||||
[submodule "glm-ocr-demo-2026"]
|
||||
path = glm-ocr-demo-2026
|
||||
url = https://github.com/abdshomad/glm-ocr-demo-2026.git
|
||||
[submodule "andrej-karpathy-skills"]
|
||||
path = andrej-karpathy-skills
|
||||
url = https://github.com/multica-ai/andrej-karpathy-skills.git
|
||||
[submodule "deepseek-ocr-2-demo-2026"]
|
||||
path = deepseek-ocr-2-demo-2026
|
||||
url = https://github.com/abdshomad/deepseek-ocr-2-demo-2026.git
|
||||
[submodule "LightOnOCR-2-1B-Demo-2026"]
|
||||
path = LightOnOCR-2-1B-Demo-2026
|
||||
url = https://github.com/abdshomad/LightOnOCR-2-1B-Demo-2026.git
|
||||
[submodule "nvidia-nemotron-ocr-v2-demo-2026"]
|
||||
path = nvidia-nemotron-ocr-v2-demo-2026
|
||||
url = https://github.com/abdshomad/nvidia-nemotron-ocr-v2-demo-2026.git
|
||||
[submodule "dots.ocr-demo-2026"]
|
||||
path = dots.ocr-demo-2026
|
||||
url = https://github.com/abdshomad/dots.ocr-demo-2026.git
|
||||
[submodule "glm-ocr-demo-2026"]
|
||||
path = glm-ocr-demo-2026
|
||||
url = https://github.com/abdshomad/glm-ocr-demo-2026.git
|
||||
[submodule "andrej-karpathy-skills"]
|
||||
path = andrej-karpathy-skills
|
||||
url = https://github.com/multica-ai/andrej-karpathy-skills.git
|
||||
+188
-188
@@ -1,188 +1,188 @@
|
||||
# AGENTS: PaddleOCR-VL-1.6 vLLM Service + Agents Settings Kit
|
||||
|
||||
This is the authoritative rules file for any AI coding agent (Claude Code, Cursor,
|
||||
GitHub Copilot, Aider, etc.) working inside `backend/`. Two unrelated concerns live
|
||||
here side by side: **Part A** is this repo's original vLLM/PaddleOCR service doc.
|
||||
**Part B** (appended 2026-07-08) is a **backend-scoped copy** of the
|
||||
[fhanyuh/agents-settings](https://github.com/fhanyuh/agents-settings) `e`/`enhance`
|
||||
and `n`/`next` workflow — see root `../AGENTS.md` for the same kit covering the
|
||||
Flutter side of this repo. The two copies are independent: this one's
|
||||
`plans/next-enhancements.md` and `docs/feature-list.md` only track backend work.
|
||||
|
||||
---
|
||||
|
||||
# Part A — vLLM Service (PaddleOCR-VL-1.6)
|
||||
|
||||
This repository serves **PaddleOCR-VL-1.6** as a dedicated VLM inference backend using **vLLM**. All Python workflows use **uv** (never bare `pip` or system Python). Full detail (client usage examples, tuning, troubleshooting, issue-file template) moved to [docs/vllm-service.md](docs/vllm-service.md) 2026-07-08 to keep this file under the Part B kit's 256-line threshold (§3) — this section keeps only the essentials.
|
||||
|
||||
## Architecture
|
||||
|
||||
```
|
||||
Client (PaddleOCR pipeline) --> HTTP /v1 --> paddleocr genai_server (vLLM backend)
|
||||
```
|
||||
|
||||
This service exposes only the VLM stage. Clients connect with `vl_rec_backend="vllm-server"` and `vl_rec_server_url="http://<host>:8118/v1"`.
|
||||
|
||||
## Quick start
|
||||
|
||||
```bash
|
||||
./scripts/install.sh # 1) Create Python 3.12 venv and install dependencies
|
||||
./scripts/serve.sh # 2) Start the vLLM-backed genai server
|
||||
```
|
||||
|
||||
Default endpoint: `http://0.0.0.0:8118/v1`. Never use `python -m pip`, `pip install`, or `python -m venv` directly in this repo — always `uv sync` / `uv run` / `uv add`.
|
||||
|
||||
## Issue recording (always follow)
|
||||
|
||||
**Every problem encountered** during install, serve, debug, or client integration must be written to `issues/{NN}-{slug}.md` before moving on — even if resolved in the same session. Naming/template details: [docs/vllm-service.md](docs/vllm-service.md#issue-recording--naming-and-template).
|
||||
|
||||
## Environment variables
|
||||
|
||||
Copy `.env.example` to `.env` and adjust as needed:
|
||||
|
||||
| Variable | Default | Description |
|
||||
|----------|---------|-------------|
|
||||
| `GENAI_HOST` | `0.0.0.0` | Bind address |
|
||||
| `GENAI_PORT` | `8118` | Service port |
|
||||
| `GENAI_MODEL` | `PaddleOCR-VL-1.6-0.9B` | Model name for `genai_server` |
|
||||
| `GENAI_BACKEND` | `vllm` | Inference backend |
|
||||
| `VLLM_CONFIG` | `config/vllm_config.yaml` | vLLM backend YAML config |
|
||||
| `CUDA_VISIBLE_DEVICES` | `1` (see `.env.example`) | GPU index(es) to use |
|
||||
|
||||
On dual-GPU hosts, pick the GPU with more free VRAM. If startup fails with a memory error, lower `gpu-memory-utilization` in `config/vllm_config.yaml` — see [docs/vllm-service.md](docs/vllm-service.md#gpu-memory-on-startup).
|
||||
|
||||
## File map
|
||||
|
||||
| Path | Purpose |
|
||||
|------|---------|
|
||||
| `issues/` | Recorded problems and fixes (`{NN}-{slug}.md`) |
|
||||
| `pyproject.toml` | uv project metadata and base dependencies |
|
||||
| `scripts/install.sh` | Bootstrap venv + vLLM server deps |
|
||||
| `scripts/serve.sh` | Start `paddleocr genai_server` |
|
||||
| `config/vllm_config.yaml` | vLLM backend tuning |
|
||||
| `.env.example` | Environment variable template |
|
||||
| `docs/vllm-service.md` | Full vLLM reference (client usage, tuning, troubleshooting) |
|
||||
|
||||
## Coding Guidelines (always follow)
|
||||
|
||||
We use the karpathy-guidelines skill to reduce common LLM coding mistakes:
|
||||
1. **Think Before Coding**: Explicitly state assumptions and surface tradeoffs instead of making silent choices.
|
||||
2. **Simplicity First**: Write the minimum amount of code to solve the problem with zero speculative configurations.
|
||||
3. **Surgical Changes**: Edit only what is required and match the existing coding style exactly.
|
||||
4. **Goal-Driven Execution**: Define verifiable success criteria and run automated tests/screenshots to confirm correctness.
|
||||
5. **SOLID Principles**: Always design, implement, and refactor code adhering to SOLID programming principles (Single Responsibility, Open/Closed, Liskov Substitution, Interface Segregation, Dependency Inversion) to ensure modularity, scalability, and maintainability.
|
||||
|
||||
## Path Guidelines (always follow)
|
||||
|
||||
Never use full paths containing the user's logged-in name (e.g., `/home/{uid}/path`). Always use relative paths instead (e.g., `.` or `./path` relative to the workspace root).
|
||||
|
||||
## App Testing Guidelines (always follow)
|
||||
|
||||
When the user intentionally asks to test the app:
|
||||
- Use browser tools to test the app.
|
||||
- Take a screenshot for each sample image, each step, and each variant/option (if any), until the OCR result appears.
|
||||
- Save the screenshots in the `/screenshots/` folder.
|
||||
- Follow the file naming convention: `{2-digit-number}-{step#}-{variant_or_options_if_any}-{slug}.jpg` (e.g., `01-step1-default-upload.jpg`).
|
||||
|
||||
---
|
||||
|
||||
# Part B — Agents Settings Kit (backend-scoped `e`/`n` workflow)
|
||||
|
||||
Backend-scoped copy of the [fhanyuh/agents-settings](https://github.com/fhanyuh/agents-settings) kit, adopted 2026-07-08. Covers only `backend/` modules (Next.js API Gateway, OCR Pipeline & Accuracy, Postgres Data Layer, DevOps/Docker) — Flutter modules are tracked by the separate copy at root `../AGENTS.md`. `../CLAUDE.md` (root) and `CLAUDE.md` (this dir) each import their own copy.
|
||||
|
||||
## B0. Adopting Into an Existing Project
|
||||
|
||||
Already done for this repo (this split *is* that adoption, mirroring root's own §0 audit). Re-run "i"/"init" here to force a re-audit of `backend/` specifically (e.g. after a large refactor).
|
||||
|
||||
## B1. Trigger "e" or "enhance"
|
||||
|
||||
- Read `plans/next-enhancements.md` (this dir) to understand current backend structure, history, and active tasks.
|
||||
- Overwrite or update the active tasks list inside it.
|
||||
- The plan must cover each backend section/module.
|
||||
- Define **exactly 3 new enhancements per section**, each with a unique number (e.g. `1.1`), a clear functional description, and status `[TODO]`.
|
||||
- Present the plan to the user in your final summary.
|
||||
|
||||
## B2. Trigger "n", "next", or "n{x}"
|
||||
|
||||
- Read `plans/next-enhancements.md` to check task status.
|
||||
- If all tasks are `[DONE]` (or none `[TODO]`), run **"e"/"enhance"** first.
|
||||
- Otherwise select the most impactful `[TODO]` task(s) by strategic value/impact — not just first-in-order. If `{x}` given, take the top `{x}` sequentially.
|
||||
|
||||
### B2a. Clarify before building ("Grill Me" step)
|
||||
|
||||
Same rule as root AGENTS.md §2a: if scope/acceptance criteria are genuinely ambiguous, ask one question at a time (`AskUserQuestion` in Claude Code) until unambiguous, and record the resolved criteria as a 1-3 line note next to the task entry before writing code. Skip when the task is already unambiguous.
|
||||
|
||||
### B2b. TDD Workflow (Test First)
|
||||
|
||||
- **Write Tests First**: Before implementing the actual feature code for a task, write automated tests defining the expected behavior.
|
||||
- **Iterate Until Green**: Run the tests to confirm they fail, then write the implementation until all tests pass perfectly.
|
||||
- **Browser Testing**: If the enhancement involves web UI or visual components, use browser tools (e.g., Chrome) to test the app visually and functionally if necessary.
|
||||
|
||||
- Implement the task(s) fully, applying the relevant role(s) from `SKILLS.md` (this dir).
|
||||
- On completion:
|
||||
1. Flip status to `[DONE]` in `plans/next-enhancements.md`.
|
||||
2. Document the feature in `docs/feature-list.md` (this dir) under the right section.
|
||||
3. **Create an Iteration Log**: Perform a code review and audit of the tasks just completed. Document this audit in `docs/iteration-log.md` (or append to it) to ensure all functions work perfectly.
|
||||
4. **Update Documentation**: Sync any architecture or workflow changes back to `CLAUDE.md` and `SKILLS.md` to keep the agent instructions current.
|
||||
- **Verify build integrity**: QA pass (golden path + edge cases + regression check on adjacent features — see `backend/CLAUDE.md`'s accuracy regression harness for OCR/parser changes specifically) and Hardware/Compatibility pass (cross-platform, GPU/VRAM footprint under Local/on-prem deployment — see Part A above).
|
||||
- State which task(s) were completed and the exact route/endpoint/menu path to see the new feature.
|
||||
|
||||
## B3. File Size & Refactoring Rules
|
||||
|
||||
Same 256-line threshold as root AGENTS.md §3, backend-wide. Applies to this file, `SKILLS.md`, and `CLAUDE.md` too — which is why Part A above was trimmed and linked out to `docs/vllm-service.md` rather than left inline.
|
||||
|
||||
## B4. Roles
|
||||
|
||||
See `SKILLS.md` (this dir) — same 5 roles as root (Architect, Backend, Frontend, QA, Hardware/Compatibility), applied to backend surfaces only (API routes, OCR pipeline, DB layer, Docker/deploy).
|
||||
|
||||
## B5. Mockup Data & Demo/Live Mode
|
||||
|
||||
Same as root AGENTS.md §5: mock data under `/data/mockup/`, a mock API layer mirroring the real backend contract, and a Demo/Live switcher. Not yet built for backend — see Adaptation Notes.
|
||||
|
||||
## B6. Cloud vs Local (On-Premise)
|
||||
|
||||
Same as root AGENTS.md §6, applied to backend service endpoints (Next.js gateway, pipeline API, vLLM server, Postgres) rather than the Flutter client's API base URL.
|
||||
|
||||
## B7. Ad-hoc Feature Requests
|
||||
|
||||
Direct feature requests not using "e"/"n": implement and document in `docs/feature-list.md` (this dir).
|
||||
|
||||
## Adaptation Notes (backend, split from root 2026-07-08)
|
||||
|
||||
- **Origin**: sections 5-8 of root `plans/next-enhancements.md` (Backend — Next.js API
|
||||
Gateway, Backend — OCR Pipeline & Accuracy, Backend — Postgres Data Layer, DevOps —
|
||||
Docker & Dev Tunnel) copied here as sections 1-4, statuses re-verified against the
|
||||
live code before the copy (not copied blind) — see task 7.1's `withTransaction`
|
||||
claim, task 5.1/5.2's dedup + timeout claims, and task 6.1's empty `models/` claim,
|
||||
all confirmed still accurate as of 2026-07-08. The root copy is frozen/archival
|
||||
(see root `AGENTS.md`'s "Scope: excludes `backend/`") rather than deleted, so this
|
||||
file — not the root one — is the single active source of truth going forward.
|
||||
- **Real commands**: `npm run dev`/`build`/`lint` in `pfm-web-app/`; accuracy
|
||||
regression harness `node pfm-web-app/scripts/accuracy-check.mts`; Python services
|
||||
via `./scripts/install.sh` + `./scripts/serve.sh` (this vLLM repo) and
|
||||
`./scripts/install-pipeline.sh` + `./scripts/serve-pipeline.sh` (pipeline API +
|
||||
classifier). Full stack: `docker compose up -d --build` **from the repo root**, not
|
||||
from inside `backend/` (see root `CLAUDE.md` — two `docker-compose.yml` files
|
||||
exist and running from here risks container-name conflicts).
|
||||
- **Pre-existing files over the 256-line threshold** (§B3 debt, not a blocker — split
|
||||
only if/when touched): `pfm-web-app/src/app/scan-pfm/page.tsx` (1169),
|
||||
`pfm-web-app/src/utils/parser.ts` (908), `pfm-web-app/src/app/page.tsx` (737),
|
||||
`config/classify_ocr_server.py` (691), `pfm-web-app/src/app/manual-label/page.tsx`
|
||||
(612), `pfm-web-app/src/app/api/parse/route.ts` (604), `pfm-web-app/src/db/init.ts`
|
||||
(477), `compare_sources_accuracy.py` (451), `pfm-web-app/src/utils/docker.ts` (362),
|
||||
`pfm-web-app/public/produk-pfm/train_classifier.py` (351), `compare_accuracy.py`
|
||||
(308), `pfm-web-app/src/app/api/arena/route.ts` (265). This file itself (`AGENTS.md`)
|
||||
was at 237 lines pre-kit and would have exceeded 256 once Part B was appended —
|
||||
hence the split into `docs/vllm-service.md`.
|
||||
- **No Demo/Live or Cloud/Local switch exists yet** (§B5, §B6) for the backend
|
||||
either. `docker-compose.override.yml` exposing `db`/`pipeline-api` directly to the
|
||||
host is a local-dev convenience, not a Cloud/Local deployment switch.
|
||||
- **Naming collision resolved by this split**: `AGENTS.md` already existed in this
|
||||
directory (vLLM service doc, committed 2026-06-30, unrelated to this kit) before
|
||||
Part B was appended — unlike root, where `AGENTS.md` didn't previously exist. Don't
|
||||
assume backend's `AGENTS.md` is kit-only when reading it from another tool; Part A
|
||||
is unrelated, pre-existing content kept for a reason.
|
||||
- **Pre-existing, unrelated governance files left as-is**: `.agents/AGENTS.md` at the
|
||||
*repo root* (different path, OCR post-processing rules) and root
|
||||
`plans/next-enhancement-plan.md` (singular, `[DONE]` QA checklist) — neither is
|
||||
part of this kit; see root `AGENTS.md`'s own Adaptation Notes.
|
||||
# AGENTS: PaddleOCR-VL-1.6 vLLM Service + Agents Settings Kit
|
||||
|
||||
This is the authoritative rules file for any AI coding agent (Claude Code, Cursor,
|
||||
GitHub Copilot, Aider, etc.) working inside `backend/`. Two unrelated concerns live
|
||||
here side by side: **Part A** is this repo's original vLLM/PaddleOCR service doc.
|
||||
**Part B** (appended 2026-07-08) is a **backend-scoped copy** of the
|
||||
[fhanyuh/agents-settings](https://github.com/fhanyuh/agents-settings) `e`/`enhance`
|
||||
and `n`/`next` workflow — see root `../AGENTS.md` for the same kit covering the
|
||||
Flutter side of this repo. The two copies are independent: this one's
|
||||
`plans/next-enhancements.md` and `docs/feature-list.md` only track backend work.
|
||||
|
||||
---
|
||||
|
||||
# Part A — vLLM Service (PaddleOCR-VL-1.6)
|
||||
|
||||
This repository serves **PaddleOCR-VL-1.6** as a dedicated VLM inference backend using **vLLM**. All Python workflows use **uv** (never bare `pip` or system Python). Full detail (client usage examples, tuning, troubleshooting, issue-file template) moved to [docs/vllm-service.md](docs/vllm-service.md) 2026-07-08 to keep this file under the Part B kit's 256-line threshold (§3) — this section keeps only the essentials.
|
||||
|
||||
## Architecture
|
||||
|
||||
```
|
||||
Client (PaddleOCR pipeline) --> HTTP /v1 --> paddleocr genai_server (vLLM backend)
|
||||
```
|
||||
|
||||
This service exposes only the VLM stage. Clients connect with `vl_rec_backend="vllm-server"` and `vl_rec_server_url="http://<host>:8118/v1"`.
|
||||
|
||||
## Quick start
|
||||
|
||||
```bash
|
||||
./scripts/install.sh # 1) Create Python 3.12 venv and install dependencies
|
||||
./scripts/serve.sh # 2) Start the vLLM-backed genai server
|
||||
```
|
||||
|
||||
Default endpoint: `http://0.0.0.0:8118/v1`. Never use `python -m pip`, `pip install`, or `python -m venv` directly in this repo — always `uv sync` / `uv run` / `uv add`.
|
||||
|
||||
## Issue recording (always follow)
|
||||
|
||||
**Every problem encountered** during install, serve, debug, or client integration must be written to `issues/{NN}-{slug}.md` before moving on — even if resolved in the same session. Naming/template details: [docs/vllm-service.md](docs/vllm-service.md#issue-recording--naming-and-template).
|
||||
|
||||
## Environment variables
|
||||
|
||||
Copy `.env.example` to `.env` and adjust as needed:
|
||||
|
||||
| Variable | Default | Description |
|
||||
|----------|---------|-------------|
|
||||
| `GENAI_HOST` | `0.0.0.0` | Bind address |
|
||||
| `GENAI_PORT` | `8118` | Service port |
|
||||
| `GENAI_MODEL` | `PaddleOCR-VL-1.6-0.9B` | Model name for `genai_server` |
|
||||
| `GENAI_BACKEND` | `vllm` | Inference backend |
|
||||
| `VLLM_CONFIG` | `config/vllm_config.yaml` | vLLM backend YAML config |
|
||||
| `CUDA_VISIBLE_DEVICES` | `1` (see `.env.example`) | GPU index(es) to use |
|
||||
|
||||
On dual-GPU hosts, pick the GPU with more free VRAM. If startup fails with a memory error, lower `gpu-memory-utilization` in `config/vllm_config.yaml` — see [docs/vllm-service.md](docs/vllm-service.md#gpu-memory-on-startup).
|
||||
|
||||
## File map
|
||||
|
||||
| Path | Purpose |
|
||||
|------|---------|
|
||||
| `issues/` | Recorded problems and fixes (`{NN}-{slug}.md`) |
|
||||
| `pyproject.toml` | uv project metadata and base dependencies |
|
||||
| `scripts/install.sh` | Bootstrap venv + vLLM server deps |
|
||||
| `scripts/serve.sh` | Start `paddleocr genai_server` |
|
||||
| `config/vllm_config.yaml` | vLLM backend tuning |
|
||||
| `.env.example` | Environment variable template |
|
||||
| `docs/vllm-service.md` | Full vLLM reference (client usage, tuning, troubleshooting) |
|
||||
|
||||
## Coding Guidelines (always follow)
|
||||
|
||||
We use the karpathy-guidelines skill to reduce common LLM coding mistakes:
|
||||
1. **Think Before Coding**: Explicitly state assumptions and surface tradeoffs instead of making silent choices.
|
||||
2. **Simplicity First**: Write the minimum amount of code to solve the problem with zero speculative configurations.
|
||||
3. **Surgical Changes**: Edit only what is required and match the existing coding style exactly.
|
||||
4. **Goal-Driven Execution**: Define verifiable success criteria and run automated tests/screenshots to confirm correctness.
|
||||
5. **SOLID Principles**: Always design, implement, and refactor code adhering to SOLID programming principles (Single Responsibility, Open/Closed, Liskov Substitution, Interface Segregation, Dependency Inversion) to ensure modularity, scalability, and maintainability.
|
||||
|
||||
## Path Guidelines (always follow)
|
||||
|
||||
Never use full paths containing the user's logged-in name (e.g., `/home/{uid}/path`). Always use relative paths instead (e.g., `.` or `./path` relative to the workspace root).
|
||||
|
||||
## App Testing Guidelines (always follow)
|
||||
|
||||
When the user intentionally asks to test the app:
|
||||
- Use browser tools to test the app.
|
||||
- Take a screenshot for each sample image, each step, and each variant/option (if any), until the OCR result appears.
|
||||
- Save the screenshots in the `/screenshots/` folder.
|
||||
- Follow the file naming convention: `{2-digit-number}-{step#}-{variant_or_options_if_any}-{slug}.jpg` (e.g., `01-step1-default-upload.jpg`).
|
||||
|
||||
---
|
||||
|
||||
# Part B — Agents Settings Kit (backend-scoped `e`/`n` workflow)
|
||||
|
||||
Backend-scoped copy of the [fhanyuh/agents-settings](https://github.com/fhanyuh/agents-settings) kit, adopted 2026-07-08. Covers only `backend/` modules (Next.js API Gateway, OCR Pipeline & Accuracy, Postgres Data Layer, DevOps/Docker) — Flutter modules are tracked by the separate copy at root `../AGENTS.md`. `../CLAUDE.md` (root) and `CLAUDE.md` (this dir) each import their own copy.
|
||||
|
||||
## B0. Adopting Into an Existing Project
|
||||
|
||||
Already done for this repo (this split *is* that adoption, mirroring root's own §0 audit). Re-run "i"/"init" here to force a re-audit of `backend/` specifically (e.g. after a large refactor).
|
||||
|
||||
## B1. Trigger "e" or "enhance"
|
||||
|
||||
- Read `plans/next-enhancements.md` (this dir) to understand current backend structure, history, and active tasks.
|
||||
- Overwrite or update the active tasks list inside it.
|
||||
- The plan must cover each backend section/module.
|
||||
- Define **exactly 3 new enhancements per section**, each with a unique number (e.g. `1.1`), a clear functional description, and status `[TODO]`.
|
||||
- Present the plan to the user in your final summary.
|
||||
|
||||
## B2. Trigger "n", "next", or "n{x}"
|
||||
|
||||
- Read `plans/next-enhancements.md` to check task status.
|
||||
- If all tasks are `[DONE]` (or none `[TODO]`), run **"e"/"enhance"** first.
|
||||
- Otherwise select the most impactful `[TODO]` task(s) by strategic value/impact — not just first-in-order. If `{x}` given, take the top `{x}` sequentially.
|
||||
|
||||
### B2a. Clarify before building ("Grill Me" step)
|
||||
|
||||
Same rule as root AGENTS.md §2a: if scope/acceptance criteria are genuinely ambiguous, ask one question at a time (`AskUserQuestion` in Claude Code) until unambiguous, and record the resolved criteria as a 1-3 line note next to the task entry before writing code. Skip when the task is already unambiguous.
|
||||
|
||||
### B2b. TDD Workflow (Test First)
|
||||
|
||||
- **Write Tests First**: Before implementing the actual feature code for a task, write automated tests defining the expected behavior.
|
||||
- **Iterate Until Green**: Run the tests to confirm they fail, then write the implementation until all tests pass perfectly.
|
||||
- **Browser Testing**: If the enhancement involves web UI or visual components, use browser tools (e.g., Chrome) to test the app visually and functionally if necessary.
|
||||
|
||||
- Implement the task(s) fully, applying the relevant role(s) from `SKILLS.md` (this dir).
|
||||
- On completion:
|
||||
1. Flip status to `[DONE]` in `plans/next-enhancements.md`.
|
||||
2. Document the feature in `docs/feature-list.md` (this dir) under the right section.
|
||||
3. **Create an Iteration Log**: Perform a code review and audit of the tasks just completed. Document this audit in `docs/iteration-log.md` (or append to it) to ensure all functions work perfectly.
|
||||
4. **Update Documentation**: Sync any architecture or workflow changes back to `CLAUDE.md` and `SKILLS.md` to keep the agent instructions current.
|
||||
- **Verify build integrity**: QA pass (golden path + edge cases + regression check on adjacent features — see `backend/CLAUDE.md`'s accuracy regression harness for OCR/parser changes specifically) and Hardware/Compatibility pass (cross-platform, GPU/VRAM footprint under Local/on-prem deployment — see Part A above).
|
||||
- State which task(s) were completed and the exact route/endpoint/menu path to see the new feature.
|
||||
|
||||
## B3. File Size & Refactoring Rules
|
||||
|
||||
Same 256-line threshold as root AGENTS.md §3, backend-wide. Applies to this file, `SKILLS.md`, and `CLAUDE.md` too — which is why Part A above was trimmed and linked out to `docs/vllm-service.md` rather than left inline.
|
||||
|
||||
## B4. Roles
|
||||
|
||||
See `SKILLS.md` (this dir) — same 5 roles as root (Architect, Backend, Frontend, QA, Hardware/Compatibility), applied to backend surfaces only (API routes, OCR pipeline, DB layer, Docker/deploy).
|
||||
|
||||
## B5. Mockup Data & Demo/Live Mode
|
||||
|
||||
Same as root AGENTS.md §5: mock data under `/data/mockup/`, a mock API layer mirroring the real backend contract, and a Demo/Live switcher. Not yet built for backend — see Adaptation Notes.
|
||||
|
||||
## B6. Cloud vs Local (On-Premise)
|
||||
|
||||
Same as root AGENTS.md §6, applied to backend service endpoints (Next.js gateway, pipeline API, vLLM server, Postgres) rather than the Flutter client's API base URL.
|
||||
|
||||
## B7. Ad-hoc Feature Requests
|
||||
|
||||
Direct feature requests not using "e"/"n": implement and document in `docs/feature-list.md` (this dir).
|
||||
|
||||
## Adaptation Notes (backend, split from root 2026-07-08)
|
||||
|
||||
- **Origin**: sections 5-8 of root `plans/next-enhancements.md` (Backend — Next.js API
|
||||
Gateway, Backend — OCR Pipeline & Accuracy, Backend — Postgres Data Layer, DevOps —
|
||||
Docker & Dev Tunnel) copied here as sections 1-4, statuses re-verified against the
|
||||
live code before the copy (not copied blind) — see task 7.1's `withTransaction`
|
||||
claim, task 5.1/5.2's dedup + timeout claims, and task 6.1's empty `models/` claim,
|
||||
all confirmed still accurate as of 2026-07-08. The root copy is frozen/archival
|
||||
(see root `AGENTS.md`'s "Scope: excludes `backend/`") rather than deleted, so this
|
||||
file — not the root one — is the single active source of truth going forward.
|
||||
- **Real commands**: `npm run dev`/`build`/`lint` in `pfm-web-app/`; accuracy
|
||||
regression harness `node pfm-web-app/scripts/accuracy-check.mts`; Python services
|
||||
via `./scripts/install.sh` + `./scripts/serve.sh` (this vLLM repo) and
|
||||
`./scripts/install-pipeline.sh` + `./scripts/serve-pipeline.sh` (pipeline API +
|
||||
classifier). Full stack: `docker compose up -d --build` **from the repo root**, not
|
||||
from inside `backend/` (see root `CLAUDE.md` — two `docker-compose.yml` files
|
||||
exist and running from here risks container-name conflicts).
|
||||
- **Pre-existing files over the 256-line threshold** (§B3 debt, not a blocker — split
|
||||
only if/when touched): `pfm-web-app/src/app/scan-pfm/page.tsx` (1169),
|
||||
`pfm-web-app/src/utils/parser.ts` (908), `pfm-web-app/src/app/page.tsx` (737),
|
||||
`config/classify_ocr_server.py` (691), `pfm-web-app/src/app/manual-label/page.tsx`
|
||||
(612), `pfm-web-app/src/app/api/parse/route.ts` (604), `pfm-web-app/src/db/init.ts`
|
||||
(477), `compare_sources_accuracy.py` (451), `pfm-web-app/src/utils/docker.ts` (362),
|
||||
`pfm-web-app/public/produk-pfm/train_classifier.py` (351), `compare_accuracy.py`
|
||||
(308), `pfm-web-app/src/app/api/arena/route.ts` (265). This file itself (`AGENTS.md`)
|
||||
was at 237 lines pre-kit and would have exceeded 256 once Part B was appended —
|
||||
hence the split into `docs/vllm-service.md`.
|
||||
- **No Demo/Live or Cloud/Local switch exists yet** (§B5, §B6) for the backend
|
||||
either. `docker-compose.override.yml` exposing `db`/`pipeline-api` directly to the
|
||||
host is a local-dev convenience, not a Cloud/Local deployment switch.
|
||||
- **Naming collision resolved by this split**: `AGENTS.md` already existed in this
|
||||
directory (vLLM service doc, committed 2026-06-30, unrelated to this kit) before
|
||||
Part B was appended — unlike root, where `AGENTS.md` didn't previously exist. Don't
|
||||
assume backend's `AGENTS.md` is kit-only when reading it from another tool; Part A
|
||||
is unrelated, pre-existing content kept for a reason.
|
||||
- **Pre-existing, unrelated governance files left as-is**: `.agents/AGENTS.md` at the
|
||||
*repo root* (different path, OCR post-processing rules) and root
|
||||
`plans/next-enhancement-plan.md` (singular, `[DONE]` QA checklist) — neither is
|
||||
part of this kit; see root `AGENTS.md`'s own Adaptation Notes.
|
||||
+87
-87
@@ -1,87 +1,87 @@
|
||||
# CLAUDE.md
|
||||
|
||||
This file provides guidance to Claude Code (claude.ai/code) when working with code in this repository.
|
||||
|
||||
## What this repo is
|
||||
|
||||
`app-pfm-ocr-v2/backend` is the **next-generation rewrite of `ai-ocr-pfm-2026`** — same underlying OCR infra (PaddleOCR-VL-1.6 on vLLM + a PaddlePaddle layout-parsing pipeline), same client (Charoen Pokphand/Primafood-branded frozen food products), but a reworked Next.js app (`pfm-web-app/`) and DB schema. If you need background on the shared OCR/vLLM infra (uv conventions, issue-recording workflow, GPU tuning), see `AGENTS.md` — it's carried over near-unchanged from the previous project.
|
||||
|
||||
**The active plan for porting the Product/SKU-scanning feature lives in [`plans/next-enhancements.md`](plans/next-enhancements.md) §2** — read it before touching anything related to `scan-pfm`, `produk-pfm`, or the product classifier, since it records exactly what's done vs. still missing and the decisions already made about how to build it. (This used to be a separate `next-implementation.md`; that file was deleted 2026-07-08 once its content was folded into the plan for traceability with the rest of the `e`/`n` backlog.)
|
||||
|
||||
## How this project differs from `ai-ocr-pfm-2026`
|
||||
|
||||
- **DO-PFM UI is consolidated into a single page.** Unlike the old project's per-route pages (`do-pfm/page.tsx`, `m-do-pfm/page.tsx`), v2's entire upload/history/item-review flow lives in one `pfm-web-app/src/app/page.tsx` (client component, local state, no separate routes). `nginx.conf` still has `/do-pfm`/`/m-do-pfm` location blocks left over from the old routing — these are currently dead (no matching Next.js route, would 404).
|
||||
- **Standalone-purpose pages still get their own route folder**, e.g. `pfm-web-app/src/app/manual-label/page.tsx` — a self-contained ground-truth annotation tool (own header, own theme, no shared chrome with the root page) backed by `api/manual-label/route.ts` and `sources/manual_labels.json`. This is the pattern to follow for any new single-purpose page (see `plans/next-enhancements.md` §2 for the Product-scan pages, which follow it).
|
||||
- **Real JWT auth, enforced on the production surface**: `src/utils/auth.ts` signs/verifies tokens (`signAccountToken`/`verifyAccountToken`/`getAccountFromAuthHeader`) against an `accounts` table, each account bound to exactly one `kode_toko` (store) — the intent being that an account's own store is used on upload instead of relying on OCR-based store-text matching. **Passwords are bcrypt-hashed** (`accounts.password`, via `bcryptjs` — chosen over native `bcrypt` since the `pfm-web-app` Docker stage is `node:20-slim` with no build toolchain for native addons; `pfm-web-app/src/db/init.ts` hashes the seed and idempotently migrates any pre-existing plaintext rows on every startup). As of 2026-07-08, `api/v1/documents/*` (list, PUT-by-id, upload) **reject requests with a missing/invalid token (401)** — this is the real production surface, and the Flutter client already does a real login and attaches `Authorization: Bearer <token>` to every request (`lib/features/auth/auth_provider.dart` + `lib/core/network/api_client.dart`). The **classic routes** (`/api/upload`, `/api/scan-pfm`, `/api/parse`, `/api/history`, etc.) and the root/`scan-pfm`/`manual-label` pages deliberately do **not** check auth at all and never will unless that decision changes — they're dev-only web UI with no login screen, not part of the production surface (see `plans/next-enhancements.md` task 1.3, cancelled, and 1.4, shipped instead).
|
||||
- **Richer SKU master data**: `pfm-web-app/import_sku.js` imports from a TSV with extended packaging columns (`standar_jumlah`, `berat_kemasan`, `isi_outer_kg`, `isi_outer_pac`, `jenis_outer`) added via `ALTER TABLE sku_master ADD COLUMN IF NOT EXISTS`, superseding the old project's bare `no_sku`/`nama_item` seed list.
|
||||
- **Accuracy regression harness** (new, doesn't exist in the old project): `pfm-web-app/scripts/accuracy-check.mts` hits the live `/api/parse` endpoint for every image in `sources/test-images/`, diffs against hand-labeled ground truth in `sources/manual_labels.json` at three post-processing stages (`layer1RawRegex` → `layer2Sanitized` → `layer3Final` — trace these stage names into `utils/parser.ts` to see where each is produced), and appends run-over-run results to `sources/accuracy_history.jsonl`. Run this after touching `parser.ts` to check for regressions:
|
||||
```bash
|
||||
node pfm-web-app/scripts/accuracy-check.mts # reuse cached OCR (fast)
|
||||
node pfm-web-app/scripts/accuracy-check.mts --refresh-ocr # force fresh pipeline run
|
||||
node pfm-web-app/scripts/accuracy-check.mts --detail <filename> # full per-stage breakdown for one image
|
||||
```
|
||||
`compare_accuracy.py` / `compare_sources_accuracy.py` / `generate_excel.py` at the repo root build human-readable Excel/HTML comparison reports from the same data (`sources/comparison_report.xlsx`, `sources/comparison_side_by_side.html`) — these are analysis tooling, not part of the running app.
|
||||
- **`api/vllm-proxy/[[...path]]/route.ts`**: a passthrough proxy to the vLLM server (`paddleocr-vllm-server:8118`) that logs every call via `logVllmCallToAll` (`utils/active-log.ts`) — used for debugging/observability, not part of the OCR pipeline itself.
|
||||
- **`docker-compose.override.yml`** exposes `db` (`5432`) and `pipeline-api` (`8090`) directly to the host for local dev — not present in the old project's compose setup.
|
||||
|
||||
## Product/SKU scanning flow — status
|
||||
|
||||
**How it works end-to-end** (architecture, endpoints, classification/OCR internals, retraining): [`docs/scan-product.md`](docs/scan-product.md). See [`plans/next-enhancements.md`](plans/next-enhancements.md) §2 (task 2.1) for full detail — kept there instead of a separate doc so status stays traceable against the rest of the `e`/`n` backlog. **Feature-complete as of 2026-07-08**: the backend (`config/classify_ocr_server.py` with DINOv2 similarity search + YOLO classifier fallback, `api/scan-pfm/route.ts`, `api/produk-pfm/route.ts`, DB schema), the reference photo dataset (`pfm-web-app/public/produk-pfm/foto-kemasan-v2/`, 81 SKU subfolders as of 2026-07-14, up from the original 16 — target ~230), the desktop frontend page (`scan-pfm/page.tsx`, full feature parity), and the trained model artifacts (`models/dinov2_index.pkl` — 2,493/2,493 photos indexed as of 2026-07-14; `models/produk-pfm-classifier-26n-100e-2026-07-14.pt` — 85.8% top-1 / 94.4% top-5 val accuracy across all 81 classes, retrained 2026-07-14 in 54m21s on an RTX 2060) all now exist and load cleanly on `pipeline-api` startup. **No mobile web page is planned**: `scan-pfm/page.tsx` is desktop-only, used to test the pipeline; real mobile product scanning goes through the Flutter app instead, so `m-scan-pfm/page.tsx` and its `nginx.conf` route are intentionally left unbuilt/dead (see plan task 2.2, cancelled 2026-07-08). Not yet done: an actual browser pass uploading a photo through `/scan-pfm` end-to-end (verified via container logs/model-loading so far, not a UI test).
|
||||
|
||||
## Confidentiality
|
||||
|
||||
Same concerns as `ai-ocr-pfm-2026` apply here, plus more surface area:
|
||||
- `pfm-web-app/src/db/init.ts` and `db/migrations/005_create_sku_master.sql` contain the client's real product catalog and real vendor/customer identities, committed directly in source.
|
||||
- `sources/` holds live business data: `Rekap SKU Aktif CPI Cikande per April 2026 v2.xlsx`, `Tabel Toko Aktif Juni 2026.xlsx`, `toko_aktif.json`, `manual_labels.json`, `ai_results.json` — real SKU/store master data and hand-labeled ground truth from real scanned documents, not fixtures.
|
||||
- `uploads/` contains real scanned delivery-order photos and their OCR JSON output.
|
||||
- The `accounts` table stores bcrypt-hashed passwords as of 2026-07-08 (see above) — still don't log or export its contents, and it's not wired into most routes yet (task 1.3), so don't treat it as a secure boundary for anything beyond the `api/v1/*` REST layer.
|
||||
|
||||
## Commands
|
||||
|
||||
Web app (`pfm-web-app/`):
|
||||
```bash
|
||||
npm run dev # next dev -H 0.0.0.0 (binds all interfaces — for LAN/tunnel access during mobile testing)
|
||||
npm run build
|
||||
npm run start
|
||||
npm run lint
|
||||
```
|
||||
|
||||
Accuracy regression check (see above) — run after any `parser.ts` change:
|
||||
```bash
|
||||
node pfm-web-app/scripts/accuracy-check.mts
|
||||
```
|
||||
|
||||
`pfm-web-app/src/utils/parser.test.ts` — same standalone `node:assert` script as the old project, covering `parseDOMetadata`/`sanitizeParsedMetadata`. Run with a TS-capable runner, e.g. `npx tsx pfm-web-app/src/utils/parser.test.ts`.
|
||||
|
||||
Python services (uv-managed, same as `ai-ocr-pfm-2026` — see `AGENTS.md`):
|
||||
```bash
|
||||
./scripts/install.sh # bootstrap .venv for vLLM server
|
||||
./scripts/install-pipeline.sh # bootstrap .venv-api
|
||||
./scripts/serve.sh # vLLM genai server on :8118
|
||||
./scripts/serve-pipeline.sh # pipeline API on :8090 + classify_ocr_server.py on :8120
|
||||
```
|
||||
|
||||
Full stack:
|
||||
```bash
|
||||
docker compose up -d --build
|
||||
```
|
||||
|
||||
## Agents Settings Kit (backend-scoped)
|
||||
|
||||
@AGENTS.md
|
||||
|
||||
`AGENTS.md` in this directory now has two parts: Part A is the pre-existing vLLM
|
||||
service doc referenced above; Part B (appended 2026-07-08) is a **backend-scoped
|
||||
copy** of the [fhanyuh/agents-settings](https://github.com/fhanyuh/agents-settings)
|
||||
`e`/`enhance`/`n`/`next` workflow, independent of the root-level copy that covers
|
||||
the Flutter side (see root `CLAUDE.md`/`AGENTS.md`). Roles are in `SKILLS.md` (this
|
||||
dir). The backlog and shipped-feature log live in `plans/next-enhancements.md` and
|
||||
`docs/feature-list.md` (this dir) — these are backend-only and separate from the
|
||||
root project's equivalents, which now only track Flutter work.
|
||||
|
||||
Claude-specific notes (same as root):
|
||||
- Spawn the relevant `SKILLS.md` role via the `Agent` tool for a fresh-context
|
||||
review/QA/architecture pass instead of continuing in the implementing context.
|
||||
- Use `AskUserQuestion` for the one-at-a-time clarification step (§B2a).
|
||||
- Use `EnterPlanMode` before writing code for any `n`/`next` task that touches
|
||||
multiple files or has more than one reasonable implementation approach.
|
||||
# CLAUDE.md
|
||||
|
||||
This file provides guidance to Claude Code (claude.ai/code) when working with code in this repository.
|
||||
|
||||
## What this repo is
|
||||
|
||||
`app-pfm-ocr-v2/backend` is the **next-generation rewrite of `ai-ocr-pfm-2026`** — same underlying OCR infra (PaddleOCR-VL-1.6 on vLLM + a PaddlePaddle layout-parsing pipeline), same client (Charoen Pokphand/Primafood-branded frozen food products), but a reworked Next.js app (`pfm-web-app/`) and DB schema. If you need background on the shared OCR/vLLM infra (uv conventions, issue-recording workflow, GPU tuning), see `AGENTS.md` — it's carried over near-unchanged from the previous project.
|
||||
|
||||
**The active plan for porting the Product/SKU-scanning feature lives in [`plans/next-enhancements.md`](plans/next-enhancements.md) §2** — read it before touching anything related to `scan-pfm`, `produk-pfm`, or the product classifier, since it records exactly what's done vs. still missing and the decisions already made about how to build it. (This used to be a separate `next-implementation.md`; that file was deleted 2026-07-08 once its content was folded into the plan for traceability with the rest of the `e`/`n` backlog.)
|
||||
|
||||
## How this project differs from `ai-ocr-pfm-2026`
|
||||
|
||||
- **DO-PFM UI is consolidated into a single page.** Unlike the old project's per-route pages (`do-pfm/page.tsx`, `m-do-pfm/page.tsx`), v2's entire upload/history/item-review flow lives in one `pfm-web-app/src/app/page.tsx` (client component, local state, no separate routes). `nginx.conf` still has `/do-pfm`/`/m-do-pfm` location blocks left over from the old routing — these are currently dead (no matching Next.js route, would 404).
|
||||
- **Standalone-purpose pages still get their own route folder**, e.g. `pfm-web-app/src/app/manual-label/page.tsx` — a self-contained ground-truth annotation tool (own header, own theme, no shared chrome with the root page) backed by `api/manual-label/route.ts` and `sources/manual_labels.json`. This is the pattern to follow for any new single-purpose page (see `plans/next-enhancements.md` §2 for the Product-scan pages, which follow it).
|
||||
- **Real JWT auth, enforced on the production surface**: `src/utils/auth.ts` signs/verifies tokens (`signAccountToken`/`verifyAccountToken`/`getAccountFromAuthHeader`) against an `accounts` table, each account bound to exactly one `kode_toko` (store) — the intent being that an account's own store is used on upload instead of relying on OCR-based store-text matching. **Passwords are bcrypt-hashed** (`accounts.password`, via `bcryptjs` — chosen over native `bcrypt` since the `pfm-web-app` Docker stage is `node:20-slim` with no build toolchain for native addons; `pfm-web-app/src/db/init.ts` hashes the seed and idempotently migrates any pre-existing plaintext rows on every startup). As of 2026-07-08, `api/v1/documents/*` (list, PUT-by-id, upload) **reject requests with a missing/invalid token (401)** — this is the real production surface, and the Flutter client already does a real login and attaches `Authorization: Bearer <token>` to every request (`lib/features/auth/auth_provider.dart` + `lib/core/network/api_client.dart`). The **classic routes** (`/api/upload`, `/api/scan-pfm`, `/api/parse`, `/api/history`, etc.) and the root/`scan-pfm`/`manual-label` pages deliberately do **not** check auth at all and never will unless that decision changes — they're dev-only web UI with no login screen, not part of the production surface (see `plans/next-enhancements.md` task 1.3, cancelled, and 1.4, shipped instead).
|
||||
- **Richer SKU master data**: `pfm-web-app/import_sku.js` imports from a TSV with extended packaging columns (`standar_jumlah`, `berat_kemasan`, `isi_outer_kg`, `isi_outer_pac`, `jenis_outer`) added via `ALTER TABLE sku_master ADD COLUMN IF NOT EXISTS`, superseding the old project's bare `no_sku`/`nama_item` seed list.
|
||||
- **Accuracy regression harness** (new, doesn't exist in the old project): `pfm-web-app/scripts/accuracy-check.mts` hits the live `/api/parse` endpoint for every image in `sources/test-images/`, diffs against hand-labeled ground truth in `sources/manual_labels.json` at three post-processing stages (`layer1RawRegex` → `layer2Sanitized` → `layer3Final` — trace these stage names into `utils/parser.ts` to see where each is produced), and appends run-over-run results to `sources/accuracy_history.jsonl`. Run this after touching `parser.ts` to check for regressions:
|
||||
```bash
|
||||
node pfm-web-app/scripts/accuracy-check.mts # reuse cached OCR (fast)
|
||||
node pfm-web-app/scripts/accuracy-check.mts --refresh-ocr # force fresh pipeline run
|
||||
node pfm-web-app/scripts/accuracy-check.mts --detail <filename> # full per-stage breakdown for one image
|
||||
```
|
||||
`compare_accuracy.py` / `compare_sources_accuracy.py` / `generate_excel.py` at the repo root build human-readable Excel/HTML comparison reports from the same data (`sources/comparison_report.xlsx`, `sources/comparison_side_by_side.html`) — these are analysis tooling, not part of the running app.
|
||||
- **`api/vllm-proxy/[[...path]]/route.ts`**: a passthrough proxy to the vLLM server (`paddleocr-vllm-server:8118`) that logs every call via `logVllmCallToAll` (`utils/active-log.ts`) — used for debugging/observability, not part of the OCR pipeline itself.
|
||||
- **`docker-compose.override.yml`** exposes `db` (`5432`) and `pipeline-api` (`8090`) directly to the host for local dev — not present in the old project's compose setup.
|
||||
|
||||
## Product/SKU scanning flow — status
|
||||
|
||||
**How it works end-to-end** (architecture, endpoints, classification/OCR internals, retraining): [`docs/scan-product.md`](docs/scan-product.md). See [`plans/next-enhancements.md`](plans/next-enhancements.md) §2 (task 2.1) for full detail — kept there instead of a separate doc so status stays traceable against the rest of the `e`/`n` backlog. **Feature-complete as of 2026-07-08**: the backend (`config/classify_ocr_server.py` with DINOv2 similarity search + YOLO classifier fallback, `api/scan-pfm/route.ts`, `api/produk-pfm/route.ts`, DB schema), the reference photo dataset (`pfm-web-app/public/produk-pfm/foto-kemasan-v2/`, 81 SKU subfolders as of 2026-07-14, up from the original 16 — target ~230), the desktop frontend page (`scan-pfm/page.tsx`, full feature parity), and the trained model artifacts (`models/dinov2_index.pkl` — 2,493/2,493 photos indexed as of 2026-07-14; `models/produk-pfm-classifier-26n-100e-2026-07-14.pt` — 85.8% top-1 / 94.4% top-5 val accuracy across all 81 classes, retrained 2026-07-14 in 54m21s on an RTX 2060) all now exist and load cleanly on `pipeline-api` startup. **No mobile web page is planned**: `scan-pfm/page.tsx` is desktop-only, used to test the pipeline; real mobile product scanning goes through the Flutter app instead, so `m-scan-pfm/page.tsx` and its `nginx.conf` route are intentionally left unbuilt/dead (see plan task 2.2, cancelled 2026-07-08). Not yet done: an actual browser pass uploading a photo through `/scan-pfm` end-to-end (verified via container logs/model-loading so far, not a UI test).
|
||||
|
||||
## Confidentiality
|
||||
|
||||
Same concerns as `ai-ocr-pfm-2026` apply here, plus more surface area:
|
||||
- `pfm-web-app/src/db/init.ts` and `db/migrations/005_create_sku_master.sql` contain the client's real product catalog and real vendor/customer identities, committed directly in source.
|
||||
- `sources/` holds live business data: `Rekap SKU Aktif CPI Cikande per April 2026 v2.xlsx`, `Tabel Toko Aktif Juni 2026.xlsx`, `toko_aktif.json`, `manual_labels.json`, `ai_results.json` — real SKU/store master data and hand-labeled ground truth from real scanned documents, not fixtures.
|
||||
- `uploads/` contains real scanned delivery-order photos and their OCR JSON output.
|
||||
- The `accounts` table stores bcrypt-hashed passwords as of 2026-07-08 (see above) — still don't log or export its contents, and it's not wired into most routes yet (task 1.3), so don't treat it as a secure boundary for anything beyond the `api/v1/*` REST layer.
|
||||
|
||||
## Commands
|
||||
|
||||
Web app (`pfm-web-app/`):
|
||||
```bash
|
||||
npm run dev # next dev -H 0.0.0.0 (binds all interfaces — for LAN/tunnel access during mobile testing)
|
||||
npm run build
|
||||
npm run start
|
||||
npm run lint
|
||||
```
|
||||
|
||||
Accuracy regression check (see above) — run after any `parser.ts` change:
|
||||
```bash
|
||||
node pfm-web-app/scripts/accuracy-check.mts
|
||||
```
|
||||
|
||||
`pfm-web-app/src/utils/parser.test.ts` — same standalone `node:assert` script as the old project, covering `parseDOMetadata`/`sanitizeParsedMetadata`. Run with a TS-capable runner, e.g. `npx tsx pfm-web-app/src/utils/parser.test.ts`.
|
||||
|
||||
Python services (uv-managed, same as `ai-ocr-pfm-2026` — see `AGENTS.md`):
|
||||
```bash
|
||||
./scripts/install.sh # bootstrap .venv for vLLM server
|
||||
./scripts/install-pipeline.sh # bootstrap .venv-api
|
||||
./scripts/serve.sh # vLLM genai server on :8118
|
||||
./scripts/serve-pipeline.sh # pipeline API on :8090 + classify_ocr_server.py on :8120
|
||||
```
|
||||
|
||||
Full stack:
|
||||
```bash
|
||||
docker compose up -d --build
|
||||
```
|
||||
|
||||
## Agents Settings Kit (backend-scoped)
|
||||
|
||||
@AGENTS.md
|
||||
|
||||
`AGENTS.md` in this directory now has two parts: Part A is the pre-existing vLLM
|
||||
service doc referenced above; Part B (appended 2026-07-08) is a **backend-scoped
|
||||
copy** of the [fhanyuh/agents-settings](https://github.com/fhanyuh/agents-settings)
|
||||
`e`/`enhance`/`n`/`next` workflow, independent of the root-level copy that covers
|
||||
the Flutter side (see root `CLAUDE.md`/`AGENTS.md`). Roles are in `SKILLS.md` (this
|
||||
dir). The backlog and shipped-feature log live in `plans/next-enhancements.md` and
|
||||
`docs/feature-list.md` (this dir) — these are backend-only and separate from the
|
||||
root project's equivalents, which now only track Flutter work.
|
||||
|
||||
Claude-specific notes (same as root):
|
||||
- Spawn the relevant `SKILLS.md` role via the `Agent` tool for a fresh-context
|
||||
review/QA/architecture pass instead of continuing in the implementing context.
|
||||
- Use `AskUserQuestion` for the one-at-a-time clarification step (§B2a).
|
||||
- Use `EnterPlanMode` before writing code for any `n`/`next` task that touches
|
||||
multiple files or has more than one reasonable implementation approach.
|
||||
+83
-83
@@ -1,83 +1,83 @@
|
||||
# Stage 0: GPU Base image
|
||||
FROM nvidia/cuda:12.6.0-devel-ubuntu22.04 AS base-gpu
|
||||
|
||||
ENV DEBIAN_FRONTEND=noninteractive
|
||||
ENV PATH="/root/.local/bin:$PATH"
|
||||
|
||||
# Install system dependencies (libgl and libglib are required for OpenCV)
|
||||
RUN apt-get update && apt-get install -y --no-install-recommends \
|
||||
curl \
|
||||
git \
|
||||
libgl1 \
|
||||
libglib2.0-0 \
|
||||
&& rm -rf /var/lib/apt/lists/*
|
||||
|
||||
# Install uv
|
||||
RUN curl -LsSf https://astral.sh/uv/install.sh | sh
|
||||
|
||||
# --- vLLM Server Stage ---
|
||||
FROM base-gpu AS vllm-server
|
||||
WORKDIR /app
|
||||
|
||||
# Install project dependencies
|
||||
COPY pyproject.toml uv.lock ./
|
||||
RUN uv python pin 3.12 && uv sync --frozen --no-dev
|
||||
|
||||
# Install prebuilt flash-attention wheel
|
||||
ARG FLASH_ATTN_WHEEL=https://github.com/mjun0812/flash-attention-prebuild-wheels/releases/download/v0.3.14/flash_attn-2.8.2+cu128torch2.8-cp312-cp312-linux_x86_64.whl
|
||||
RUN uv pip install --python .venv "${FLASH_ATTN_WHEEL}"
|
||||
|
||||
COPY . /app
|
||||
RUN chmod +x /app/scripts/serve.sh
|
||||
|
||||
EXPOSE 8118
|
||||
CMD ["./scripts/serve.sh"]
|
||||
|
||||
# --- Pipeline API Stage ---
|
||||
FROM base-gpu AS pipeline-api
|
||||
WORKDIR /app
|
||||
|
||||
# Build paddlepaddle and paddlex virtual env
|
||||
RUN uv venv .venv-api --python 3.12
|
||||
RUN uv pip install --python .venv-api paddlepaddle-gpu -i https://www.paddlepaddle.org.cn/packages/stable/cu126/
|
||||
RUN uv pip install --python .venv-api "paddleocr[doc-parser]>=3.3.0"
|
||||
RUN uv pip install --python .venv-api "aiohttp>=3.9" "filetype>=1.2" "fastapi>=0.110" "starlette>=0.36" "uvicorn>=0.16" "ultralytics>=8.0"
|
||||
|
||||
COPY . /app
|
||||
RUN chmod +x /app/scripts/serve-pipeline.sh
|
||||
|
||||
EXPOSE 8090
|
||||
CMD ["./scripts/serve-pipeline.sh"]
|
||||
|
||||
# --- Gradio UI Stage ---
|
||||
FROM python:3.12-slim AS gradio-ui
|
||||
WORKDIR /app/PaddleOCR-VL-1.6_Online_Demo
|
||||
|
||||
RUN apt-get update && apt-get install -y --no-install-recommends \
|
||||
curl \
|
||||
&& rm -rf /var/lib/apt/lists/*
|
||||
|
||||
COPY PaddleOCR-VL-1.6_Online_Demo/requirements.txt ./
|
||||
RUN pip install --no-cache-dir -r requirements.txt
|
||||
|
||||
COPY PaddleOCR-VL-1.6_Online_Demo ./
|
||||
|
||||
EXPOSE 7870
|
||||
|
||||
ENV GRADIO_SERVER_NAME="0.0.0.0"
|
||||
ENV GRADIO_SERVER_PORT="7870"
|
||||
|
||||
CMD ["python", "app.py"]
|
||||
|
||||
# --- Next.js Web App Stage ---
|
||||
FROM node:20-slim AS pfm-web-app
|
||||
WORKDIR /app
|
||||
COPY pfm-web-app/package.json pfm-web-app/package-lock.json ./
|
||||
ENV PUPPETEER_SKIP_DOWNLOAD=true
|
||||
RUN npm ci
|
||||
COPY pfm-web-app/ ./
|
||||
ENV NODE_ENV=production
|
||||
RUN npm run build
|
||||
EXPOSE 3000
|
||||
CMD ["npm", "start"]
|
||||
|
||||
# Stage 0: GPU Base image
|
||||
FROM nvidia/cuda:12.6.0-devel-ubuntu22.04 AS base-gpu
|
||||
|
||||
ENV DEBIAN_FRONTEND=noninteractive
|
||||
ENV PATH="/root/.local/bin:$PATH"
|
||||
|
||||
# Install system dependencies (libgl and libglib are required for OpenCV)
|
||||
RUN apt-get update && apt-get install -y --no-install-recommends \
|
||||
curl \
|
||||
git \
|
||||
libgl1 \
|
||||
libglib2.0-0 \
|
||||
&& rm -rf /var/lib/apt/lists/*
|
||||
|
||||
# Install uv
|
||||
RUN curl -LsSf https://astral.sh/uv/install.sh | sh
|
||||
|
||||
# --- vLLM Server Stage ---
|
||||
FROM base-gpu AS vllm-server
|
||||
WORKDIR /app
|
||||
|
||||
# Install project dependencies
|
||||
COPY pyproject.toml uv.lock ./
|
||||
RUN uv python pin 3.12 && uv sync --frozen --no-dev
|
||||
|
||||
# Install prebuilt flash-attention wheel
|
||||
ARG FLASH_ATTN_WHEEL=https://github.com/mjun0812/flash-attention-prebuild-wheels/releases/download/v0.3.14/flash_attn-2.8.2+cu128torch2.8-cp312-cp312-linux_x86_64.whl
|
||||
RUN uv pip install --python .venv "${FLASH_ATTN_WHEEL}"
|
||||
|
||||
COPY . /app
|
||||
RUN chmod +x /app/scripts/serve.sh
|
||||
|
||||
EXPOSE 8118
|
||||
CMD ["./scripts/serve.sh"]
|
||||
|
||||
# --- Pipeline API Stage ---
|
||||
FROM base-gpu AS pipeline-api
|
||||
WORKDIR /app
|
||||
|
||||
# Build paddlepaddle and paddlex virtual env
|
||||
RUN uv venv .venv-api --python 3.12
|
||||
RUN uv pip install --python .venv-api paddlepaddle-gpu -i https://www.paddlepaddle.org.cn/packages/stable/cu126/
|
||||
RUN uv pip install --python .venv-api "paddleocr[doc-parser]>=3.3.0"
|
||||
RUN uv pip install --python .venv-api "aiohttp>=3.9" "filetype>=1.2" "fastapi>=0.110" "starlette>=0.36" "uvicorn>=0.16" "ultralytics>=8.0"
|
||||
|
||||
COPY . /app
|
||||
RUN chmod +x /app/scripts/serve-pipeline.sh
|
||||
|
||||
EXPOSE 8090
|
||||
CMD ["./scripts/serve-pipeline.sh"]
|
||||
|
||||
# --- Gradio UI Stage ---
|
||||
FROM python:3.12-slim AS gradio-ui
|
||||
WORKDIR /app/PaddleOCR-VL-1.6_Online_Demo
|
||||
|
||||
RUN apt-get update && apt-get install -y --no-install-recommends \
|
||||
curl \
|
||||
&& rm -rf /var/lib/apt/lists/*
|
||||
|
||||
COPY PaddleOCR-VL-1.6_Online_Demo/requirements.txt ./
|
||||
RUN pip install --no-cache-dir -r requirements.txt
|
||||
|
||||
COPY PaddleOCR-VL-1.6_Online_Demo ./
|
||||
|
||||
EXPOSE 7870
|
||||
|
||||
ENV GRADIO_SERVER_NAME="0.0.0.0"
|
||||
ENV GRADIO_SERVER_PORT="7870"
|
||||
|
||||
CMD ["python", "app.py"]
|
||||
|
||||
# --- Next.js Web App Stage ---
|
||||
FROM node:20-slim AS pfm-web-app
|
||||
WORKDIR /app
|
||||
COPY pfm-web-app/package.json pfm-web-app/package-lock.json ./
|
||||
ENV PUPPETEER_SKIP_DOWNLOAD=true
|
||||
RUN npm ci
|
||||
COPY pfm-web-app/ ./
|
||||
ENV NODE_ENV=production
|
||||
RUN npm run build
|
||||
EXPOSE 3000
|
||||
CMD ["npm", "start"]
|
||||
|
||||
+151
-151
@@ -1,151 +1,151 @@
|
||||
# PaddleOCR-VL-1.6 on vLLM
|
||||
|
||||
Local deployment of [PaddleOCR-VL-1.6](https://www.paddleocr.ai/latest/en/version3.x/pipeline_usage/PaddleOCR-VL.html) using **vLLM** as the VLM inference backend. All Python workflows use **[uv](https://docs.astral.sh/uv/)**.
|
||||
|
||||
## Architecture
|
||||
|
||||
```
|
||||
Gradio demo (7870)
|
||||
│
|
||||
▼
|
||||
Pipeline API (8090) ── layout + preprocessing (PaddlePaddle GPU)
|
||||
│
|
||||
▼
|
||||
vLLM genai server (8118) ── PaddleOCR-VL-1.6 VLM
|
||||
```
|
||||
|
||||
| Service | Script | Default URL |
|
||||
|---------|--------|-------------|
|
||||
| vLLM VLM server | `./scripts/serve.sh` | `http://127.0.0.1:8118/v1` |
|
||||
| Full pipeline API | `./scripts/serve-pipeline.sh` | `http://127.0.0.1:8090/layout-parsing` |
|
||||
| Online demo UI | `./scripts/run-demo.sh` | `http://127.0.0.1:7870` |
|
||||
|
||||
The vLLM server exposes only the VLM stage. For HTTP document parsing (layout + OCR), run the pipeline API, which calls vLLM via `config/pipeline_config_vllm.yaml`.
|
||||
|
||||
## Prerequisites
|
||||
|
||||
- Linux with NVIDIA GPU (CC ≥ 8.0 recommended; CUDA 12.6+ driver)
|
||||
- [uv](https://docs.astral.sh/uv/) installed
|
||||
- ~16 GB GPU VRAM for default vLLM settings (tune in `config/vllm_config.yaml`)
|
||||
|
||||
## Quick start
|
||||
|
||||
```bash
|
||||
git clone <repo-url> ai-ocr-pfm-2026
|
||||
cd ai-ocr-pfm-2026
|
||||
|
||||
cp .env.example .env # adjust CUDA_VISIBLE_DEVICES if needed
|
||||
|
||||
# 1) Install vLLM server (.venv)
|
||||
FLASH_ATTN_WHEEL="https://github.com/mjun0812/flash-attention-prebuild-wheels/releases/download/v0.3.14/flash_attn-2.8.2+cu128torch2.8-cp312-cp312-linux_x86_64.whl" \
|
||||
./scripts/install.sh
|
||||
|
||||
# 2) Install pipeline API (.venv-api) — optional, needed for demo / full HTTP API
|
||||
./scripts/install-pipeline.sh
|
||||
```
|
||||
|
||||
Start services (three terminals, or background each):
|
||||
|
||||
```bash
|
||||
./scripts/serve.sh # vLLM on :8118
|
||||
./scripts/serve-pipeline.sh # pipeline on :8090
|
||||
./scripts/run-demo.sh # Gradio on :7870
|
||||
```
|
||||
|
||||
Health checks:
|
||||
|
||||
```bash
|
||||
curl -s http://127.0.0.1:8118/v1/models | jq .
|
||||
curl -s http://127.0.0.1:8090/health
|
||||
curl -s -o /dev/null -w "%{http_code}\n" http://127.0.0.1:7870/
|
||||
```
|
||||
|
||||
## Configuration
|
||||
|
||||
Copy `.env.example` to `.env`:
|
||||
|
||||
| Variable | Default | Description |
|
||||
|----------|---------|-------------|
|
||||
| `GENAI_HOST` | `0.0.0.0` | vLLM bind address |
|
||||
| `GENAI_PORT` | `8118` | vLLM port |
|
||||
| `GENAI_MODEL` | `PaddleOCR-VL-1.6-0.9B` | Model name |
|
||||
| `VLLM_CONFIG` | `config/vllm_config.yaml` | vLLM tuning |
|
||||
| `CUDA_VISIBLE_DEVICES` | `1` | GPU for vLLM (use least-busy GPU) |
|
||||
| `PIPELINE_PORT` | `8090` | Pipeline API port |
|
||||
| `PIPELINE_DEVICE` | `gpu:0` | GPU for layout/preprocessing |
|
||||
| `GRADIO_PORT` | `7870` | Demo UI port |
|
||||
|
||||
vLLM tuning (`config/vllm_config.yaml`):
|
||||
|
||||
```yaml
|
||||
gpu-memory-utilization: 0.75
|
||||
max-num-seqs: 128
|
||||
```
|
||||
|
||||
## Client usage
|
||||
|
||||
### Python (vLLM only)
|
||||
|
||||
```python
|
||||
from paddleocr import PaddleOCRVL
|
||||
|
||||
pipeline = PaddleOCRVL(
|
||||
vl_rec_backend="vllm-server",
|
||||
vl_rec_server_url="http://127.0.0.1:8118/v1",
|
||||
)
|
||||
output = pipeline.predict("path/to/image.png")
|
||||
```
|
||||
|
||||
Run the client in a **separate** environment if it needs PaddlePaddle GPU alongside Transformers.
|
||||
|
||||
### CLI
|
||||
|
||||
```bash
|
||||
uv run paddleocr doc_parser \
|
||||
--input demo.png \
|
||||
--vl_rec_backend vllm-server \
|
||||
--vl_rec_server_url http://127.0.0.1:8118/v1
|
||||
```
|
||||
|
||||
### HTTP (full pipeline)
|
||||
|
||||
```bash
|
||||
curl -X POST http://127.0.0.1:8090/layout-parsing \
|
||||
-H "Content-Type: application/json" \
|
||||
-d '{"file":"<base64>", "fileType": 1, "useLayoutDetection": true}'
|
||||
```
|
||||
|
||||
## Project layout
|
||||
|
||||
```
|
||||
config/
|
||||
vllm_config.yaml # vLLM backend tuning
|
||||
pipeline_config_vllm.yaml # pipeline → vLLM server URL
|
||||
scripts/
|
||||
install.sh # bootstrap .venv (vLLM)
|
||||
install-pipeline.sh # bootstrap .venv-api (pipeline)
|
||||
serve.sh # start vLLM genai server
|
||||
serve-pipeline.sh # start pipeline API
|
||||
run-demo.sh # start Gradio demo
|
||||
PaddleOCR-VL-1.6_Online_Demo/ # bundled Hugging Face-style demo
|
||||
issues/ # recorded problems and fixes
|
||||
AGENTS.md # agent / contributor guide
|
||||
```
|
||||
|
||||
## Troubleshooting
|
||||
|
||||
See [issues/](issues/) for detailed write-ups. Common fixes:
|
||||
|
||||
| Symptom | Fix |
|
||||
|---------|-----|
|
||||
| GPU OOM on vLLM startup | Lower `gpu-memory-utilization` or set `CUDA_VISIBLE_DEVICES` to a free GPU |
|
||||
| flash-attn build failure | Use prebuilt wheel via `FLASH_ATTN_WHEEL=... ./scripts/install.sh` |
|
||||
| Port 8080 in use | Pipeline defaults to **8090**; demo defaults to **7870** |
|
||||
|
||||
Agent conventions and issue-recording rules: [AGENTS.md](AGENTS.md).
|
||||
|
||||
## References
|
||||
|
||||
- [PaddleOCR-VL usage tutorial](https://www.paddleocr.ai/latest/en/version3.x/pipeline_usage/PaddleOCR-VL.html)
|
||||
- [PaddleOCR genai_server FAQ](https://github.com/PaddlePaddle/PaddleOCR/discussions/16822)
|
||||
- [flash-attention prebuild wheels](https://mjunya.com/flash-attention-prebuild-wheels/)
|
||||
# PaddleOCR-VL-1.6 on vLLM
|
||||
|
||||
Local deployment of [PaddleOCR-VL-1.6](https://www.paddleocr.ai/latest/en/version3.x/pipeline_usage/PaddleOCR-VL.html) using **vLLM** as the VLM inference backend. All Python workflows use **[uv](https://docs.astral.sh/uv/)**.
|
||||
|
||||
## Architecture
|
||||
|
||||
```
|
||||
Gradio demo (7870)
|
||||
│
|
||||
▼
|
||||
Pipeline API (8090) ── layout + preprocessing (PaddlePaddle GPU)
|
||||
│
|
||||
▼
|
||||
vLLM genai server (8118) ── PaddleOCR-VL-1.6 VLM
|
||||
```
|
||||
|
||||
| Service | Script | Default URL |
|
||||
|---------|--------|-------------|
|
||||
| vLLM VLM server | `./scripts/serve.sh` | `http://127.0.0.1:8118/v1` |
|
||||
| Full pipeline API | `./scripts/serve-pipeline.sh` | `http://127.0.0.1:8090/layout-parsing` |
|
||||
| Online demo UI | `./scripts/run-demo.sh` | `http://127.0.0.1:7870` |
|
||||
|
||||
The vLLM server exposes only the VLM stage. For HTTP document parsing (layout + OCR), run the pipeline API, which calls vLLM via `config/pipeline_config_vllm.yaml`.
|
||||
|
||||
## Prerequisites
|
||||
|
||||
- Linux with NVIDIA GPU (CC ≥ 8.0 recommended; CUDA 12.6+ driver)
|
||||
- [uv](https://docs.astral.sh/uv/) installed
|
||||
- ~16 GB GPU VRAM for default vLLM settings (tune in `config/vllm_config.yaml`)
|
||||
|
||||
## Quick start
|
||||
|
||||
```bash
|
||||
git clone <repo-url> ai-ocr-pfm-2026
|
||||
cd ai-ocr-pfm-2026
|
||||
|
||||
cp .env.example .env # adjust CUDA_VISIBLE_DEVICES if needed
|
||||
|
||||
# 1) Install vLLM server (.venv)
|
||||
FLASH_ATTN_WHEEL="https://github.com/mjun0812/flash-attention-prebuild-wheels/releases/download/v0.3.14/flash_attn-2.8.2+cu128torch2.8-cp312-cp312-linux_x86_64.whl" \
|
||||
./scripts/install.sh
|
||||
|
||||
# 2) Install pipeline API (.venv-api) — optional, needed for demo / full HTTP API
|
||||
./scripts/install-pipeline.sh
|
||||
```
|
||||
|
||||
Start services (three terminals, or background each):
|
||||
|
||||
```bash
|
||||
./scripts/serve.sh # vLLM on :8118
|
||||
./scripts/serve-pipeline.sh # pipeline on :8090
|
||||
./scripts/run-demo.sh # Gradio on :7870
|
||||
```
|
||||
|
||||
Health checks:
|
||||
|
||||
```bash
|
||||
curl -s http://127.0.0.1:8118/v1/models | jq .
|
||||
curl -s http://127.0.0.1:8090/health
|
||||
curl -s -o /dev/null -w "%{http_code}\n" http://127.0.0.1:7870/
|
||||
```
|
||||
|
||||
## Configuration
|
||||
|
||||
Copy `.env.example` to `.env`:
|
||||
|
||||
| Variable | Default | Description |
|
||||
|----------|---------|-------------|
|
||||
| `GENAI_HOST` | `0.0.0.0` | vLLM bind address |
|
||||
| `GENAI_PORT` | `8118` | vLLM port |
|
||||
| `GENAI_MODEL` | `PaddleOCR-VL-1.6-0.9B` | Model name |
|
||||
| `VLLM_CONFIG` | `config/vllm_config.yaml` | vLLM tuning |
|
||||
| `CUDA_VISIBLE_DEVICES` | `1` | GPU for vLLM (use least-busy GPU) |
|
||||
| `PIPELINE_PORT` | `8090` | Pipeline API port |
|
||||
| `PIPELINE_DEVICE` | `gpu:0` | GPU for layout/preprocessing |
|
||||
| `GRADIO_PORT` | `7870` | Demo UI port |
|
||||
|
||||
vLLM tuning (`config/vllm_config.yaml`):
|
||||
|
||||
```yaml
|
||||
gpu-memory-utilization: 0.75
|
||||
max-num-seqs: 128
|
||||
```
|
||||
|
||||
## Client usage
|
||||
|
||||
### Python (vLLM only)
|
||||
|
||||
```python
|
||||
from paddleocr import PaddleOCRVL
|
||||
|
||||
pipeline = PaddleOCRVL(
|
||||
vl_rec_backend="vllm-server",
|
||||
vl_rec_server_url="http://127.0.0.1:8118/v1",
|
||||
)
|
||||
output = pipeline.predict("path/to/image.png")
|
||||
```
|
||||
|
||||
Run the client in a **separate** environment if it needs PaddlePaddle GPU alongside Transformers.
|
||||
|
||||
### CLI
|
||||
|
||||
```bash
|
||||
uv run paddleocr doc_parser \
|
||||
--input demo.png \
|
||||
--vl_rec_backend vllm-server \
|
||||
--vl_rec_server_url http://127.0.0.1:8118/v1
|
||||
```
|
||||
|
||||
### HTTP (full pipeline)
|
||||
|
||||
```bash
|
||||
curl -X POST http://127.0.0.1:8090/layout-parsing \
|
||||
-H "Content-Type: application/json" \
|
||||
-d '{"file":"<base64>", "fileType": 1, "useLayoutDetection": true}'
|
||||
```
|
||||
|
||||
## Project layout
|
||||
|
||||
```
|
||||
config/
|
||||
vllm_config.yaml # vLLM backend tuning
|
||||
pipeline_config_vllm.yaml # pipeline → vLLM server URL
|
||||
scripts/
|
||||
install.sh # bootstrap .venv (vLLM)
|
||||
install-pipeline.sh # bootstrap .venv-api (pipeline)
|
||||
serve.sh # start vLLM genai server
|
||||
serve-pipeline.sh # start pipeline API
|
||||
run-demo.sh # start Gradio demo
|
||||
PaddleOCR-VL-1.6_Online_Demo/ # bundled Hugging Face-style demo
|
||||
issues/ # recorded problems and fixes
|
||||
AGENTS.md # agent / contributor guide
|
||||
```
|
||||
|
||||
## Troubleshooting
|
||||
|
||||
See [issues/](issues/) for detailed write-ups. Common fixes:
|
||||
|
||||
| Symptom | Fix |
|
||||
|---------|-----|
|
||||
| GPU OOM on vLLM startup | Lower `gpu-memory-utilization` or set `CUDA_VISIBLE_DEVICES` to a free GPU |
|
||||
| flash-attn build failure | Use prebuilt wheel via `FLASH_ATTN_WHEEL=... ./scripts/install.sh` |
|
||||
| Port 8080 in use | Pipeline defaults to **8090**; demo defaults to **7870** |
|
||||
|
||||
Agent conventions and issue-recording rules: [AGENTS.md](AGENTS.md).
|
||||
|
||||
## References
|
||||
|
||||
- [PaddleOCR-VL usage tutorial](https://www.paddleocr.ai/latest/en/version3.x/pipeline_usage/PaddleOCR-VL.html)
|
||||
- [PaddleOCR genai_server FAQ](https://github.com/PaddlePaddle/PaddleOCR/discussions/16822)
|
||||
- [flash-attention prebuild wheels](https://mjunya.com/flash-attention-prebuild-wheels/)
|
||||
+102
-102
@@ -1,102 +1,102 @@
|
||||
# Skills & Roles (backend)
|
||||
|
||||
Backend-scoped copy of the root `SKILLS.md` — same five roles, applied to
|
||||
`backend/` surfaces (Next.js API gateway, OCR pipeline, Postgres, Docker/deploy)
|
||||
during `n`/`next` execution (see `AGENTS.md` Part B, this dir). One agent can play
|
||||
all of them in sequence; a multi-agent harness may spawn each as a separate
|
||||
subagent for a fresh-context pass. Order matters: Architect → Backend/Frontend →
|
||||
QA → Hardware/Compatibility.
|
||||
|
||||
## 1. Software Architect
|
||||
|
||||
**Responsibilities**
|
||||
- Decide where new backend code lives; keep module boundaries clean (API routes vs.
|
||||
`utils/` business logic vs. `db/` layer vs. the Python pipeline in `config/`).
|
||||
- Prefer deep modules (few, well-bounded files with simple interfaces) over shallow
|
||||
ones — this is what keeps the codebase navigable for an agent.
|
||||
- Own the 256-LOC split rule (`AGENTS.md` Part B §B3): when a file crosses the
|
||||
threshold, decide the split boundary before anyone patches around it.
|
||||
- Keep `plans/next-enhancements.md` (this dir) structured by real backend module
|
||||
boundaries, not arbitrary groupings.
|
||||
- Owns the existing-project audit (`AGENTS.md` Part B §B0) for backend specifically.
|
||||
|
||||
**When invoked**: start of every `e`/`enhance` run; start of every `n`/`next` task,
|
||||
before implementation begins.
|
||||
|
||||
**Handoff**: hands the Backend/Frontend roles a target file layout and interface
|
||||
contract, not just a task description.
|
||||
|
||||
## 2. Backend Engineer
|
||||
|
||||
**Responsibilities**
|
||||
- Implement Next.js API route logic, DB access (`src/db/`), and the Python OCR
|
||||
pipeline (`config/classify_ocr_server.py`, pipeline API) as the task requires.
|
||||
- Wire the mock-vs-live routing required by the Demo/Live switch (`AGENTS.md` Part B
|
||||
§B5) and the Cloud/Local endpoint switch (§B6) if/when built — both must resolve
|
||||
through the same contract so swapping either setting never changes calling code.
|
||||
- Keep business logic out of route handlers (`src/app/api/**/route.ts`); route
|
||||
handlers stay thin, matching the existing `utils/parser.ts`-style separation.
|
||||
- Use `withTransaction` (`src/db/index.ts`) for any multi-statement write that must
|
||||
be atomic — see task 7.1 in the pre-kit history for why this matters here.
|
||||
|
||||
**When invoked**: any task touching API routes, the DB layer, or the OCR pipeline.
|
||||
|
||||
**Handoff**: gives Frontend a stable contract (types/response shape) to build
|
||||
against; gives QA the list of new/changed endpoints and their expected error modes.
|
||||
|
||||
## 3. Frontend Engineer
|
||||
|
||||
**Responsibilities**
|
||||
- Implement UI for the task inside `pfm-web-app/src/app/`, including Demo/Live and
|
||||
Cloud/Local switcher controls where relevant.
|
||||
- Follow this repo's existing page pattern: consolidated single-page flows (root
|
||||
`page.tsx`) vs. standalone-purpose route folders (`manual-label/page.tsx`,
|
||||
`scan-pfm/page.tsx`) — see backend `CLAUDE.md` for which pattern a given feature
|
||||
should follow.
|
||||
- Consume the Backend Engineer's contract rather than reaching around it.
|
||||
- Keep components small and composable, respecting the 256-LOC rule.
|
||||
|
||||
**When invoked**: any task with a user-facing surface inside `pfm-web-app/`.
|
||||
|
||||
**Handoff**: gives QA the golden-path user flow and the edge cases it's aware of.
|
||||
|
||||
## 4. QA / Test Engineer
|
||||
|
||||
**Responsibilities**
|
||||
- During clarification (`AGENTS.md` Part B §B2a), turn resolved answers into
|
||||
concrete acceptance criteria — what "done" verifiably means.
|
||||
- Write/extend automated tests (`parser.test.ts` pattern) for the change.
|
||||
- For anything touching `parser.ts` or the OCR pipeline, run the accuracy
|
||||
regression harness (`node pfm-web-app/scripts/accuracy-check.mts` or `accuracy-check-scan.mts`) and check for
|
||||
regressions against the current baseline (see `sources/accuracy_history.jsonl` and `sources/product_accuracy_history.jsonl` for latest metrics), not just "it compiles."
|
||||
- Run the **verify build integrity** pass: golden path + edge cases + regression
|
||||
check on adjacent features.
|
||||
- Reject work back to the relevant role if acceptance criteria aren't met — don't
|
||||
patch around a failing check.
|
||||
|
||||
**When invoked**: acceptance-criteria drafting during §B2a; final verification pass
|
||||
before a task is marked `[DONE]`.
|
||||
|
||||
**Handoff**: reports pass/fail with specifics (what broke, under what input) back to
|
||||
whichever role owns that surface.
|
||||
|
||||
## 5. Hardware & Performance Compatibility Reviewer
|
||||
|
||||
**Responsibilities**
|
||||
- Check the change against this stack's real constraints: single vs. dual-GPU dev
|
||||
mode (`docker-compose.yml` runs `npm run dev`, a known throughput ceiling), VRAM
|
||||
budget for vLLM (`gpu-memory-utilization` in `config/vllm_config.yaml`), and
|
||||
behavior under the Local/on-prem deployment mode from `AGENTS.md` Part B §B6.
|
||||
- Flag newly introduced heavy Python/Node dependencies, GPU-specific assumptions, or
|
||||
anything that would break the isolated vLLM-server-only environment (no
|
||||
`paddlepaddle-gpu` in this venv — see `AGENTS.md` Part A / `docs/vllm-service.md`).
|
||||
- Flag anything that would degrade badly on lower-spec hardware or slower networks
|
||||
(e.g. the mobile app's 2s polling loop against a slow backend response), and
|
||||
suggest a lighter-weight alternative when one exists.
|
||||
|
||||
**When invoked**: final verification pass, alongside QA, before a task is marked
|
||||
`[DONE]`; also whenever a task adds a new dependency or changes the deployment/
|
||||
runtime surface.
|
||||
|
||||
**Handoff**: blocks `[DONE]` status until concerns are resolved or explicitly
|
||||
accepted as a documented trade-off in `docs/feature-list.md` (this dir).
|
||||
# Skills & Roles (backend)
|
||||
|
||||
Backend-scoped copy of the root `SKILLS.md` — same five roles, applied to
|
||||
`backend/` surfaces (Next.js API gateway, OCR pipeline, Postgres, Docker/deploy)
|
||||
during `n`/`next` execution (see `AGENTS.md` Part B, this dir). One agent can play
|
||||
all of them in sequence; a multi-agent harness may spawn each as a separate
|
||||
subagent for a fresh-context pass. Order matters: Architect → Backend/Frontend →
|
||||
QA → Hardware/Compatibility.
|
||||
|
||||
## 1. Software Architect
|
||||
|
||||
**Responsibilities**
|
||||
- Decide where new backend code lives; keep module boundaries clean (API routes vs.
|
||||
`utils/` business logic vs. `db/` layer vs. the Python pipeline in `config/`).
|
||||
- Prefer deep modules (few, well-bounded files with simple interfaces) over shallow
|
||||
ones — this is what keeps the codebase navigable for an agent.
|
||||
- Own the 256-LOC split rule (`AGENTS.md` Part B §B3): when a file crosses the
|
||||
threshold, decide the split boundary before anyone patches around it.
|
||||
- Keep `plans/next-enhancements.md` (this dir) structured by real backend module
|
||||
boundaries, not arbitrary groupings.
|
||||
- Owns the existing-project audit (`AGENTS.md` Part B §B0) for backend specifically.
|
||||
|
||||
**When invoked**: start of every `e`/`enhance` run; start of every `n`/`next` task,
|
||||
before implementation begins.
|
||||
|
||||
**Handoff**: hands the Backend/Frontend roles a target file layout and interface
|
||||
contract, not just a task description.
|
||||
|
||||
## 2. Backend Engineer
|
||||
|
||||
**Responsibilities**
|
||||
- Implement Next.js API route logic, DB access (`src/db/`), and the Python OCR
|
||||
pipeline (`config/classify_ocr_server.py`, pipeline API) as the task requires.
|
||||
- Wire the mock-vs-live routing required by the Demo/Live switch (`AGENTS.md` Part B
|
||||
§B5) and the Cloud/Local endpoint switch (§B6) if/when built — both must resolve
|
||||
through the same contract so swapping either setting never changes calling code.
|
||||
- Keep business logic out of route handlers (`src/app/api/**/route.ts`); route
|
||||
handlers stay thin, matching the existing `utils/parser.ts`-style separation.
|
||||
- Use `withTransaction` (`src/db/index.ts`) for any multi-statement write that must
|
||||
be atomic — see task 7.1 in the pre-kit history for why this matters here.
|
||||
|
||||
**When invoked**: any task touching API routes, the DB layer, or the OCR pipeline.
|
||||
|
||||
**Handoff**: gives Frontend a stable contract (types/response shape) to build
|
||||
against; gives QA the list of new/changed endpoints and their expected error modes.
|
||||
|
||||
## 3. Frontend Engineer
|
||||
|
||||
**Responsibilities**
|
||||
- Implement UI for the task inside `pfm-web-app/src/app/`, including Demo/Live and
|
||||
Cloud/Local switcher controls where relevant.
|
||||
- Follow this repo's existing page pattern: consolidated single-page flows (root
|
||||
`page.tsx`) vs. standalone-purpose route folders (`manual-label/page.tsx`,
|
||||
`scan-pfm/page.tsx`) — see backend `CLAUDE.md` for which pattern a given feature
|
||||
should follow.
|
||||
- Consume the Backend Engineer's contract rather than reaching around it.
|
||||
- Keep components small and composable, respecting the 256-LOC rule.
|
||||
|
||||
**When invoked**: any task with a user-facing surface inside `pfm-web-app/`.
|
||||
|
||||
**Handoff**: gives QA the golden-path user flow and the edge cases it's aware of.
|
||||
|
||||
## 4. QA / Test Engineer
|
||||
|
||||
**Responsibilities**
|
||||
- During clarification (`AGENTS.md` Part B §B2a), turn resolved answers into
|
||||
concrete acceptance criteria — what "done" verifiably means.
|
||||
- Write/extend automated tests (`parser.test.ts` pattern) for the change.
|
||||
- For anything touching `parser.ts` or the OCR pipeline, run the accuracy
|
||||
regression harness (`node pfm-web-app/scripts/accuracy-check.mts` or `accuracy-check-scan.mts`) and check for
|
||||
regressions against the current baseline (see `sources/accuracy_history.jsonl` and `sources/product_accuracy_history.jsonl` for latest metrics), not just "it compiles."
|
||||
- Run the **verify build integrity** pass: golden path + edge cases + regression
|
||||
check on adjacent features.
|
||||
- Reject work back to the relevant role if acceptance criteria aren't met — don't
|
||||
patch around a failing check.
|
||||
|
||||
**When invoked**: acceptance-criteria drafting during §B2a; final verification pass
|
||||
before a task is marked `[DONE]`.
|
||||
|
||||
**Handoff**: reports pass/fail with specifics (what broke, under what input) back to
|
||||
whichever role owns that surface.
|
||||
|
||||
## 5. Hardware & Performance Compatibility Reviewer
|
||||
|
||||
**Responsibilities**
|
||||
- Check the change against this stack's real constraints: single vs. dual-GPU dev
|
||||
mode (`docker-compose.yml` runs `npm run dev`, a known throughput ceiling), VRAM
|
||||
budget for vLLM (`gpu-memory-utilization` in `config/vllm_config.yaml`), and
|
||||
behavior under the Local/on-prem deployment mode from `AGENTS.md` Part B §B6.
|
||||
- Flag newly introduced heavy Python/Node dependencies, GPU-specific assumptions, or
|
||||
anything that would break the isolated vLLM-server-only environment (no
|
||||
`paddlepaddle-gpu` in this venv — see `AGENTS.md` Part A / `docs/vllm-service.md`).
|
||||
- Flag anything that would degrade badly on lower-spec hardware or slower networks
|
||||
(e.g. the mobile app's 2s polling loop against a slow backend response), and
|
||||
suggest a lighter-weight alternative when one exists.
|
||||
|
||||
**When invoked**: final verification pass, alongside QA, before a task is marked
|
||||
`[DONE]`; also whenever a task adds a new dependency or changes the deployment/
|
||||
runtime surface.
|
||||
|
||||
**Handoff**: blocks `[DONE]` status until concerns are resolved or explicitly
|
||||
accepted as a documented trade-off in `docs/feature-list.md` (this dir).
|
||||
+308
-308
@@ -1,308 +1,308 @@
|
||||
import json
|
||||
import os
|
||||
import glob
|
||||
import pandas as pd
|
||||
from openpyxl.styles import Font, PatternFill, Alignment, Border, Side
|
||||
from openpyxl.utils import get_column_letter
|
||||
|
||||
def clean_val(val):
|
||||
if val is None:
|
||||
return ""
|
||||
s = str(val).strip().upper()
|
||||
s = " ".join(s.split())
|
||||
s = s.replace("PT. ", "PT.")
|
||||
s = s.replace("✓", "").replace("✔", "").strip()
|
||||
return s
|
||||
|
||||
def main():
|
||||
jsonl_file = "backend/uploads/test_images_results.jsonl"
|
||||
manual_labels_pattern = "backend/uploads/manual_label_*.json"
|
||||
xlsx_file = "backend/pfm-web-app/public/comparison_report.xlsx"
|
||||
|
||||
if not os.path.exists(jsonl_file):
|
||||
# Fallback to backend/uploads if run from different dir
|
||||
jsonl_file = "uploads/test_images_results.jsonl"
|
||||
manual_labels_pattern = "uploads/manual_label_*.json"
|
||||
xlsx_file = "pfm-web-app/public/comparison_report.xlsx"
|
||||
|
||||
if not os.path.exists(jsonl_file):
|
||||
print(f"Error: JSONL file not found at {jsonl_file}")
|
||||
return
|
||||
|
||||
# Load automated results
|
||||
auto_results = {}
|
||||
with open(jsonl_file, "r", encoding="utf-8") as f:
|
||||
for line in f:
|
||||
if not line.strip():
|
||||
continue
|
||||
try:
|
||||
data = json.loads(line)
|
||||
filename = data.get("filename")
|
||||
if filename:
|
||||
auto_results[filename] = data
|
||||
except Exception as e:
|
||||
print(f"Skipping line: {e}")
|
||||
|
||||
# Load manual labels
|
||||
manual_files = glob.glob(manual_labels_pattern)
|
||||
manual_labels = {}
|
||||
for mf in manual_files:
|
||||
try:
|
||||
with open(mf, "r", encoding="utf-8") as f:
|
||||
data = json.load(f)
|
||||
filename = data.get("filename")
|
||||
if filename:
|
||||
manual_labels[filename] = data
|
||||
except Exception as e:
|
||||
print(f"Error reading manual label {mf}: {e}")
|
||||
|
||||
print(f"Loaded {len(auto_results)} automated results.")
|
||||
print(f"Loaded {len(manual_labels)} manual labels.")
|
||||
|
||||
# Fields to compare in headers
|
||||
header_fields = [
|
||||
("noPO", "noPO", "PO Number"),
|
||||
("noSO", "noSO", "SO Number"),
|
||||
("noDO", "noDO", "DO Number"),
|
||||
("tanggal", "tanggal", "Date"),
|
||||
("plat", "platTruk", "Plat Nomor"),
|
||||
("customer", "customerInfo", "Customer Name"),
|
||||
("store", "orderUntuk", "Store Name"),
|
||||
("alamat", "alamat", "Alamat")
|
||||
]
|
||||
|
||||
doc_comparison_rows = []
|
||||
item_comparison_rows = []
|
||||
|
||||
# Counters for accuracy calculation
|
||||
stats = {
|
||||
"PO Number": {"match": 0, "total": 0},
|
||||
"SO Number": {"match": 0, "total": 0},
|
||||
"DO Number": {"match": 0, "total": 0},
|
||||
"Date": {"match": 0, "total": 0},
|
||||
"Plat Nomor": {"match": 0, "total": 0},
|
||||
"Customer Name": {"match": 0, "total": 0},
|
||||
"Store Name": {"match": 0, "total": 0},
|
||||
"Alamat": {"match": 0, "total": 0},
|
||||
"Item SKU": {"match": 0, "total": 0},
|
||||
"Item Banyak": {"match": 0, "total": 0},
|
||||
"Item Jumlah": {"match": 0, "total": 0}
|
||||
}
|
||||
|
||||
for filename, manual in manual_labels.items():
|
||||
auto = auto_results.get(filename)
|
||||
if not auto:
|
||||
print(f"Warning: Automated result not found for {filename}")
|
||||
continue
|
||||
|
||||
auto_meta = auto.get("metadata", {})
|
||||
|
||||
# 1. Compare header fields
|
||||
for manual_key, auto_key, field_label in header_fields:
|
||||
m_val = clean_val(manual.get(manual_key))
|
||||
a_val = clean_val(auto_meta.get(auto_key))
|
||||
is_match = (m_val == a_val)
|
||||
|
||||
doc_comparison_rows.append({
|
||||
"Filename": filename,
|
||||
"Field": field_label,
|
||||
"Automated Value (OCR)": a_val if a_val else "(empty)",
|
||||
"Manual Value (Ground Truth)": m_val if m_val else "(empty)",
|
||||
"Match": "Match" if is_match else "Mismatch"
|
||||
})
|
||||
|
||||
stats[field_label]["total"] += 1
|
||||
if is_match:
|
||||
stats[field_label]["match"] += 1
|
||||
|
||||
# 2. Compare items
|
||||
m_items = manual.get("items", [])
|
||||
# We also look at auto.get("items") or auto_meta.get("items")
|
||||
a_items = auto.get("items", [])
|
||||
if not a_items and "items" in auto_meta:
|
||||
a_items = auto_meta.get("items", [])
|
||||
|
||||
# Create dictionaries of items indexed by codeBarang (SKU)
|
||||
m_items_dict = {clean_val(item.get("kodeBarang")): item for item in m_items if clean_val(item.get("kodeBarang"))}
|
||||
a_items_dict = {clean_val(item.get("kodeBarang")): item for item in a_items if clean_val(item.get("kodeBarang"))}
|
||||
|
||||
# Check all unique SKUs across both manual and automated
|
||||
all_skus = set(list(m_items_dict.keys()) + list(a_items_dict.keys()))
|
||||
|
||||
for sku in all_skus:
|
||||
m_item = m_items_dict.get(sku)
|
||||
a_item = a_items_dict.get(sku)
|
||||
|
||||
# Check SKU existence match
|
||||
sku_match = (m_item is not None) and (a_item is not None)
|
||||
stats["Item SKU"]["total"] += 1
|
||||
if sku_match:
|
||||
stats["Item SKU"]["match"] += 1
|
||||
|
||||
m_banyak = clean_val(m_item.get("banyak")) if m_item else ""
|
||||
a_banyak = clean_val(a_item.get("banyak")) if a_item else ""
|
||||
banyak_match = (m_banyak == a_banyak)
|
||||
|
||||
stats["Item Banyak"]["total"] += 1
|
||||
if banyak_match:
|
||||
stats["Item Banyak"]["match"] += 1
|
||||
|
||||
m_jumlah = clean_val(m_item.get("jumlah")) if m_item else ""
|
||||
a_jumlah = clean_val(a_item.get("jumlah")) if a_item else ""
|
||||
jumlah_match = (m_jumlah == a_jumlah)
|
||||
|
||||
stats["Item Jumlah"]["total"] += 1
|
||||
if jumlah_match:
|
||||
stats["Item Jumlah"]["match"] += 1
|
||||
|
||||
# Log code comparison
|
||||
item_comparison_rows.append({
|
||||
"Filename": filename,
|
||||
"Kode Barang (SKU)": sku,
|
||||
"Field": "SKU Existence",
|
||||
"Automated Value (OCR)": sku if a_item else "(not found)",
|
||||
"Manual Value (Ground Truth)": sku if m_item else "(not found)",
|
||||
"Match": "Match" if sku_match else "Mismatch"
|
||||
})
|
||||
|
||||
# Log Banyak comparison
|
||||
item_comparison_rows.append({
|
||||
"Filename": filename,
|
||||
"Kode Barang (SKU)": sku,
|
||||
"Field": "Banyak (Qty Package)",
|
||||
"Automated Value (OCR)": a_banyak if a_banyak else "(empty)",
|
||||
"Manual Value (Ground Truth)": m_banyak if m_banyak else "(empty)",
|
||||
"Match": "Match" if banyak_match else "Mismatch"
|
||||
})
|
||||
|
||||
# Log Jumlah comparison
|
||||
item_comparison_rows.append({
|
||||
"Filename": filename,
|
||||
"Kode Barang (SKU)": sku,
|
||||
"Field": "Jumlah (Qty Unit)",
|
||||
"Automated Value (OCR)": a_jumlah if a_jumlah else "(empty)",
|
||||
"Manual Value (Ground Truth)": m_jumlah if m_jumlah else "(empty)",
|
||||
"Match": "Match" if jumlah_match else "Mismatch"
|
||||
})
|
||||
|
||||
# Prepare summary data
|
||||
summary_rows = []
|
||||
total_matches = 0
|
||||
total_fields = 0
|
||||
for field_label, counts in stats.items():
|
||||
match_cnt = counts["match"]
|
||||
total_cnt = counts["total"]
|
||||
pct = (match_cnt / total_cnt * 100.0) if total_cnt > 0 else 100.0
|
||||
summary_rows.append({
|
||||
"Field / Area": field_label,
|
||||
"Total Checks": total_cnt,
|
||||
"Matches": match_cnt,
|
||||
"Mismatches": total_cnt - match_cnt,
|
||||
"Accuracy (%)": round(pct, 2)
|
||||
})
|
||||
total_matches += match_cnt
|
||||
total_fields += total_cnt
|
||||
|
||||
overall_accuracy = (total_matches / total_fields * 100.0) if total_fields > 0 else 100.0
|
||||
summary_rows.append({
|
||||
"Field / Area": "OVERALL TOTAL",
|
||||
"Total Checks": total_fields,
|
||||
"Matches": total_matches,
|
||||
"Mismatches": total_fields - total_matches,
|
||||
"Accuracy (%)": round(overall_accuracy, 2)
|
||||
})
|
||||
|
||||
df_summary = pd.DataFrame(summary_rows)
|
||||
df_docs = pd.DataFrame(doc_comparison_rows)
|
||||
df_items = pd.DataFrame(item_comparison_rows)
|
||||
|
||||
# Styling setup
|
||||
font_family = "Segoe UI"
|
||||
header_font = Font(name=font_family, size=11, bold=True, color="FFFFFF")
|
||||
regular_font = Font(name=font_family, size=10)
|
||||
bold_font = Font(name=font_family, size=10, bold=True)
|
||||
|
||||
header_fill = PatternFill(start_color="1F4E78", end_color="1F4E78", fill_type="solid") # Dark Blue
|
||||
zebra_fill = PatternFill(start_color="F2F5F8", end_color="F2F5F8", fill_type="solid") # Zebra light blue-gray
|
||||
match_fill = PatternFill(start_color="E2EFDA", end_color="E2EFDA", fill_type="solid") # Light green
|
||||
mismatch_fill = PatternFill(start_color="FCE4D6", end_color="FCE4D6", fill_type="solid") # Light orange
|
||||
|
||||
center_align = Alignment(horizontal="center", vertical="center")
|
||||
left_align = Alignment(horizontal="left", vertical="center")
|
||||
right_align = Alignment(horizontal="right", vertical="center")
|
||||
|
||||
thin_side = Side(border_style="thin", color="D9D9D9")
|
||||
cell_border = Border(left=thin_side, right=thin_side, top=thin_side, bottom=thin_side)
|
||||
|
||||
# Save to Excel
|
||||
os.makedirs(os.path.dirname(xlsx_file), exist_ok=True)
|
||||
with pd.ExcelWriter(xlsx_file, engine='openpyxl') as writer:
|
||||
df_summary.to_excel(writer, sheet_name='Summary Accuracy', index=False)
|
||||
df_docs.to_excel(writer, sheet_name='Header Field Comparison', index=False)
|
||||
df_items.to_excel(writer, sheet_name='Item SKU Comparison', index=False)
|
||||
|
||||
# Style worksheets
|
||||
for sheet_name in ['Summary Accuracy', 'Header Field Comparison', 'Item SKU Comparison']:
|
||||
ws = writer.sheets[sheet_name]
|
||||
max_row = ws.max_row
|
||||
max_col = ws.max_column
|
||||
|
||||
# Header row styling
|
||||
for col in range(1, max_col + 1):
|
||||
cell = ws.cell(row=1, column=col)
|
||||
cell.font = header_font
|
||||
cell.fill = header_fill
|
||||
cell.alignment = center_align
|
||||
|
||||
# Data rows styling
|
||||
for row in range(2, max_row + 1):
|
||||
is_zebra = (row % 2 == 0)
|
||||
|
||||
# Check for Match/Mismatch to apply colors on sheets 2 & 3
|
||||
match_val = None
|
||||
if sheet_name in ['Header Field Comparison', 'Item SKU Comparison']:
|
||||
# Match column is the last column
|
||||
match_cell = ws.cell(row=row, column=max_col)
|
||||
match_val = match_cell.value
|
||||
|
||||
for col in range(1, max_col + 1):
|
||||
cell = ws.cell(row=row, column=col)
|
||||
cell.font = regular_font
|
||||
cell.border = cell_border
|
||||
|
||||
# Apply alignments based on column
|
||||
if sheet_name == 'Summary Accuracy':
|
||||
if col == 1:
|
||||
cell.alignment = left_align
|
||||
else:
|
||||
cell.alignment = right_align
|
||||
|
||||
# Highlight overall total row
|
||||
if row == max_row:
|
||||
cell.font = bold_font
|
||||
cell.fill = match_fill if overall_accuracy > 80 else mismatch_fill
|
||||
else:
|
||||
# For detail sheets
|
||||
if col in [1, 3, 4]:
|
||||
cell.alignment = left_align
|
||||
else:
|
||||
cell.alignment = center_align
|
||||
|
||||
# Color match / mismatch
|
||||
if match_val == "Match":
|
||||
cell.fill = match_fill
|
||||
elif match_val == "Mismatch":
|
||||
cell.fill = mismatch_fill
|
||||
elif is_zebra:
|
||||
cell.fill = zebra_fill
|
||||
|
||||
# Auto-fit columns
|
||||
for col in ws.columns:
|
||||
max_len = max(len(str(cell.value or '')) for cell in col)
|
||||
col_letter = get_column_letter(col[0].column)
|
||||
ws.column_dimensions[col_letter].width = max(max_len + 4, 12)
|
||||
|
||||
print(f"Comparison report generated at {xlsx_file}")
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
import json
|
||||
import os
|
||||
import glob
|
||||
import pandas as pd
|
||||
from openpyxl.styles import Font, PatternFill, Alignment, Border, Side
|
||||
from openpyxl.utils import get_column_letter
|
||||
|
||||
def clean_val(val):
|
||||
if val is None:
|
||||
return ""
|
||||
s = str(val).strip().upper()
|
||||
s = " ".join(s.split())
|
||||
s = s.replace("PT. ", "PT.")
|
||||
s = s.replace("✓", "").replace("✔", "").strip()
|
||||
return s
|
||||
|
||||
def main():
|
||||
jsonl_file = "backend/uploads/test_images_results.jsonl"
|
||||
manual_labels_pattern = "backend/uploads/manual_label_*.json"
|
||||
xlsx_file = "backend/pfm-web-app/public/comparison_report.xlsx"
|
||||
|
||||
if not os.path.exists(jsonl_file):
|
||||
# Fallback to backend/uploads if run from different dir
|
||||
jsonl_file = "uploads/test_images_results.jsonl"
|
||||
manual_labels_pattern = "uploads/manual_label_*.json"
|
||||
xlsx_file = "pfm-web-app/public/comparison_report.xlsx"
|
||||
|
||||
if not os.path.exists(jsonl_file):
|
||||
print(f"Error: JSONL file not found at {jsonl_file}")
|
||||
return
|
||||
|
||||
# Load automated results
|
||||
auto_results = {}
|
||||
with open(jsonl_file, "r", encoding="utf-8") as f:
|
||||
for line in f:
|
||||
if not line.strip():
|
||||
continue
|
||||
try:
|
||||
data = json.loads(line)
|
||||
filename = data.get("filename")
|
||||
if filename:
|
||||
auto_results[filename] = data
|
||||
except Exception as e:
|
||||
print(f"Skipping line: {e}")
|
||||
|
||||
# Load manual labels
|
||||
manual_files = glob.glob(manual_labels_pattern)
|
||||
manual_labels = {}
|
||||
for mf in manual_files:
|
||||
try:
|
||||
with open(mf, "r", encoding="utf-8") as f:
|
||||
data = json.load(f)
|
||||
filename = data.get("filename")
|
||||
if filename:
|
||||
manual_labels[filename] = data
|
||||
except Exception as e:
|
||||
print(f"Error reading manual label {mf}: {e}")
|
||||
|
||||
print(f"Loaded {len(auto_results)} automated results.")
|
||||
print(f"Loaded {len(manual_labels)} manual labels.")
|
||||
|
||||
# Fields to compare in headers
|
||||
header_fields = [
|
||||
("noPO", "noPO", "PO Number"),
|
||||
("noSO", "noSO", "SO Number"),
|
||||
("noDO", "noDO", "DO Number"),
|
||||
("tanggal", "tanggal", "Date"),
|
||||
("plat", "platTruk", "Plat Nomor"),
|
||||
("customer", "customerInfo", "Customer Name"),
|
||||
("store", "orderUntuk", "Store Name"),
|
||||
("alamat", "alamat", "Alamat")
|
||||
]
|
||||
|
||||
doc_comparison_rows = []
|
||||
item_comparison_rows = []
|
||||
|
||||
# Counters for accuracy calculation
|
||||
stats = {
|
||||
"PO Number": {"match": 0, "total": 0},
|
||||
"SO Number": {"match": 0, "total": 0},
|
||||
"DO Number": {"match": 0, "total": 0},
|
||||
"Date": {"match": 0, "total": 0},
|
||||
"Plat Nomor": {"match": 0, "total": 0},
|
||||
"Customer Name": {"match": 0, "total": 0},
|
||||
"Store Name": {"match": 0, "total": 0},
|
||||
"Alamat": {"match": 0, "total": 0},
|
||||
"Item SKU": {"match": 0, "total": 0},
|
||||
"Item Banyak": {"match": 0, "total": 0},
|
||||
"Item Jumlah": {"match": 0, "total": 0}
|
||||
}
|
||||
|
||||
for filename, manual in manual_labels.items():
|
||||
auto = auto_results.get(filename)
|
||||
if not auto:
|
||||
print(f"Warning: Automated result not found for {filename}")
|
||||
continue
|
||||
|
||||
auto_meta = auto.get("metadata", {})
|
||||
|
||||
# 1. Compare header fields
|
||||
for manual_key, auto_key, field_label in header_fields:
|
||||
m_val = clean_val(manual.get(manual_key))
|
||||
a_val = clean_val(auto_meta.get(auto_key))
|
||||
is_match = (m_val == a_val)
|
||||
|
||||
doc_comparison_rows.append({
|
||||
"Filename": filename,
|
||||
"Field": field_label,
|
||||
"Automated Value (OCR)": a_val if a_val else "(empty)",
|
||||
"Manual Value (Ground Truth)": m_val if m_val else "(empty)",
|
||||
"Match": "Match" if is_match else "Mismatch"
|
||||
})
|
||||
|
||||
stats[field_label]["total"] += 1
|
||||
if is_match:
|
||||
stats[field_label]["match"] += 1
|
||||
|
||||
# 2. Compare items
|
||||
m_items = manual.get("items", [])
|
||||
# We also look at auto.get("items") or auto_meta.get("items")
|
||||
a_items = auto.get("items", [])
|
||||
if not a_items and "items" in auto_meta:
|
||||
a_items = auto_meta.get("items", [])
|
||||
|
||||
# Create dictionaries of items indexed by codeBarang (SKU)
|
||||
m_items_dict = {clean_val(item.get("kodeBarang")): item for item in m_items if clean_val(item.get("kodeBarang"))}
|
||||
a_items_dict = {clean_val(item.get("kodeBarang")): item for item in a_items if clean_val(item.get("kodeBarang"))}
|
||||
|
||||
# Check all unique SKUs across both manual and automated
|
||||
all_skus = set(list(m_items_dict.keys()) + list(a_items_dict.keys()))
|
||||
|
||||
for sku in all_skus:
|
||||
m_item = m_items_dict.get(sku)
|
||||
a_item = a_items_dict.get(sku)
|
||||
|
||||
# Check SKU existence match
|
||||
sku_match = (m_item is not None) and (a_item is not None)
|
||||
stats["Item SKU"]["total"] += 1
|
||||
if sku_match:
|
||||
stats["Item SKU"]["match"] += 1
|
||||
|
||||
m_banyak = clean_val(m_item.get("banyak")) if m_item else ""
|
||||
a_banyak = clean_val(a_item.get("banyak")) if a_item else ""
|
||||
banyak_match = (m_banyak == a_banyak)
|
||||
|
||||
stats["Item Banyak"]["total"] += 1
|
||||
if banyak_match:
|
||||
stats["Item Banyak"]["match"] += 1
|
||||
|
||||
m_jumlah = clean_val(m_item.get("jumlah")) if m_item else ""
|
||||
a_jumlah = clean_val(a_item.get("jumlah")) if a_item else ""
|
||||
jumlah_match = (m_jumlah == a_jumlah)
|
||||
|
||||
stats["Item Jumlah"]["total"] += 1
|
||||
if jumlah_match:
|
||||
stats["Item Jumlah"]["match"] += 1
|
||||
|
||||
# Log code comparison
|
||||
item_comparison_rows.append({
|
||||
"Filename": filename,
|
||||
"Kode Barang (SKU)": sku,
|
||||
"Field": "SKU Existence",
|
||||
"Automated Value (OCR)": sku if a_item else "(not found)",
|
||||
"Manual Value (Ground Truth)": sku if m_item else "(not found)",
|
||||
"Match": "Match" if sku_match else "Mismatch"
|
||||
})
|
||||
|
||||
# Log Banyak comparison
|
||||
item_comparison_rows.append({
|
||||
"Filename": filename,
|
||||
"Kode Barang (SKU)": sku,
|
||||
"Field": "Banyak (Qty Package)",
|
||||
"Automated Value (OCR)": a_banyak if a_banyak else "(empty)",
|
||||
"Manual Value (Ground Truth)": m_banyak if m_banyak else "(empty)",
|
||||
"Match": "Match" if banyak_match else "Mismatch"
|
||||
})
|
||||
|
||||
# Log Jumlah comparison
|
||||
item_comparison_rows.append({
|
||||
"Filename": filename,
|
||||
"Kode Barang (SKU)": sku,
|
||||
"Field": "Jumlah (Qty Unit)",
|
||||
"Automated Value (OCR)": a_jumlah if a_jumlah else "(empty)",
|
||||
"Manual Value (Ground Truth)": m_jumlah if m_jumlah else "(empty)",
|
||||
"Match": "Match" if jumlah_match else "Mismatch"
|
||||
})
|
||||
|
||||
# Prepare summary data
|
||||
summary_rows = []
|
||||
total_matches = 0
|
||||
total_fields = 0
|
||||
for field_label, counts in stats.items():
|
||||
match_cnt = counts["match"]
|
||||
total_cnt = counts["total"]
|
||||
pct = (match_cnt / total_cnt * 100.0) if total_cnt > 0 else 100.0
|
||||
summary_rows.append({
|
||||
"Field / Area": field_label,
|
||||
"Total Checks": total_cnt,
|
||||
"Matches": match_cnt,
|
||||
"Mismatches": total_cnt - match_cnt,
|
||||
"Accuracy (%)": round(pct, 2)
|
||||
})
|
||||
total_matches += match_cnt
|
||||
total_fields += total_cnt
|
||||
|
||||
overall_accuracy = (total_matches / total_fields * 100.0) if total_fields > 0 else 100.0
|
||||
summary_rows.append({
|
||||
"Field / Area": "OVERALL TOTAL",
|
||||
"Total Checks": total_fields,
|
||||
"Matches": total_matches,
|
||||
"Mismatches": total_fields - total_matches,
|
||||
"Accuracy (%)": round(overall_accuracy, 2)
|
||||
})
|
||||
|
||||
df_summary = pd.DataFrame(summary_rows)
|
||||
df_docs = pd.DataFrame(doc_comparison_rows)
|
||||
df_items = pd.DataFrame(item_comparison_rows)
|
||||
|
||||
# Styling setup
|
||||
font_family = "Segoe UI"
|
||||
header_font = Font(name=font_family, size=11, bold=True, color="FFFFFF")
|
||||
regular_font = Font(name=font_family, size=10)
|
||||
bold_font = Font(name=font_family, size=10, bold=True)
|
||||
|
||||
header_fill = PatternFill(start_color="1F4E78", end_color="1F4E78", fill_type="solid") # Dark Blue
|
||||
zebra_fill = PatternFill(start_color="F2F5F8", end_color="F2F5F8", fill_type="solid") # Zebra light blue-gray
|
||||
match_fill = PatternFill(start_color="E2EFDA", end_color="E2EFDA", fill_type="solid") # Light green
|
||||
mismatch_fill = PatternFill(start_color="FCE4D6", end_color="FCE4D6", fill_type="solid") # Light orange
|
||||
|
||||
center_align = Alignment(horizontal="center", vertical="center")
|
||||
left_align = Alignment(horizontal="left", vertical="center")
|
||||
right_align = Alignment(horizontal="right", vertical="center")
|
||||
|
||||
thin_side = Side(border_style="thin", color="D9D9D9")
|
||||
cell_border = Border(left=thin_side, right=thin_side, top=thin_side, bottom=thin_side)
|
||||
|
||||
# Save to Excel
|
||||
os.makedirs(os.path.dirname(xlsx_file), exist_ok=True)
|
||||
with pd.ExcelWriter(xlsx_file, engine='openpyxl') as writer:
|
||||
df_summary.to_excel(writer, sheet_name='Summary Accuracy', index=False)
|
||||
df_docs.to_excel(writer, sheet_name='Header Field Comparison', index=False)
|
||||
df_items.to_excel(writer, sheet_name='Item SKU Comparison', index=False)
|
||||
|
||||
# Style worksheets
|
||||
for sheet_name in ['Summary Accuracy', 'Header Field Comparison', 'Item SKU Comparison']:
|
||||
ws = writer.sheets[sheet_name]
|
||||
max_row = ws.max_row
|
||||
max_col = ws.max_column
|
||||
|
||||
# Header row styling
|
||||
for col in range(1, max_col + 1):
|
||||
cell = ws.cell(row=1, column=col)
|
||||
cell.font = header_font
|
||||
cell.fill = header_fill
|
||||
cell.alignment = center_align
|
||||
|
||||
# Data rows styling
|
||||
for row in range(2, max_row + 1):
|
||||
is_zebra = (row % 2 == 0)
|
||||
|
||||
# Check for Match/Mismatch to apply colors on sheets 2 & 3
|
||||
match_val = None
|
||||
if sheet_name in ['Header Field Comparison', 'Item SKU Comparison']:
|
||||
# Match column is the last column
|
||||
match_cell = ws.cell(row=row, column=max_col)
|
||||
match_val = match_cell.value
|
||||
|
||||
for col in range(1, max_col + 1):
|
||||
cell = ws.cell(row=row, column=col)
|
||||
cell.font = regular_font
|
||||
cell.border = cell_border
|
||||
|
||||
# Apply alignments based on column
|
||||
if sheet_name == 'Summary Accuracy':
|
||||
if col == 1:
|
||||
cell.alignment = left_align
|
||||
else:
|
||||
cell.alignment = right_align
|
||||
|
||||
# Highlight overall total row
|
||||
if row == max_row:
|
||||
cell.font = bold_font
|
||||
cell.fill = match_fill if overall_accuracy > 80 else mismatch_fill
|
||||
else:
|
||||
# For detail sheets
|
||||
if col in [1, 3, 4]:
|
||||
cell.alignment = left_align
|
||||
else:
|
||||
cell.alignment = center_align
|
||||
|
||||
# Color match / mismatch
|
||||
if match_val == "Match":
|
||||
cell.fill = match_fill
|
||||
elif match_val == "Mismatch":
|
||||
cell.fill = mismatch_fill
|
||||
elif is_zebra:
|
||||
cell.fill = zebra_fill
|
||||
|
||||
# Auto-fit columns
|
||||
for col in ws.columns:
|
||||
max_len = max(len(str(cell.value or '')) for cell in col)
|
||||
col_letter = get_column_letter(col[0].column)
|
||||
ws.column_dimensions[col_letter].width = max(max_len + 4, 12)
|
||||
|
||||
print(f"Comparison report generated at {xlsx_file}")
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
+451
-451
@@ -1,451 +1,451 @@
|
||||
import json
|
||||
import os
|
||||
import pandas as pd
|
||||
from openpyxl.styles import Font, PatternFill, Alignment, Border, Side
|
||||
from openpyxl.utils import get_column_letter
|
||||
|
||||
def clean_val(val):
|
||||
if val is None:
|
||||
return ""
|
||||
s = str(val).strip().upper()
|
||||
if s in ["N/A", "NOT FOUND", "NOTFOUND", "EMPTY", "NONE", "-", "N / A"]:
|
||||
return ""
|
||||
s = " ".join(s.split())
|
||||
s = s.replace("PT. ", "PT.")
|
||||
s = s.replace("✓", "").replace("✔", "").strip()
|
||||
return s
|
||||
|
||||
def main():
|
||||
ai_file = "sources/ai_results.json"
|
||||
manual_file = "sources/manual_labels.json"
|
||||
xlsx_file = "sources/comparison_report.xlsx"
|
||||
images_dir = "sources/test-images"
|
||||
|
||||
if not os.path.exists(ai_file):
|
||||
# Fallback to backend/sources
|
||||
ai_file = "backend/sources/ai_results.json"
|
||||
manual_file = "backend/sources/manual_labels.json"
|
||||
xlsx_file = "backend/sources/comparison_report.xlsx"
|
||||
images_dir = "backend/sources/test-images"
|
||||
|
||||
if not os.path.exists(ai_file):
|
||||
print(f"Error: AI results file not found at {ai_file}")
|
||||
return
|
||||
|
||||
if not os.path.exists(manual_file):
|
||||
print(f"Error: Manual labels file not found at {manual_file}")
|
||||
return
|
||||
|
||||
# Load data
|
||||
with open(ai_file, "r", encoding="utf-8") as f:
|
||||
ai_data = json.load(f)
|
||||
|
||||
with open(manual_file, "r", encoding="utf-8") as f:
|
||||
manual_data = json.load(f)
|
||||
|
||||
# Convert to dict for lookup by filename
|
||||
ai_dict = {item.get("filename"): item for item in ai_data if item.get("filename")}
|
||||
manual_dict = {item.get("filename"): item for item in manual_data if item.get("filename")}
|
||||
|
||||
print(f"Loaded {len(ai_dict)} AI results from file.")
|
||||
print(f"Loaded {len(manual_dict)} manual labels from file.")
|
||||
|
||||
# Scan for physical image files in test-images folder
|
||||
existing_images = None
|
||||
if os.path.exists(images_dir):
|
||||
existing_images = set(os.listdir(images_dir))
|
||||
print(f"Found {len(existing_images)} physical images in '{images_dir}'.")
|
||||
else:
|
||||
print(f"Warning: Images directory not found at '{images_dir}'.")
|
||||
|
||||
# Find mismatches in file lists
|
||||
only_in_ai = set(ai_dict.keys()) - set(manual_dict.keys())
|
||||
only_in_manual = set(manual_dict.keys()) - set(ai_dict.keys())
|
||||
if only_in_ai:
|
||||
print(f"Warning: {len(only_in_ai)} files exist only in AI results: {only_in_ai}")
|
||||
if only_in_manual:
|
||||
print(f"Warning: {len(only_in_manual)} files exist only in Manual labels: {only_in_manual}")
|
||||
|
||||
# Determine files to compare (must exist in AI results, Manual labels, and physically as images if directory is available)
|
||||
common_filenames = set(ai_dict.keys()) & set(manual_dict.keys())
|
||||
|
||||
if existing_images is not None:
|
||||
deleted_images = common_filenames - existing_images
|
||||
if deleted_images:
|
||||
print(f"Info: Excluded {len(deleted_images)} files that were physically deleted from images folder: {deleted_images}")
|
||||
all_filenames = sorted(list(common_filenames & existing_images))
|
||||
else:
|
||||
all_filenames = sorted(list(common_filenames))
|
||||
|
||||
print(f"Comparing {len(all_filenames)} matching images.")
|
||||
|
||||
header_fields = [
|
||||
("noPO", "noPO", "PO Number"),
|
||||
("noSO", "noSO", "SO Number"),
|
||||
("noDO", "noDO", "DO Number"),
|
||||
("tanggal", "tanggal", "Date"),
|
||||
("plat", "platTruk", "Plat Nomor"),
|
||||
("customer", "customerInfo", "Customer Name"),
|
||||
("store", "orderUntuk", "Store Name"),
|
||||
("alamat", "alamat", "Alamat")
|
||||
]
|
||||
|
||||
doc_comparison_rows = []
|
||||
item_comparison_rows = []
|
||||
|
||||
# Counters for accuracy calculation
|
||||
stats = {
|
||||
"PO Number": {"match": 0, "total": 0},
|
||||
"SO Number": {"match": 0, "total": 0},
|
||||
"DO Number": {"match": 0, "total": 0},
|
||||
"Date": {"match": 0, "total": 0},
|
||||
"Plat Nomor": {"match": 0, "total": 0},
|
||||
"Customer Name": {"match": 0, "total": 0},
|
||||
"Store Name": {"match": 0, "total": 0},
|
||||
"Alamat": {"match": 0, "total": 0},
|
||||
"Item SKU": {"match": 0, "total": 0},
|
||||
"Item Banyak": {"match": 0, "total": 0},
|
||||
"Item Jumlah": {"match": 0, "total": 0}
|
||||
}
|
||||
|
||||
# Document-level side-by-side rows
|
||||
doc_side_by_side_rows = []
|
||||
|
||||
for filename in all_filenames:
|
||||
manual = manual_dict.get(filename)
|
||||
ai = ai_dict.get(filename)
|
||||
|
||||
if not manual:
|
||||
print(f"Warning: Manual label not found for {filename} (exists only in AI results)")
|
||||
continue
|
||||
if not ai:
|
||||
print(f"Warning: AI result not found for {filename} (exists only in Manual labels)")
|
||||
continue
|
||||
|
||||
ai_meta = ai.get("layer3Final", {})
|
||||
|
||||
# 1. Compare header fields (Vertical format for filtering)
|
||||
sxs_row = {"Filename": filename}
|
||||
for manual_key, ai_key, field_label in header_fields:
|
||||
m_val = clean_val(manual.get(manual_key))
|
||||
a_val = clean_val(ai_meta.get(ai_key))
|
||||
is_match = (m_val == a_val)
|
||||
|
||||
doc_comparison_rows.append({
|
||||
"Filename": filename,
|
||||
"Field": field_label,
|
||||
"AI Value (OCR)": a_val if a_val else "(empty)",
|
||||
"Manual Value (Ground Truth)": m_val if m_val else "(empty)",
|
||||
"Match": "Match" if is_match else "Mismatch"
|
||||
})
|
||||
|
||||
# Side-by-side
|
||||
sxs_row[f"{field_label} (AI)"] = a_val if a_val else ""
|
||||
sxs_row[f"{field_label} (Manual)"] = m_val if m_val else ""
|
||||
sxs_row[f"{field_label} Status"] = "Match" if is_match else "Mismatch"
|
||||
|
||||
stats[field_label]["total"] += 1
|
||||
if is_match:
|
||||
stats[field_label]["match"] += 1
|
||||
|
||||
doc_side_by_side_rows.append(sxs_row)
|
||||
|
||||
# 2. Compare items
|
||||
m_items = manual.get("items", [])
|
||||
ai_items = ai.get("items", [])
|
||||
if not ai_items and "items" in ai_meta:
|
||||
ai_items = ai_meta.get("items", [])
|
||||
|
||||
# Create dictionaries of items indexed by codeBarang (SKU)
|
||||
m_items_dict = {clean_val(item.get("kodeBarang")): item for item in m_items if clean_val(item.get("kodeBarang"))}
|
||||
ai_items_dict = {clean_val(item.get("kodeBarang")): item for item in ai_items if clean_val(item.get("kodeBarang"))}
|
||||
|
||||
# Check all unique SKUs across both manual and AI
|
||||
all_skus = set(list(m_items_dict.keys()) + list(ai_items_dict.keys()))
|
||||
|
||||
for sku in all_skus:
|
||||
m_item = m_items_dict.get(sku)
|
||||
ai_item = ai_items_dict.get(sku)
|
||||
|
||||
# Check SKU existence match
|
||||
sku_match = (m_item is not None) and (ai_item is not None)
|
||||
stats["Item SKU"]["total"] += 1
|
||||
if sku_match:
|
||||
stats["Item SKU"]["match"] += 1
|
||||
|
||||
m_banyak = clean_val(m_item.get("banyak")) if m_item else ""
|
||||
ai_banyak = clean_val(ai_item.get("banyak")) if ai_item else ""
|
||||
banyak_match = (m_banyak == ai_banyak)
|
||||
|
||||
stats["Item Banyak"]["total"] += 1
|
||||
if banyak_match:
|
||||
stats["Item Banyak"]["match"] += 1
|
||||
|
||||
m_jumlah = clean_val(m_item.get("jumlah")) if m_item else ""
|
||||
ai_jumlah = clean_val(ai_item.get("jumlah")) if ai_item else ""
|
||||
jumlah_match = (m_jumlah == ai_jumlah)
|
||||
|
||||
stats["Item Jumlah"]["total"] += 1
|
||||
if jumlah_match:
|
||||
stats["Item Jumlah"]["match"] += 1
|
||||
|
||||
# Log code comparison
|
||||
item_comparison_rows.append({
|
||||
"Filename": filename,
|
||||
"Kode Barang (SKU)": sku,
|
||||
"Field": "SKU Existence",
|
||||
"AI Value (OCR)": sku if ai_item else "(not found)",
|
||||
"Manual Value (Ground Truth)": sku if m_item else "(not found)",
|
||||
"Match": "Match" if sku_match else "Mismatch"
|
||||
})
|
||||
|
||||
# Log Banyak comparison
|
||||
item_comparison_rows.append({
|
||||
"Filename": filename,
|
||||
"Kode Barang (SKU)": sku,
|
||||
"Field": "Banyak (Qty Package)",
|
||||
"AI Value (OCR)": ai_banyak if ai_banyak else "(empty)",
|
||||
"Manual Value (Ground Truth)": m_banyak if m_banyak else "(empty)",
|
||||
"Match": "Match" if banyak_match else "Mismatch"
|
||||
})
|
||||
|
||||
# Log Jumlah comparison
|
||||
item_comparison_rows.append({
|
||||
"Filename": filename,
|
||||
"Kode Barang (SKU)": sku,
|
||||
"Field": "Jumlah (Qty Unit)",
|
||||
"AI Value (OCR)": ai_jumlah if ai_jumlah else "(empty)",
|
||||
"Manual Value (Ground Truth)": m_jumlah if m_jumlah else "(empty)",
|
||||
"Match": "Match" if jumlah_match else "Mismatch"
|
||||
})
|
||||
|
||||
# Prepare summary data
|
||||
summary_rows = []
|
||||
total_matches = 0
|
||||
total_fields = 0
|
||||
for field_label, counts in stats.items():
|
||||
match_cnt = counts["match"]
|
||||
total_cnt = counts["total"]
|
||||
pct = (match_cnt / total_cnt * 100.0) if total_cnt > 0 else 100.0
|
||||
summary_rows.append({
|
||||
"Field / Area": field_label,
|
||||
"Total Checks": total_cnt,
|
||||
"Matches": match_cnt,
|
||||
"Mismatches": total_cnt - match_cnt,
|
||||
"Accuracy (%)": round(pct, 2)
|
||||
})
|
||||
total_matches += match_cnt
|
||||
total_fields += total_cnt
|
||||
|
||||
overall_accuracy = (total_matches / total_fields * 100.0) if total_fields > 0 else 100.0
|
||||
summary_rows.append({
|
||||
"Field / Area": "OVERALL TOTAL",
|
||||
"Total Checks": total_fields,
|
||||
"Matches": total_matches,
|
||||
"Mismatches": total_fields - total_matches,
|
||||
"Accuracy (%)": round(overall_accuracy, 2)
|
||||
})
|
||||
|
||||
df_summary = pd.DataFrame(summary_rows)
|
||||
df_docs = pd.DataFrame(doc_comparison_rows)
|
||||
df_sxs = pd.DataFrame(doc_side_by_side_rows)
|
||||
df_items = pd.DataFrame(item_comparison_rows)
|
||||
|
||||
# Styling setup
|
||||
font_family = "Segoe UI"
|
||||
header_font = Font(name=font_family, size=11, bold=True, color="FFFFFF")
|
||||
regular_font = Font(name=font_family, size=10)
|
||||
bold_font = Font(name=font_family, size=10, bold=True)
|
||||
title_font = Font(name=font_family, size=16, bold=True, color="1F4E78")
|
||||
|
||||
header_fill = PatternFill(start_color="1F4E78", end_color="1F4E78", fill_type="solid") # Dark Blue
|
||||
zebra_fill = PatternFill(start_color="F2F5F8", end_color="F2F5F8", fill_type="solid") # Zebra light blue-gray
|
||||
match_fill = PatternFill(start_color="E2EFDA", end_color="E2EFDA", fill_type="solid") # Light green
|
||||
mismatch_fill = PatternFill(start_color="FCE4D6", end_color="FCE4D6", fill_type="solid") # Light orange
|
||||
|
||||
center_align = Alignment(horizontal="center", vertical="center")
|
||||
left_align = Alignment(horizontal="left", vertical="center")
|
||||
right_align = Alignment(horizontal="right", vertical="center")
|
||||
|
||||
thin_side = Side(border_style="thin", color="D9D9D9")
|
||||
cell_border = Border(left=thin_side, right=thin_side, top=thin_side, bottom=thin_side)
|
||||
|
||||
# Save to Excel
|
||||
os.makedirs(os.path.dirname(xlsx_file), exist_ok=True)
|
||||
|
||||
writer = None
|
||||
for attempt in range(1, 10):
|
||||
try:
|
||||
writer = pd.ExcelWriter(xlsx_file, engine='openpyxl')
|
||||
break
|
||||
except PermissionError:
|
||||
base_dir = os.path.dirname(xlsx_file)
|
||||
filename = os.path.basename(xlsx_file)
|
||||
name, ext = os.path.splitext(filename)
|
||||
if "_" in name and name.split("_")[-1].isdigit():
|
||||
name = "_".join(name.split("_")[:-1])
|
||||
xlsx_file = os.path.join(base_dir, f"{name}_{attempt}{ext}")
|
||||
|
||||
if writer is None:
|
||||
print("Error: Could not open the Excel writer because the file is locked.")
|
||||
return
|
||||
|
||||
with writer:
|
||||
df_summary.to_excel(writer, sheet_name='Summary Accuracy', index=False, startrow=3)
|
||||
df_sxs.to_excel(writer, sheet_name='Side-by-Side Comparison', index=False)
|
||||
df_docs.to_excel(writer, sheet_name='Header Field Comparison', index=False)
|
||||
df_items.to_excel(writer, sheet_name='Item SKU Comparison', index=False)
|
||||
|
||||
# 1. Style Summary Sheet with a Title Banner
|
||||
ws_sum = writer.sheets['Summary Accuracy']
|
||||
ws_sum.views.sheetView[0].showGridLines = True
|
||||
ws_sum.cell(row=1, column=1, value="AI OCR vs. Manual Ground Truth Accuracy Report").font = title_font
|
||||
ws_sum.row_dimensions[1].height = 30
|
||||
|
||||
# Style Summary Table Headers
|
||||
max_col_sum = df_summary.shape[1]
|
||||
for col in range(1, max_col_sum + 1):
|
||||
cell = ws_sum.cell(row=4, column=col)
|
||||
cell.font = header_font
|
||||
cell.fill = header_fill
|
||||
cell.alignment = center_align
|
||||
cell.border = cell_border
|
||||
|
||||
# Style Summary Data
|
||||
max_row_sum = ws_sum.max_row
|
||||
for row in range(5, max_row_sum + 1):
|
||||
for col in range(1, max_col_sum + 1):
|
||||
cell = ws_sum.cell(row=row, column=col)
|
||||
cell.font = regular_font
|
||||
cell.border = cell_border
|
||||
if col == 1:
|
||||
cell.alignment = left_align
|
||||
else:
|
||||
cell.alignment = right_align
|
||||
|
||||
# Zebra style
|
||||
if row % 2 == 0 and row != max_row_sum:
|
||||
cell.fill = zebra_fill
|
||||
|
||||
# Format percentage
|
||||
if col == 5 and isinstance(cell.value, (int, float)):
|
||||
cell.number_format = '0.00"%"'
|
||||
|
||||
# Bold overall total row
|
||||
if row == max_row_sum:
|
||||
for col in range(1, max_col_sum + 1):
|
||||
c = ws_sum.cell(row=row, column=col)
|
||||
c.font = bold_font
|
||||
c.fill = match_fill if overall_accuracy > 80 else mismatch_fill
|
||||
|
||||
# Auto-adjust column width for Summary
|
||||
for col in ws_sum.columns:
|
||||
max_len = max(len(str(cell.value or '')) for cell in col)
|
||||
col_letter = get_column_letter(col[0].column)
|
||||
ws_sum.column_dimensions[col_letter].width = max(max_len + 4, 12)
|
||||
|
||||
# Style detail worksheets
|
||||
for sheet_name in ['Side-by-Side Comparison', 'Header Field Comparison', 'Item SKU Comparison']:
|
||||
ws = writer.sheets[sheet_name]
|
||||
ws.views.sheetView[0].showGridLines = True
|
||||
max_row = ws.max_row
|
||||
max_col = ws.max_column
|
||||
|
||||
# Header row styling
|
||||
for col in range(1, max_col + 1):
|
||||
cell = ws.cell(row=1, column=col)
|
||||
cell.font = header_font
|
||||
cell.fill = header_fill
|
||||
cell.alignment = center_align
|
||||
cell.border = cell_border
|
||||
|
||||
# Data rows styling
|
||||
for row in range(2, max_row + 1):
|
||||
is_zebra = (row % 2 == 0)
|
||||
|
||||
# For Side-by-Side Comparison
|
||||
if sheet_name == 'Side-by-Side Comparison':
|
||||
for col in range(1, max_col + 1):
|
||||
cell = ws.cell(row=row, column=col)
|
||||
cell.font = regular_font
|
||||
cell.border = cell_border
|
||||
|
||||
if col == 1:
|
||||
cell.alignment = left_align
|
||||
if is_zebra:
|
||||
cell.fill = zebra_fill
|
||||
else:
|
||||
# Apply alignments and color mismatch/match
|
||||
# Format of headers:
|
||||
# Col 1: Filename
|
||||
# Col 2: PO AI, Col 3: PO Manual, Col 4: PO Status
|
||||
# ... and so on
|
||||
# So status is at index col where (col - 1) % 3 == 0 (4, 7, 10, 13, 16, 19, 22, 25)
|
||||
col_pos = col - 1
|
||||
if col_pos % 3 == 0: # This is a Status column
|
||||
status_val = cell.value
|
||||
cell.alignment = center_align
|
||||
if status_val == "Match":
|
||||
cell.fill = match_fill
|
||||
else:
|
||||
cell.fill = mismatch_fill
|
||||
else: # This is AI or Manual value column
|
||||
cell.alignment = left_align
|
||||
# Match background of the cell with its corresponding status cell (two columns to the right if AI, one if Manual)
|
||||
status_col_idx = col + (2 if col_pos % 3 == 1 else 1)
|
||||
status_val = ws.cell(row=row, column=status_col_idx).value
|
||||
if status_val == "Match":
|
||||
if is_zebra:
|
||||
# Let's keep it subtle
|
||||
pass
|
||||
else:
|
||||
# Highlight mismatches clearly
|
||||
cell.fill = mismatch_fill
|
||||
|
||||
# For vertical comparison sheets
|
||||
else:
|
||||
# Match column is the last column
|
||||
match_cell = ws.cell(row=row, column=max_col)
|
||||
match_val = match_cell.value
|
||||
|
||||
for col in range(1, max_col + 1):
|
||||
cell = ws.cell(row=row, column=col)
|
||||
cell.font = regular_font
|
||||
cell.border = cell_border
|
||||
|
||||
# Apply alignments based on column
|
||||
if col in [1, 2, 3, 4]:
|
||||
cell.alignment = left_align
|
||||
else:
|
||||
cell.alignment = center_align
|
||||
|
||||
# Color match / mismatch
|
||||
if match_val == "Match":
|
||||
cell.fill = match_fill
|
||||
elif match_val == "Mismatch":
|
||||
cell.fill = mismatch_fill
|
||||
elif is_zebra:
|
||||
cell.fill = zebra_fill
|
||||
|
||||
# Auto-fit columns
|
||||
for col in ws.columns:
|
||||
max_len = 0
|
||||
for cell in col:
|
||||
val_str = str(cell.value or '')
|
||||
# Limit long text like Alamat from making column excessively wide
|
||||
if sheet_name == 'Side-by-Side Comparison' and cell.column in [22, 23]: # Alamat
|
||||
max_len = max(max_len, min(len(val_str), 30))
|
||||
elif sheet_name == 'Header Field Comparison' and cell.column in [3, 4]: # Values
|
||||
max_len = max(max_len, min(len(val_str), 40))
|
||||
else:
|
||||
max_len = max(max_len, len(val_str))
|
||||
col_letter = get_column_letter(col[0].column)
|
||||
ws.column_dimensions[col_letter].width = max(max_len + 4, 12)
|
||||
|
||||
print("\n=== Accuracy Report Summary ===")
|
||||
print(df_summary.to_string(index=False))
|
||||
print("===============================\n")
|
||||
print(f"Comparison report generated at {xlsx_file}")
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
import json
|
||||
import os
|
||||
import pandas as pd
|
||||
from openpyxl.styles import Font, PatternFill, Alignment, Border, Side
|
||||
from openpyxl.utils import get_column_letter
|
||||
|
||||
def clean_val(val):
|
||||
if val is None:
|
||||
return ""
|
||||
s = str(val).strip().upper()
|
||||
if s in ["N/A", "NOT FOUND", "NOTFOUND", "EMPTY", "NONE", "-", "N / A"]:
|
||||
return ""
|
||||
s = " ".join(s.split())
|
||||
s = s.replace("PT. ", "PT.")
|
||||
s = s.replace("✓", "").replace("✔", "").strip()
|
||||
return s
|
||||
|
||||
def main():
|
||||
ai_file = "sources/ai_results.json"
|
||||
manual_file = "sources/manual_labels.json"
|
||||
xlsx_file = "sources/comparison_report.xlsx"
|
||||
images_dir = "sources/test-images"
|
||||
|
||||
if not os.path.exists(ai_file):
|
||||
# Fallback to backend/sources
|
||||
ai_file = "backend/sources/ai_results.json"
|
||||
manual_file = "backend/sources/manual_labels.json"
|
||||
xlsx_file = "backend/sources/comparison_report.xlsx"
|
||||
images_dir = "backend/sources/test-images"
|
||||
|
||||
if not os.path.exists(ai_file):
|
||||
print(f"Error: AI results file not found at {ai_file}")
|
||||
return
|
||||
|
||||
if not os.path.exists(manual_file):
|
||||
print(f"Error: Manual labels file not found at {manual_file}")
|
||||
return
|
||||
|
||||
# Load data
|
||||
with open(ai_file, "r", encoding="utf-8") as f:
|
||||
ai_data = json.load(f)
|
||||
|
||||
with open(manual_file, "r", encoding="utf-8") as f:
|
||||
manual_data = json.load(f)
|
||||
|
||||
# Convert to dict for lookup by filename
|
||||
ai_dict = {item.get("filename"): item for item in ai_data if item.get("filename")}
|
||||
manual_dict = {item.get("filename"): item for item in manual_data if item.get("filename")}
|
||||
|
||||
print(f"Loaded {len(ai_dict)} AI results from file.")
|
||||
print(f"Loaded {len(manual_dict)} manual labels from file.")
|
||||
|
||||
# Scan for physical image files in test-images folder
|
||||
existing_images = None
|
||||
if os.path.exists(images_dir):
|
||||
existing_images = set(os.listdir(images_dir))
|
||||
print(f"Found {len(existing_images)} physical images in '{images_dir}'.")
|
||||
else:
|
||||
print(f"Warning: Images directory not found at '{images_dir}'.")
|
||||
|
||||
# Find mismatches in file lists
|
||||
only_in_ai = set(ai_dict.keys()) - set(manual_dict.keys())
|
||||
only_in_manual = set(manual_dict.keys()) - set(ai_dict.keys())
|
||||
if only_in_ai:
|
||||
print(f"Warning: {len(only_in_ai)} files exist only in AI results: {only_in_ai}")
|
||||
if only_in_manual:
|
||||
print(f"Warning: {len(only_in_manual)} files exist only in Manual labels: {only_in_manual}")
|
||||
|
||||
# Determine files to compare (must exist in AI results, Manual labels, and physically as images if directory is available)
|
||||
common_filenames = set(ai_dict.keys()) & set(manual_dict.keys())
|
||||
|
||||
if existing_images is not None:
|
||||
deleted_images = common_filenames - existing_images
|
||||
if deleted_images:
|
||||
print(f"Info: Excluded {len(deleted_images)} files that were physically deleted from images folder: {deleted_images}")
|
||||
all_filenames = sorted(list(common_filenames & existing_images))
|
||||
else:
|
||||
all_filenames = sorted(list(common_filenames))
|
||||
|
||||
print(f"Comparing {len(all_filenames)} matching images.")
|
||||
|
||||
header_fields = [
|
||||
("noPO", "noPO", "PO Number"),
|
||||
("noSO", "noSO", "SO Number"),
|
||||
("noDO", "noDO", "DO Number"),
|
||||
("tanggal", "tanggal", "Date"),
|
||||
("plat", "platTruk", "Plat Nomor"),
|
||||
("customer", "customerInfo", "Customer Name"),
|
||||
("store", "orderUntuk", "Store Name"),
|
||||
("alamat", "alamat", "Alamat")
|
||||
]
|
||||
|
||||
doc_comparison_rows = []
|
||||
item_comparison_rows = []
|
||||
|
||||
# Counters for accuracy calculation
|
||||
stats = {
|
||||
"PO Number": {"match": 0, "total": 0},
|
||||
"SO Number": {"match": 0, "total": 0},
|
||||
"DO Number": {"match": 0, "total": 0},
|
||||
"Date": {"match": 0, "total": 0},
|
||||
"Plat Nomor": {"match": 0, "total": 0},
|
||||
"Customer Name": {"match": 0, "total": 0},
|
||||
"Store Name": {"match": 0, "total": 0},
|
||||
"Alamat": {"match": 0, "total": 0},
|
||||
"Item SKU": {"match": 0, "total": 0},
|
||||
"Item Banyak": {"match": 0, "total": 0},
|
||||
"Item Jumlah": {"match": 0, "total": 0}
|
||||
}
|
||||
|
||||
# Document-level side-by-side rows
|
||||
doc_side_by_side_rows = []
|
||||
|
||||
for filename in all_filenames:
|
||||
manual = manual_dict.get(filename)
|
||||
ai = ai_dict.get(filename)
|
||||
|
||||
if not manual:
|
||||
print(f"Warning: Manual label not found for {filename} (exists only in AI results)")
|
||||
continue
|
||||
if not ai:
|
||||
print(f"Warning: AI result not found for {filename} (exists only in Manual labels)")
|
||||
continue
|
||||
|
||||
ai_meta = ai.get("layer3Final", {})
|
||||
|
||||
# 1. Compare header fields (Vertical format for filtering)
|
||||
sxs_row = {"Filename": filename}
|
||||
for manual_key, ai_key, field_label in header_fields:
|
||||
m_val = clean_val(manual.get(manual_key))
|
||||
a_val = clean_val(ai_meta.get(ai_key))
|
||||
is_match = (m_val == a_val)
|
||||
|
||||
doc_comparison_rows.append({
|
||||
"Filename": filename,
|
||||
"Field": field_label,
|
||||
"AI Value (OCR)": a_val if a_val else "(empty)",
|
||||
"Manual Value (Ground Truth)": m_val if m_val else "(empty)",
|
||||
"Match": "Match" if is_match else "Mismatch"
|
||||
})
|
||||
|
||||
# Side-by-side
|
||||
sxs_row[f"{field_label} (AI)"] = a_val if a_val else ""
|
||||
sxs_row[f"{field_label} (Manual)"] = m_val if m_val else ""
|
||||
sxs_row[f"{field_label} Status"] = "Match" if is_match else "Mismatch"
|
||||
|
||||
stats[field_label]["total"] += 1
|
||||
if is_match:
|
||||
stats[field_label]["match"] += 1
|
||||
|
||||
doc_side_by_side_rows.append(sxs_row)
|
||||
|
||||
# 2. Compare items
|
||||
m_items = manual.get("items", [])
|
||||
ai_items = ai.get("items", [])
|
||||
if not ai_items and "items" in ai_meta:
|
||||
ai_items = ai_meta.get("items", [])
|
||||
|
||||
# Create dictionaries of items indexed by codeBarang (SKU)
|
||||
m_items_dict = {clean_val(item.get("kodeBarang")): item for item in m_items if clean_val(item.get("kodeBarang"))}
|
||||
ai_items_dict = {clean_val(item.get("kodeBarang")): item for item in ai_items if clean_val(item.get("kodeBarang"))}
|
||||
|
||||
# Check all unique SKUs across both manual and AI
|
||||
all_skus = set(list(m_items_dict.keys()) + list(ai_items_dict.keys()))
|
||||
|
||||
for sku in all_skus:
|
||||
m_item = m_items_dict.get(sku)
|
||||
ai_item = ai_items_dict.get(sku)
|
||||
|
||||
# Check SKU existence match
|
||||
sku_match = (m_item is not None) and (ai_item is not None)
|
||||
stats["Item SKU"]["total"] += 1
|
||||
if sku_match:
|
||||
stats["Item SKU"]["match"] += 1
|
||||
|
||||
m_banyak = clean_val(m_item.get("banyak")) if m_item else ""
|
||||
ai_banyak = clean_val(ai_item.get("banyak")) if ai_item else ""
|
||||
banyak_match = (m_banyak == ai_banyak)
|
||||
|
||||
stats["Item Banyak"]["total"] += 1
|
||||
if banyak_match:
|
||||
stats["Item Banyak"]["match"] += 1
|
||||
|
||||
m_jumlah = clean_val(m_item.get("jumlah")) if m_item else ""
|
||||
ai_jumlah = clean_val(ai_item.get("jumlah")) if ai_item else ""
|
||||
jumlah_match = (m_jumlah == ai_jumlah)
|
||||
|
||||
stats["Item Jumlah"]["total"] += 1
|
||||
if jumlah_match:
|
||||
stats["Item Jumlah"]["match"] += 1
|
||||
|
||||
# Log code comparison
|
||||
item_comparison_rows.append({
|
||||
"Filename": filename,
|
||||
"Kode Barang (SKU)": sku,
|
||||
"Field": "SKU Existence",
|
||||
"AI Value (OCR)": sku if ai_item else "(not found)",
|
||||
"Manual Value (Ground Truth)": sku if m_item else "(not found)",
|
||||
"Match": "Match" if sku_match else "Mismatch"
|
||||
})
|
||||
|
||||
# Log Banyak comparison
|
||||
item_comparison_rows.append({
|
||||
"Filename": filename,
|
||||
"Kode Barang (SKU)": sku,
|
||||
"Field": "Banyak (Qty Package)",
|
||||
"AI Value (OCR)": ai_banyak if ai_banyak else "(empty)",
|
||||
"Manual Value (Ground Truth)": m_banyak if m_banyak else "(empty)",
|
||||
"Match": "Match" if banyak_match else "Mismatch"
|
||||
})
|
||||
|
||||
# Log Jumlah comparison
|
||||
item_comparison_rows.append({
|
||||
"Filename": filename,
|
||||
"Kode Barang (SKU)": sku,
|
||||
"Field": "Jumlah (Qty Unit)",
|
||||
"AI Value (OCR)": ai_jumlah if ai_jumlah else "(empty)",
|
||||
"Manual Value (Ground Truth)": m_jumlah if m_jumlah else "(empty)",
|
||||
"Match": "Match" if jumlah_match else "Mismatch"
|
||||
})
|
||||
|
||||
# Prepare summary data
|
||||
summary_rows = []
|
||||
total_matches = 0
|
||||
total_fields = 0
|
||||
for field_label, counts in stats.items():
|
||||
match_cnt = counts["match"]
|
||||
total_cnt = counts["total"]
|
||||
pct = (match_cnt / total_cnt * 100.0) if total_cnt > 0 else 100.0
|
||||
summary_rows.append({
|
||||
"Field / Area": field_label,
|
||||
"Total Checks": total_cnt,
|
||||
"Matches": match_cnt,
|
||||
"Mismatches": total_cnt - match_cnt,
|
||||
"Accuracy (%)": round(pct, 2)
|
||||
})
|
||||
total_matches += match_cnt
|
||||
total_fields += total_cnt
|
||||
|
||||
overall_accuracy = (total_matches / total_fields * 100.0) if total_fields > 0 else 100.0
|
||||
summary_rows.append({
|
||||
"Field / Area": "OVERALL TOTAL",
|
||||
"Total Checks": total_fields,
|
||||
"Matches": total_matches,
|
||||
"Mismatches": total_fields - total_matches,
|
||||
"Accuracy (%)": round(overall_accuracy, 2)
|
||||
})
|
||||
|
||||
df_summary = pd.DataFrame(summary_rows)
|
||||
df_docs = pd.DataFrame(doc_comparison_rows)
|
||||
df_sxs = pd.DataFrame(doc_side_by_side_rows)
|
||||
df_items = pd.DataFrame(item_comparison_rows)
|
||||
|
||||
# Styling setup
|
||||
font_family = "Segoe UI"
|
||||
header_font = Font(name=font_family, size=11, bold=True, color="FFFFFF")
|
||||
regular_font = Font(name=font_family, size=10)
|
||||
bold_font = Font(name=font_family, size=10, bold=True)
|
||||
title_font = Font(name=font_family, size=16, bold=True, color="1F4E78")
|
||||
|
||||
header_fill = PatternFill(start_color="1F4E78", end_color="1F4E78", fill_type="solid") # Dark Blue
|
||||
zebra_fill = PatternFill(start_color="F2F5F8", end_color="F2F5F8", fill_type="solid") # Zebra light blue-gray
|
||||
match_fill = PatternFill(start_color="E2EFDA", end_color="E2EFDA", fill_type="solid") # Light green
|
||||
mismatch_fill = PatternFill(start_color="FCE4D6", end_color="FCE4D6", fill_type="solid") # Light orange
|
||||
|
||||
center_align = Alignment(horizontal="center", vertical="center")
|
||||
left_align = Alignment(horizontal="left", vertical="center")
|
||||
right_align = Alignment(horizontal="right", vertical="center")
|
||||
|
||||
thin_side = Side(border_style="thin", color="D9D9D9")
|
||||
cell_border = Border(left=thin_side, right=thin_side, top=thin_side, bottom=thin_side)
|
||||
|
||||
# Save to Excel
|
||||
os.makedirs(os.path.dirname(xlsx_file), exist_ok=True)
|
||||
|
||||
writer = None
|
||||
for attempt in range(1, 10):
|
||||
try:
|
||||
writer = pd.ExcelWriter(xlsx_file, engine='openpyxl')
|
||||
break
|
||||
except PermissionError:
|
||||
base_dir = os.path.dirname(xlsx_file)
|
||||
filename = os.path.basename(xlsx_file)
|
||||
name, ext = os.path.splitext(filename)
|
||||
if "_" in name and name.split("_")[-1].isdigit():
|
||||
name = "_".join(name.split("_")[:-1])
|
||||
xlsx_file = os.path.join(base_dir, f"{name}_{attempt}{ext}")
|
||||
|
||||
if writer is None:
|
||||
print("Error: Could not open the Excel writer because the file is locked.")
|
||||
return
|
||||
|
||||
with writer:
|
||||
df_summary.to_excel(writer, sheet_name='Summary Accuracy', index=False, startrow=3)
|
||||
df_sxs.to_excel(writer, sheet_name='Side-by-Side Comparison', index=False)
|
||||
df_docs.to_excel(writer, sheet_name='Header Field Comparison', index=False)
|
||||
df_items.to_excel(writer, sheet_name='Item SKU Comparison', index=False)
|
||||
|
||||
# 1. Style Summary Sheet with a Title Banner
|
||||
ws_sum = writer.sheets['Summary Accuracy']
|
||||
ws_sum.views.sheetView[0].showGridLines = True
|
||||
ws_sum.cell(row=1, column=1, value="AI OCR vs. Manual Ground Truth Accuracy Report").font = title_font
|
||||
ws_sum.row_dimensions[1].height = 30
|
||||
|
||||
# Style Summary Table Headers
|
||||
max_col_sum = df_summary.shape[1]
|
||||
for col in range(1, max_col_sum + 1):
|
||||
cell = ws_sum.cell(row=4, column=col)
|
||||
cell.font = header_font
|
||||
cell.fill = header_fill
|
||||
cell.alignment = center_align
|
||||
cell.border = cell_border
|
||||
|
||||
# Style Summary Data
|
||||
max_row_sum = ws_sum.max_row
|
||||
for row in range(5, max_row_sum + 1):
|
||||
for col in range(1, max_col_sum + 1):
|
||||
cell = ws_sum.cell(row=row, column=col)
|
||||
cell.font = regular_font
|
||||
cell.border = cell_border
|
||||
if col == 1:
|
||||
cell.alignment = left_align
|
||||
else:
|
||||
cell.alignment = right_align
|
||||
|
||||
# Zebra style
|
||||
if row % 2 == 0 and row != max_row_sum:
|
||||
cell.fill = zebra_fill
|
||||
|
||||
# Format percentage
|
||||
if col == 5 and isinstance(cell.value, (int, float)):
|
||||
cell.number_format = '0.00"%"'
|
||||
|
||||
# Bold overall total row
|
||||
if row == max_row_sum:
|
||||
for col in range(1, max_col_sum + 1):
|
||||
c = ws_sum.cell(row=row, column=col)
|
||||
c.font = bold_font
|
||||
c.fill = match_fill if overall_accuracy > 80 else mismatch_fill
|
||||
|
||||
# Auto-adjust column width for Summary
|
||||
for col in ws_sum.columns:
|
||||
max_len = max(len(str(cell.value or '')) for cell in col)
|
||||
col_letter = get_column_letter(col[0].column)
|
||||
ws_sum.column_dimensions[col_letter].width = max(max_len + 4, 12)
|
||||
|
||||
# Style detail worksheets
|
||||
for sheet_name in ['Side-by-Side Comparison', 'Header Field Comparison', 'Item SKU Comparison']:
|
||||
ws = writer.sheets[sheet_name]
|
||||
ws.views.sheetView[0].showGridLines = True
|
||||
max_row = ws.max_row
|
||||
max_col = ws.max_column
|
||||
|
||||
# Header row styling
|
||||
for col in range(1, max_col + 1):
|
||||
cell = ws.cell(row=1, column=col)
|
||||
cell.font = header_font
|
||||
cell.fill = header_fill
|
||||
cell.alignment = center_align
|
||||
cell.border = cell_border
|
||||
|
||||
# Data rows styling
|
||||
for row in range(2, max_row + 1):
|
||||
is_zebra = (row % 2 == 0)
|
||||
|
||||
# For Side-by-Side Comparison
|
||||
if sheet_name == 'Side-by-Side Comparison':
|
||||
for col in range(1, max_col + 1):
|
||||
cell = ws.cell(row=row, column=col)
|
||||
cell.font = regular_font
|
||||
cell.border = cell_border
|
||||
|
||||
if col == 1:
|
||||
cell.alignment = left_align
|
||||
if is_zebra:
|
||||
cell.fill = zebra_fill
|
||||
else:
|
||||
# Apply alignments and color mismatch/match
|
||||
# Format of headers:
|
||||
# Col 1: Filename
|
||||
# Col 2: PO AI, Col 3: PO Manual, Col 4: PO Status
|
||||
# ... and so on
|
||||
# So status is at index col where (col - 1) % 3 == 0 (4, 7, 10, 13, 16, 19, 22, 25)
|
||||
col_pos = col - 1
|
||||
if col_pos % 3 == 0: # This is a Status column
|
||||
status_val = cell.value
|
||||
cell.alignment = center_align
|
||||
if status_val == "Match":
|
||||
cell.fill = match_fill
|
||||
else:
|
||||
cell.fill = mismatch_fill
|
||||
else: # This is AI or Manual value column
|
||||
cell.alignment = left_align
|
||||
# Match background of the cell with its corresponding status cell (two columns to the right if AI, one if Manual)
|
||||
status_col_idx = col + (2 if col_pos % 3 == 1 else 1)
|
||||
status_val = ws.cell(row=row, column=status_col_idx).value
|
||||
if status_val == "Match":
|
||||
if is_zebra:
|
||||
# Let's keep it subtle
|
||||
pass
|
||||
else:
|
||||
# Highlight mismatches clearly
|
||||
cell.fill = mismatch_fill
|
||||
|
||||
# For vertical comparison sheets
|
||||
else:
|
||||
# Match column is the last column
|
||||
match_cell = ws.cell(row=row, column=max_col)
|
||||
match_val = match_cell.value
|
||||
|
||||
for col in range(1, max_col + 1):
|
||||
cell = ws.cell(row=row, column=col)
|
||||
cell.font = regular_font
|
||||
cell.border = cell_border
|
||||
|
||||
# Apply alignments based on column
|
||||
if col in [1, 2, 3, 4]:
|
||||
cell.alignment = left_align
|
||||
else:
|
||||
cell.alignment = center_align
|
||||
|
||||
# Color match / mismatch
|
||||
if match_val == "Match":
|
||||
cell.fill = match_fill
|
||||
elif match_val == "Mismatch":
|
||||
cell.fill = mismatch_fill
|
||||
elif is_zebra:
|
||||
cell.fill = zebra_fill
|
||||
|
||||
# Auto-fit columns
|
||||
for col in ws.columns:
|
||||
max_len = 0
|
||||
for cell in col:
|
||||
val_str = str(cell.value or '')
|
||||
# Limit long text like Alamat from making column excessively wide
|
||||
if sheet_name == 'Side-by-Side Comparison' and cell.column in [22, 23]: # Alamat
|
||||
max_len = max(max_len, min(len(val_str), 30))
|
||||
elif sheet_name == 'Header Field Comparison' and cell.column in [3, 4]: # Values
|
||||
max_len = max(max_len, min(len(val_str), 40))
|
||||
else:
|
||||
max_len = max(max_len, len(val_str))
|
||||
col_letter = get_column_letter(col[0].column)
|
||||
ws.column_dimensions[col_letter].width = max(max_len + 4, 12)
|
||||
|
||||
print("\n=== Accuracy Report Summary ===")
|
||||
print(df_summary.to_string(index=False))
|
||||
print("===============================\n")
|
||||
print(f"Comparison report generated at {xlsx_file}")
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
File diff suppressed because it is too large.
Load diff
+266
-266
@@ -1,266 +1,266 @@
|
||||
# Expiry-date extraction cascade, split out of classify_ocr_server.py so it
|
||||
# can be imported (and offline-tested against captured OCR lines) without
|
||||
# loading any models. Pure regex/string logic - no torch/paddle imports.
|
||||
import re
|
||||
|
||||
EXP_KEYWORD_RE = re.compile(
|
||||
r'(?:exp(?:\.|ired)?|tgl(?:\s*exp)?|expiry|bbd|best\s*before|before|best|baik\s*digunakan|\bbb\b)',
|
||||
re.IGNORECASE,
|
||||
)
|
||||
DD_MM_YYYY_RE = re.compile(
|
||||
r'(?<!\d)(0[1-9]|[12]\d|3[01]).*?(0[1-9]|1[0-2]).*?(20\d{2})(?!\d)'
|
||||
)
|
||||
DDMMYYYY_RE = re.compile(
|
||||
r'(?<!\d)(0[1-9]|[12]\d|3[01])(0[1-9]|1[0-2])(20\d{2})(?!\d)'
|
||||
)
|
||||
BB_ATTACHED_DATE_RE = re.compile(
|
||||
r'\b(?:bb|bestbefore)\s*[:.-]?\s*(0[1-9]|[12]\d|3[01])(0[1-9]|1[0-2])(20\d{2})(?!\d)',
|
||||
re.IGNORECASE,
|
||||
)
|
||||
KEYWORD_DIGITS_RE = re.compile(
|
||||
r'(?:exp|expired|tgl|expiry|bbd|before|best|bb|baik|digunakan)\s*[:.-]?\s*(\d{6,8})\b',
|
||||
re.IGNORECASE,
|
||||
)
|
||||
LENIENT_DATE_RE = re.compile(
|
||||
r'(?<!\d)(\d{1,2}).*?(\d{1,2}).*?((?:20)?\d{2})(?!\d)'
|
||||
)
|
||||
# Store price-tag / label-printer lines ("Printed:04/05/2026 19:53",
|
||||
# "Rp.6,800/PC"). The date on these is the moment the shelf label was
|
||||
# printed, never the product's expiry - excluded from the keyword-less
|
||||
# stages so it can't shadow the real date elsewhere on the package.
|
||||
PRICE_TAG_RE = re.compile(r'(?i)printed\s*[:.]?|rp\s*[.,]?\s*\d')
|
||||
|
||||
def is_valid_ddmmyyyy_digits(val: str) -> bool:
|
||||
if len(val) != 8 or not val.isdigit():
|
||||
return False
|
||||
day, month, year = int(val[0:2]), int(val[2:4]), int(val[4:8])
|
||||
return 1 <= day <= 31 and 1 <= month <= 12 and 2000 <= year <= 2099
|
||||
|
||||
def is_plausible_date_parts(day: str, month: str, year: str) -> bool:
|
||||
# Sanity gate for the lenient stage: it happily assembles junk like
|
||||
# "00/22/26" or "1/3/06" out of garbled digit runs. A frozen-food
|
||||
# expiry is always a real calendar day within a few years of today.
|
||||
if not (day.isdigit() and month.isdigit() and year.isdigit()):
|
||||
return False
|
||||
d, m = int(day), int(month)
|
||||
y = int(year) if len(year) == 4 else 2000 + int(year)
|
||||
return 1 <= d <= 31 and 1 <= m <= 12 and 2020 <= y <= 2039
|
||||
|
||||
def format_ddmmyyyy(val: str) -> str:
|
||||
if is_valid_ddmmyyyy_digits(val):
|
||||
return f"{val[0:2]}/{val[2:4]}/{val[4:8]}"
|
||||
return val.upper()
|
||||
|
||||
def format_ddmmyy(val: str) -> str:
|
||||
if len(val) == 6 and val.isdigit():
|
||||
day, month = int(val[0:2]), int(val[2:4])
|
||||
if 1 <= day <= 31 and 1 <= month <= 12:
|
||||
return f"{val[0:2]}/{val[2:4]}/{val[4:6]}"
|
||||
return val.upper()
|
||||
|
||||
def line_has_exp_keyword(line: str) -> bool:
|
||||
if EXP_KEYWORD_RE.search(line):
|
||||
return True
|
||||
# BB05032027 — keyword directly followed by digits
|
||||
return bool(re.search(r'(?i)\b(?:bb|bestbefore)(?:\s*[:.-]?\s*)?\d', line))
|
||||
|
||||
def clean_date_line(line: str) -> str:
|
||||
# 1) Replace "1)" with "0"
|
||||
cleaned = line.replace("1)", "0")
|
||||
|
||||
# 2) Replace "()" with "0"
|
||||
cleaned = cleaned.replace("()", "0")
|
||||
|
||||
# Clean BB misrecognitions (convert B8, 8B, 88 to BB when followed by digits)
|
||||
cleaned = re.sub(r'\b(?:B8|8B|88)(?=\d)', 'BB', cleaned)
|
||||
cleaned = re.sub(r'^(?:B8|8B|88)(?=\d)', 'BB', cleaned)
|
||||
|
||||
# Clean 012/112 month misrecognitions (e.g. 020122027 -> 02022027) —
|
||||
# but only when the line does NOT already hold a valid date: a real
|
||||
# "01122026" (= 01/12/2026) also matches the 112 pattern (0+112+2026)
|
||||
# and would be mangled into 7-digit junk.
|
||||
if not (DDMMYYYY_RE.search(cleaned) or DD_MM_YYYY_RE.search(cleaned)):
|
||||
cleaned = re.sub(r'(?<!\d)(\d{1,2})012(20\d{2})(?!\d)', r'\g<1>02\g<2>', cleaned)
|
||||
cleaned = re.sub(r'(?<!\d)(\d{1,2})([-./\s]+)012([-./\s]+)(20\d{2})(?!\d)', r'\g<1>\g<2>02\g<3>\g<4>', cleaned)
|
||||
cleaned = re.sub(r'(?<!\d)(\d{1,2})112(20\d{2})(?!\d)', r'\g<1>02\g<2>', cleaned)
|
||||
cleaned = re.sub(r'(?<!\d)(\d{1,2})([-./\s]+)112([-./\s]+)(20\d{2})(?!\d)', r'\g<1>\g<2>02\g<3>\g<4>', cleaned)
|
||||
|
||||
# Run contextual replacements
|
||||
for _ in range(3):
|
||||
# letter o/O flanked by digits or boundary -> 0
|
||||
cleaned = re.compile(r'(\d)[oO](\d|\b)').sub(r'\g<1>0\g<2>', cleaned)
|
||||
cleaned = re.compile(r'(\b|\d)[oO](\d)').sub(r'\g<1>0\g<2>', cleaned)
|
||||
|
||||
# letter I/i/l/| flanked by digits -> 1
|
||||
cleaned = re.compile(r'(\d)[Ii|l](\d|\b)').sub(r'\g<1>1\g<2>', cleaned)
|
||||
cleaned = re.compile(r'(\b|\d)[Ii|l](\d)').sub(r'\g<1>1\g<2>', cleaned)
|
||||
|
||||
# letter S/s flanked by digits -> 5
|
||||
cleaned = re.compile(r'(\d)[Ss](\d|\b)').sub(r'\g<1>5\g<2>', cleaned)
|
||||
cleaned = re.compile(r'(\b|\d)[Ss](\d)').sub(r'\g<1>5\g<2>', cleaned)
|
||||
|
||||
# letter Z/z flanked by digits -> 2
|
||||
cleaned = re.compile(r'(\d)[Zz](\d|\b)').sub(r'\g<1>2\g<2>', cleaned)
|
||||
cleaned = re.compile(r'(\b|\d)[Zz](\d)').sub(r'\g<1>2\g<2>', cleaned)
|
||||
|
||||
# letter B flanked by digits -> 8
|
||||
cleaned = re.compile(r'(\d)B(\d|\b)').sub(r'\g<1>8\g<2>', cleaned)
|
||||
cleaned = re.compile(r'(\b|\d)B(\d)').sub(r'\g<1>8\g<2>', cleaned)
|
||||
|
||||
return cleaned
|
||||
|
||||
def extract_expired_date(text_lines):
|
||||
"""Return (formatted_date, line_index, source_line). Prioritises BB/EXP + DDMMYYYY or DD MM YYYY."""
|
||||
if not text_lines:
|
||||
return None, None, None
|
||||
|
||||
cleaned_lines = [clean_date_line(line) for line in text_lines]
|
||||
|
||||
def pick(match, idx, cleaned_line, formatter=None):
|
||||
raw = match.group(0)
|
||||
original_line = text_lines[idx].strip()
|
||||
if match.lastindex and match.lastindex >= 3:
|
||||
formatted = f"{match.group(1)}/{match.group(2)}/{match.group(3)}"
|
||||
elif match.lastindex and match.lastindex >= 1 and match.group(1).isdigit():
|
||||
digits = match.group(1)
|
||||
if len(digits) == 8:
|
||||
formatted = format_ddmmyyyy(digits)
|
||||
elif len(digits) == 6:
|
||||
formatted = format_ddmmyy(digits)
|
||||
else:
|
||||
formatted = digits
|
||||
elif formatter:
|
||||
formatted = formatter(raw)
|
||||
else:
|
||||
formatted = raw.strip().upper()
|
||||
return formatted, idx, original_line
|
||||
|
||||
# 1) BB/EXP keyword lines — compact DDMMYYYY (e.g. BB05032027, EXP 05032027)
|
||||
for idx, line in enumerate(cleaned_lines):
|
||||
if not line_has_exp_keyword(line):
|
||||
continue
|
||||
match = BB_ATTACHED_DATE_RE.search(line) or DDMMYYYY_RE.search(line)
|
||||
if match:
|
||||
return pick(match, idx, line)
|
||||
|
||||
# 2) BB/EXP keyword lines — spaced DD MM YYYY (e.g. BB 05 03 2027)
|
||||
for idx, line in enumerate(cleaned_lines):
|
||||
if not line_has_exp_keyword(line):
|
||||
continue
|
||||
match = DD_MM_YYYY_RE.search(line)
|
||||
if match:
|
||||
return pick(match, idx, line)
|
||||
|
||||
# 3) Keyword + 6–8 digit run (BB05032027 via keyword_digits)
|
||||
for idx, line in enumerate(cleaned_lines):
|
||||
match = KEYWORD_DIGITS_RE.search(line)
|
||||
if match:
|
||||
digits = match.group(1)
|
||||
if len(digits) == 8 and is_valid_ddmmyyyy_digits(digits):
|
||||
return format_ddmmyyyy(digits), idx, text_lines[idx].strip()
|
||||
if len(digits) == 6:
|
||||
return format_ddmmyy(digits), idx, text_lines[idx].strip()
|
||||
|
||||
# 3.5) BB/EXP keyword lines — lenient check for unclear/noisy date formats (e.g. BB 02J 132027)
|
||||
for idx, line in enumerate(cleaned_lines):
|
||||
if not line_has_exp_keyword(line):
|
||||
continue
|
||||
match = LENIENT_DATE_RE.search(line)
|
||||
if match and is_plausible_date_parts(match.group(1), match.group(2), match.group(3)):
|
||||
return pick(match, idx, line)
|
||||
|
||||
# 3.6) Keyword line + date split onto an adjacent line (PaddleOCR sometimes
|
||||
# detects "BB"/"Baik digunakan" as its own box, separate from the date
|
||||
# digits in a neighboring box, e.g. "BB" / "05032027" as two lines).
|
||||
for idx, line in enumerate(cleaned_lines):
|
||||
if not line_has_exp_keyword(line):
|
||||
continue
|
||||
for j in (idx + 1, idx - 1, idx + 2):
|
||||
if j < 0 or j >= len(cleaned_lines) or j == idx:
|
||||
continue
|
||||
neighbor = cleaned_lines[j]
|
||||
combined = f"{line} {neighbor}" if j > idx else f"{neighbor} {line}"
|
||||
match = (BB_ATTACHED_DATE_RE.search(combined)
|
||||
or DDMMYYYY_RE.search(combined)
|
||||
or DD_MM_YYYY_RE.search(combined))
|
||||
if match:
|
||||
report_idx = j if sum(c.isdigit() for c in neighbor) > sum(c.isdigit() for c in line) else idx
|
||||
return pick(match, report_idx, combined)
|
||||
|
||||
# 4) Any line — spaced DD MM YYYY (excluding store price-tag lines)
|
||||
for idx, line in enumerate(cleaned_lines):
|
||||
if PRICE_TAG_RE.search(line):
|
||||
continue
|
||||
match = DD_MM_YYYY_RE.search(line)
|
||||
if match:
|
||||
return pick(match, idx, line)
|
||||
|
||||
# 5) Any line — compact DDMMYYYY (skip likely SKU: same line has 8-digit product code context)
|
||||
for idx, line in enumerate(cleaned_lines):
|
||||
if PRICE_TAG_RE.search(line):
|
||||
continue
|
||||
for match in DDMMYYYY_RE.finditer(line):
|
||||
digits = f"{match.group(1)}{match.group(2)}{match.group(3)}"
|
||||
if is_valid_ddmmyyyy_digits(digits):
|
||||
# Skip if this 8-digit block is the only digits and looks like SKU on label top
|
||||
if re.search(r'\b\d{8}\b', line) and not line_has_exp_keyword(line):
|
||||
if re.search(r'(?:nugget|chicken|fiesta|champ|okey|akumo|frozen|gr)', line, re.I):
|
||||
continue
|
||||
return format_ddmmyyyy(digits), idx, text_lines[idx].strip()
|
||||
|
||||
# 6) Legacy patterns (slashes, month names, etc.)
|
||||
date_patterns = [
|
||||
r'\b\d{2}[-./]\d{2}[-./]\d{2,4}\b',
|
||||
r'\b\d{4}[-./]\d{2}[-./]\d{2}\b',
|
||||
r'\b\d{2}\s+(?:JAN|FEB|MAR|APR|MAY|JUN|JUL|AUG|SEP|OCT|NOV|DEC)[a-zA-Z]*\s+\d{2,4}\b',
|
||||
]
|
||||
for idx, line in enumerate(cleaned_lines):
|
||||
if not line_has_exp_keyword(line):
|
||||
continue
|
||||
for pat in date_patterns:
|
||||
match = re.search(pat, line, re.IGNORECASE)
|
||||
if match:
|
||||
return match.group(0).upper(), idx, text_lines[idx].strip()
|
||||
|
||||
return None, None, None
|
||||
|
||||
def line_contains_expired_date(line: str, expired_date: str) -> bool:
|
||||
if not line or not expired_date:
|
||||
return False
|
||||
digits_only = re.sub(r"\D", "", expired_date)
|
||||
line_digits = re.sub(r"\D", "", line)
|
||||
if len(digits_only) >= 6 and digits_only in line_digits:
|
||||
return True
|
||||
compact = expired_date.replace("/", "")
|
||||
return compact in line.replace(" ", "") or expired_date in line
|
||||
|
||||
def find_expired_crop_index(text_lines, expired_idx, expired_date, polys_len):
|
||||
"""Pick OCR box index for cropping; prefer the line that actually contains the date."""
|
||||
if not expired_date or polys_len <= 0:
|
||||
return None
|
||||
|
||||
if (
|
||||
expired_idx is not None
|
||||
and expired_idx < polys_len
|
||||
and expired_idx < len(text_lines)
|
||||
and line_contains_expired_date(text_lines[expired_idx], expired_date)
|
||||
):
|
||||
return expired_idx
|
||||
|
||||
keyword_match = None
|
||||
for idx, line in enumerate(text_lines):
|
||||
if idx >= polys_len:
|
||||
break
|
||||
if not line_contains_expired_date(line, expired_date):
|
||||
continue
|
||||
if line_has_exp_keyword(line):
|
||||
return idx
|
||||
if keyword_match is None:
|
||||
keyword_match = idx
|
||||
|
||||
if keyword_match is not None:
|
||||
return keyword_match
|
||||
|
||||
if expired_idx is not None and expired_idx < polys_len:
|
||||
return expired_idx
|
||||
return None
|
||||
# Expiry-date extraction cascade, split out of classify_ocr_server.py so it
|
||||
# can be imported (and offline-tested against captured OCR lines) without
|
||||
# loading any models. Pure regex/string logic - no torch/paddle imports.
|
||||
import re
|
||||
|
||||
EXP_KEYWORD_RE = re.compile(
|
||||
r'(?:exp(?:\.|ired)?|tgl(?:\s*exp)?|expiry|bbd|best\s*before|before|best|baik\s*digunakan|\bbb\b)',
|
||||
re.IGNORECASE,
|
||||
)
|
||||
DD_MM_YYYY_RE = re.compile(
|
||||
r'(?<!\d)(0[1-9]|[12]\d|3[01]).*?(0[1-9]|1[0-2]).*?(20\d{2})(?!\d)'
|
||||
)
|
||||
DDMMYYYY_RE = re.compile(
|
||||
r'(?<!\d)(0[1-9]|[12]\d|3[01])(0[1-9]|1[0-2])(20\d{2})(?!\d)'
|
||||
)
|
||||
BB_ATTACHED_DATE_RE = re.compile(
|
||||
r'\b(?:bb|bestbefore)\s*[:.-]?\s*(0[1-9]|[12]\d|3[01])(0[1-9]|1[0-2])(20\d{2})(?!\d)',
|
||||
re.IGNORECASE,
|
||||
)
|
||||
KEYWORD_DIGITS_RE = re.compile(
|
||||
r'(?:exp|expired|tgl|expiry|bbd|before|best|bb|baik|digunakan)\s*[:.-]?\s*(\d{6,8})\b',
|
||||
re.IGNORECASE,
|
||||
)
|
||||
LENIENT_DATE_RE = re.compile(
|
||||
r'(?<!\d)(\d{1,2}).*?(\d{1,2}).*?((?:20)?\d{2})(?!\d)'
|
||||
)
|
||||
# Store price-tag / label-printer lines ("Printed:04/05/2026 19:53",
|
||||
# "Rp.6,800/PC"). The date on these is the moment the shelf label was
|
||||
# printed, never the product's expiry - excluded from the keyword-less
|
||||
# stages so it can't shadow the real date elsewhere on the package.
|
||||
PRICE_TAG_RE = re.compile(r'(?i)printed\s*[:.]?|rp\s*[.,]?\s*\d')
|
||||
|
||||
def is_valid_ddmmyyyy_digits(val: str) -> bool:
|
||||
if len(val) != 8 or not val.isdigit():
|
||||
return False
|
||||
day, month, year = int(val[0:2]), int(val[2:4]), int(val[4:8])
|
||||
return 1 <= day <= 31 and 1 <= month <= 12 and 2000 <= year <= 2099
|
||||
|
||||
def is_plausible_date_parts(day: str, month: str, year: str) -> bool:
|
||||
# Sanity gate for the lenient stage: it happily assembles junk like
|
||||
# "00/22/26" or "1/3/06" out of garbled digit runs. A frozen-food
|
||||
# expiry is always a real calendar day within a few years of today.
|
||||
if not (day.isdigit() and month.isdigit() and year.isdigit()):
|
||||
return False
|
||||
d, m = int(day), int(month)
|
||||
y = int(year) if len(year) == 4 else 2000 + int(year)
|
||||
return 1 <= d <= 31 and 1 <= m <= 12 and 2020 <= y <= 2039
|
||||
|
||||
def format_ddmmyyyy(val: str) -> str:
|
||||
if is_valid_ddmmyyyy_digits(val):
|
||||
return f"{val[0:2]}/{val[2:4]}/{val[4:8]}"
|
||||
return val.upper()
|
||||
|
||||
def format_ddmmyy(val: str) -> str:
|
||||
if len(val) == 6 and val.isdigit():
|
||||
day, month = int(val[0:2]), int(val[2:4])
|
||||
if 1 <= day <= 31 and 1 <= month <= 12:
|
||||
return f"{val[0:2]}/{val[2:4]}/{val[4:6]}"
|
||||
return val.upper()
|
||||
|
||||
def line_has_exp_keyword(line: str) -> bool:
|
||||
if EXP_KEYWORD_RE.search(line):
|
||||
return True
|
||||
# BB05032027 — keyword directly followed by digits
|
||||
return bool(re.search(r'(?i)\b(?:bb|bestbefore)(?:\s*[:.-]?\s*)?\d', line))
|
||||
|
||||
def clean_date_line(line: str) -> str:
|
||||
# 1) Replace "1)" with "0"
|
||||
cleaned = line.replace("1)", "0")
|
||||
|
||||
# 2) Replace "()" with "0"
|
||||
cleaned = cleaned.replace("()", "0")
|
||||
|
||||
# Clean BB misrecognitions (convert B8, 8B, 88 to BB when followed by digits)
|
||||
cleaned = re.sub(r'\b(?:B8|8B|88)(?=\d)', 'BB', cleaned)
|
||||
cleaned = re.sub(r'^(?:B8|8B|88)(?=\d)', 'BB', cleaned)
|
||||
|
||||
# Clean 012/112 month misrecognitions (e.g. 020122027 -> 02022027) —
|
||||
# but only when the line does NOT already hold a valid date: a real
|
||||
# "01122026" (= 01/12/2026) also matches the 112 pattern (0+112+2026)
|
||||
# and would be mangled into 7-digit junk.
|
||||
if not (DDMMYYYY_RE.search(cleaned) or DD_MM_YYYY_RE.search(cleaned)):
|
||||
cleaned = re.sub(r'(?<!\d)(\d{1,2})012(20\d{2})(?!\d)', r'\g<1>02\g<2>', cleaned)
|
||||
cleaned = re.sub(r'(?<!\d)(\d{1,2})([-./\s]+)012([-./\s]+)(20\d{2})(?!\d)', r'\g<1>\g<2>02\g<3>\g<4>', cleaned)
|
||||
cleaned = re.sub(r'(?<!\d)(\d{1,2})112(20\d{2})(?!\d)', r'\g<1>02\g<2>', cleaned)
|
||||
cleaned = re.sub(r'(?<!\d)(\d{1,2})([-./\s]+)112([-./\s]+)(20\d{2})(?!\d)', r'\g<1>\g<2>02\g<3>\g<4>', cleaned)
|
||||
|
||||
# Run contextual replacements
|
||||
for _ in range(3):
|
||||
# letter o/O flanked by digits or boundary -> 0
|
||||
cleaned = re.compile(r'(\d)[oO](\d|\b)').sub(r'\g<1>0\g<2>', cleaned)
|
||||
cleaned = re.compile(r'(\b|\d)[oO](\d)').sub(r'\g<1>0\g<2>', cleaned)
|
||||
|
||||
# letter I/i/l/| flanked by digits -> 1
|
||||
cleaned = re.compile(r'(\d)[Ii|l](\d|\b)').sub(r'\g<1>1\g<2>', cleaned)
|
||||
cleaned = re.compile(r'(\b|\d)[Ii|l](\d)').sub(r'\g<1>1\g<2>', cleaned)
|
||||
|
||||
# letter S/s flanked by digits -> 5
|
||||
cleaned = re.compile(r'(\d)[Ss](\d|\b)').sub(r'\g<1>5\g<2>', cleaned)
|
||||
cleaned = re.compile(r'(\b|\d)[Ss](\d)').sub(r'\g<1>5\g<2>', cleaned)
|
||||
|
||||
# letter Z/z flanked by digits -> 2
|
||||
cleaned = re.compile(r'(\d)[Zz](\d|\b)').sub(r'\g<1>2\g<2>', cleaned)
|
||||
cleaned = re.compile(r'(\b|\d)[Zz](\d)').sub(r'\g<1>2\g<2>', cleaned)
|
||||
|
||||
# letter B flanked by digits -> 8
|
||||
cleaned = re.compile(r'(\d)B(\d|\b)').sub(r'\g<1>8\g<2>', cleaned)
|
||||
cleaned = re.compile(r'(\b|\d)B(\d)').sub(r'\g<1>8\g<2>', cleaned)
|
||||
|
||||
return cleaned
|
||||
|
||||
def extract_expired_date(text_lines):
|
||||
"""Return (formatted_date, line_index, source_line). Prioritises BB/EXP + DDMMYYYY or DD MM YYYY."""
|
||||
if not text_lines:
|
||||
return None, None, None
|
||||
|
||||
cleaned_lines = [clean_date_line(line) for line in text_lines]
|
||||
|
||||
def pick(match, idx, cleaned_line, formatter=None):
|
||||
raw = match.group(0)
|
||||
original_line = text_lines[idx].strip()
|
||||
if match.lastindex and match.lastindex >= 3:
|
||||
formatted = f"{match.group(1)}/{match.group(2)}/{match.group(3)}"
|
||||
elif match.lastindex and match.lastindex >= 1 and match.group(1).isdigit():
|
||||
digits = match.group(1)
|
||||
if len(digits) == 8:
|
||||
formatted = format_ddmmyyyy(digits)
|
||||
elif len(digits) == 6:
|
||||
formatted = format_ddmmyy(digits)
|
||||
else:
|
||||
formatted = digits
|
||||
elif formatter:
|
||||
formatted = formatter(raw)
|
||||
else:
|
||||
formatted = raw.strip().upper()
|
||||
return formatted, idx, original_line
|
||||
|
||||
# 1) BB/EXP keyword lines — compact DDMMYYYY (e.g. BB05032027, EXP 05032027)
|
||||
for idx, line in enumerate(cleaned_lines):
|
||||
if not line_has_exp_keyword(line):
|
||||
continue
|
||||
match = BB_ATTACHED_DATE_RE.search(line) or DDMMYYYY_RE.search(line)
|
||||
if match:
|
||||
return pick(match, idx, line)
|
||||
|
||||
# 2) BB/EXP keyword lines — spaced DD MM YYYY (e.g. BB 05 03 2027)
|
||||
for idx, line in enumerate(cleaned_lines):
|
||||
if not line_has_exp_keyword(line):
|
||||
continue
|
||||
match = DD_MM_YYYY_RE.search(line)
|
||||
if match:
|
||||
return pick(match, idx, line)
|
||||
|
||||
# 3) Keyword + 6–8 digit run (BB05032027 via keyword_digits)
|
||||
for idx, line in enumerate(cleaned_lines):
|
||||
match = KEYWORD_DIGITS_RE.search(line)
|
||||
if match:
|
||||
digits = match.group(1)
|
||||
if len(digits) == 8 and is_valid_ddmmyyyy_digits(digits):
|
||||
return format_ddmmyyyy(digits), idx, text_lines[idx].strip()
|
||||
if len(digits) == 6:
|
||||
return format_ddmmyy(digits), idx, text_lines[idx].strip()
|
||||
|
||||
# 3.5) BB/EXP keyword lines — lenient check for unclear/noisy date formats (e.g. BB 02J 132027)
|
||||
for idx, line in enumerate(cleaned_lines):
|
||||
if not line_has_exp_keyword(line):
|
||||
continue
|
||||
match = LENIENT_DATE_RE.search(line)
|
||||
if match and is_plausible_date_parts(match.group(1), match.group(2), match.group(3)):
|
||||
return pick(match, idx, line)
|
||||
|
||||
# 3.6) Keyword line + date split onto an adjacent line (PaddleOCR sometimes
|
||||
# detects "BB"/"Baik digunakan" as its own box, separate from the date
|
||||
# digits in a neighboring box, e.g. "BB" / "05032027" as two lines).
|
||||
for idx, line in enumerate(cleaned_lines):
|
||||
if not line_has_exp_keyword(line):
|
||||
continue
|
||||
for j in (idx + 1, idx - 1, idx + 2):
|
||||
if j < 0 or j >= len(cleaned_lines) or j == idx:
|
||||
continue
|
||||
neighbor = cleaned_lines[j]
|
||||
combined = f"{line} {neighbor}" if j > idx else f"{neighbor} {line}"
|
||||
match = (BB_ATTACHED_DATE_RE.search(combined)
|
||||
or DDMMYYYY_RE.search(combined)
|
||||
or DD_MM_YYYY_RE.search(combined))
|
||||
if match:
|
||||
report_idx = j if sum(c.isdigit() for c in neighbor) > sum(c.isdigit() for c in line) else idx
|
||||
return pick(match, report_idx, combined)
|
||||
|
||||
# 4) Any line — spaced DD MM YYYY (excluding store price-tag lines)
|
||||
for idx, line in enumerate(cleaned_lines):
|
||||
if PRICE_TAG_RE.search(line):
|
||||
continue
|
||||
match = DD_MM_YYYY_RE.search(line)
|
||||
if match:
|
||||
return pick(match, idx, line)
|
||||
|
||||
# 5) Any line — compact DDMMYYYY (skip likely SKU: same line has 8-digit product code context)
|
||||
for idx, line in enumerate(cleaned_lines):
|
||||
if PRICE_TAG_RE.search(line):
|
||||
continue
|
||||
for match in DDMMYYYY_RE.finditer(line):
|
||||
digits = f"{match.group(1)}{match.group(2)}{match.group(3)}"
|
||||
if is_valid_ddmmyyyy_digits(digits):
|
||||
# Skip if this 8-digit block is the only digits and looks like SKU on label top
|
||||
if re.search(r'\b\d{8}\b', line) and not line_has_exp_keyword(line):
|
||||
if re.search(r'(?:nugget|chicken|fiesta|champ|okey|akumo|frozen|gr)', line, re.I):
|
||||
continue
|
||||
return format_ddmmyyyy(digits), idx, text_lines[idx].strip()
|
||||
|
||||
# 6) Legacy patterns (slashes, month names, etc.)
|
||||
date_patterns = [
|
||||
r'\b\d{2}[-./]\d{2}[-./]\d{2,4}\b',
|
||||
r'\b\d{4}[-./]\d{2}[-./]\d{2}\b',
|
||||
r'\b\d{2}\s+(?:JAN|FEB|MAR|APR|MAY|JUN|JUL|AUG|SEP|OCT|NOV|DEC)[a-zA-Z]*\s+\d{2,4}\b',
|
||||
]
|
||||
for idx, line in enumerate(cleaned_lines):
|
||||
if not line_has_exp_keyword(line):
|
||||
continue
|
||||
for pat in date_patterns:
|
||||
match = re.search(pat, line, re.IGNORECASE)
|
||||
if match:
|
||||
return match.group(0).upper(), idx, text_lines[idx].strip()
|
||||
|
||||
return None, None, None
|
||||
|
||||
def line_contains_expired_date(line: str, expired_date: str) -> bool:
|
||||
if not line or not expired_date:
|
||||
return False
|
||||
digits_only = re.sub(r"\D", "", expired_date)
|
||||
line_digits = re.sub(r"\D", "", line)
|
||||
if len(digits_only) >= 6 and digits_only in line_digits:
|
||||
return True
|
||||
compact = expired_date.replace("/", "")
|
||||
return compact in line.replace(" ", "") or expired_date in line
|
||||
|
||||
def find_expired_crop_index(text_lines, expired_idx, expired_date, polys_len):
|
||||
"""Pick OCR box index for cropping; prefer the line that actually contains the date."""
|
||||
if not expired_date or polys_len <= 0:
|
||||
return None
|
||||
|
||||
if (
|
||||
expired_idx is not None
|
||||
and expired_idx < polys_len
|
||||
and expired_idx < len(text_lines)
|
||||
and line_contains_expired_date(text_lines[expired_idx], expired_date)
|
||||
):
|
||||
return expired_idx
|
||||
|
||||
keyword_match = None
|
||||
for idx, line in enumerate(text_lines):
|
||||
if idx >= polys_len:
|
||||
break
|
||||
if not line_contains_expired_date(line, expired_date):
|
||||
continue
|
||||
if line_has_exp_keyword(line):
|
||||
return idx
|
||||
if keyword_match is None:
|
||||
keyword_match = idx
|
||||
|
||||
if keyword_match is not None:
|
||||
return keyword_match
|
||||
|
||||
if expired_idx is not None and expired_idx < polys_len:
|
||||
return expired_idx
|
||||
return None
|
||||
@@ -1,85 +1,85 @@
|
||||
pipeline_name: PaddleOCR-VL-1.6
|
||||
|
||||
batch_size: 64
|
||||
|
||||
use_queues: True
|
||||
|
||||
use_doc_preprocessor: True
|
||||
use_layout_detection: True
|
||||
use_chart_recognition: False
|
||||
use_seal_recognition: False
|
||||
format_block_content: False
|
||||
merge_layout_blocks: True
|
||||
markdown_ignore_labels: []
|
||||
# - number
|
||||
# - footnote
|
||||
# - header
|
||||
# - header_image
|
||||
# - footer
|
||||
# - footer_image
|
||||
# - aside_text
|
||||
|
||||
SubModules:
|
||||
LayoutDetection:
|
||||
module_name: layout_detection
|
||||
model_name: PP-DocLayoutV3
|
||||
model_dir: null
|
||||
batch_size: 8
|
||||
threshold: 0.2
|
||||
layout_nms: True
|
||||
layout_unclip_ratio: [1.0, 1.0]
|
||||
layout_merge_bboxes_mode:
|
||||
0: "union"
|
||||
1: "union"
|
||||
2: "union"
|
||||
3: "large"
|
||||
4: "union"
|
||||
5: "large"
|
||||
6: "large"
|
||||
7: "union"
|
||||
8: "union"
|
||||
9: "union"
|
||||
10: "union"
|
||||
11: "union"
|
||||
12: "union"
|
||||
13: "union"
|
||||
14: "union"
|
||||
15: "large"
|
||||
16: "union"
|
||||
17: "large"
|
||||
18: "union"
|
||||
19: "union"
|
||||
20: "union"
|
||||
21: "union"
|
||||
22: "union"
|
||||
23: "union"
|
||||
24: "union"
|
||||
VLRecognition:
|
||||
module_name: vl_recognition
|
||||
model_name: PaddleOCR-VL-1.6-0.9B
|
||||
model_dir: null
|
||||
batch_size: 4096
|
||||
genai_config:
|
||||
backend: vllm-server
|
||||
server_url: http://127.0.0.1:8118/v1
|
||||
|
||||
SubPipelines:
|
||||
DocPreprocessor:
|
||||
pipeline_name: doc_preprocessor
|
||||
batch_size: 8
|
||||
use_doc_orientation_classify: True
|
||||
use_doc_unwarping: True
|
||||
SubModules:
|
||||
DocOrientationClassify:
|
||||
module_name: doc_text_orientation
|
||||
model_name: PP-LCNet_x1_0_doc_ori
|
||||
model_dir: null
|
||||
batch_size: 8
|
||||
DocUnwarping:
|
||||
module_name: image_unwarping
|
||||
model_name: UVDoc
|
||||
model_dir: null
|
||||
|
||||
Serving:
|
||||
extra:
|
||||
max_num_input_imgs: null
|
||||
pipeline_name: PaddleOCR-VL-1.6
|
||||
|
||||
batch_size: 64
|
||||
|
||||
use_queues: True
|
||||
|
||||
use_doc_preprocessor: True
|
||||
use_layout_detection: True
|
||||
use_chart_recognition: False
|
||||
use_seal_recognition: False
|
||||
format_block_content: False
|
||||
merge_layout_blocks: True
|
||||
markdown_ignore_labels: []
|
||||
# - number
|
||||
# - footnote
|
||||
# - header
|
||||
# - header_image
|
||||
# - footer
|
||||
# - footer_image
|
||||
# - aside_text
|
||||
|
||||
SubModules:
|
||||
LayoutDetection:
|
||||
module_name: layout_detection
|
||||
model_name: PP-DocLayoutV3
|
||||
model_dir: null
|
||||
batch_size: 8
|
||||
threshold: 0.2
|
||||
layout_nms: True
|
||||
layout_unclip_ratio: [1.0, 1.0]
|
||||
layout_merge_bboxes_mode:
|
||||
0: "union"
|
||||
1: "union"
|
||||
2: "union"
|
||||
3: "large"
|
||||
4: "union"
|
||||
5: "large"
|
||||
6: "large"
|
||||
7: "union"
|
||||
8: "union"
|
||||
9: "union"
|
||||
10: "union"
|
||||
11: "union"
|
||||
12: "union"
|
||||
13: "union"
|
||||
14: "union"
|
||||
15: "large"
|
||||
16: "union"
|
||||
17: "large"
|
||||
18: "union"
|
||||
19: "union"
|
||||
20: "union"
|
||||
21: "union"
|
||||
22: "union"
|
||||
23: "union"
|
||||
24: "union"
|
||||
VLRecognition:
|
||||
module_name: vl_recognition
|
||||
model_name: PaddleOCR-VL-1.6-0.9B
|
||||
model_dir: null
|
||||
batch_size: 4096
|
||||
genai_config:
|
||||
backend: vllm-server
|
||||
server_url: http://127.0.0.1:8118/v1
|
||||
|
||||
SubPipelines:
|
||||
DocPreprocessor:
|
||||
pipeline_name: doc_preprocessor
|
||||
batch_size: 8
|
||||
use_doc_orientation_classify: True
|
||||
use_doc_unwarping: True
|
||||
SubModules:
|
||||
DocOrientationClassify:
|
||||
module_name: doc_text_orientation
|
||||
model_name: PP-LCNet_x1_0_doc_ori
|
||||
model_dir: null
|
||||
batch_size: 8
|
||||
DocUnwarping:
|
||||
module_name: image_unwarping
|
||||
model_name: UVDoc
|
||||
model_dir: null
|
||||
|
||||
Serving:
|
||||
extra:
|
||||
max_num_input_imgs: null
|
||||
@@ -1,69 +1,69 @@
|
||||
import re
|
||||
|
||||
def clean_date_line(line: str) -> str:
|
||||
# 1) Replace "1)" with "0"
|
||||
cleaned = line.replace("1)", "0")
|
||||
|
||||
# 2) Replace "()" with "0"
|
||||
cleaned = cleaned.replace("()", "0")
|
||||
|
||||
# Clean BB misrecognitions (convert B8, 8B, 88 to BB when followed by digits)
|
||||
cleaned = re.sub(r'\b(?:B8|8B|88)(?=\d)', 'BB', cleaned)
|
||||
cleaned = re.sub(r'^(?:B8|8B|88)(?=\d)', 'BB', cleaned)
|
||||
|
||||
# Clean 012 month misrecognition (e.g. 020122027 -> 02022027)
|
||||
cleaned = re.sub(r'(?<!\d)(\d{1,2})012(20\d{2})(?!\d)', r'\g<1>02\g<2>', cleaned)
|
||||
cleaned = re.sub(r'(?<!\d)(\d{1,2})([-./\s]+)012([-./\s]+)(20\d{2})(?!\d)', r'\g<1>\g<2>02\g<3>\g<4>', cleaned)
|
||||
|
||||
# Clean 112 month misrecognition (e.g. 021122027 -> 02022027)
|
||||
cleaned = re.sub(r'(?<!\d)(\d{1,2})112(20\d{2})(?!\d)', r'\g<1>02\g<2>', cleaned)
|
||||
cleaned = re.sub(r'(?<!\d)(\d{1,2})([-./\s]+)112([-./\s]+)(20\d{2})(?!\d)', r'\g<1>\g<2>02\g<3>\g<4>', cleaned)
|
||||
|
||||
# Run contextual replacements
|
||||
for _ in range(3):
|
||||
# letter o/O flanked by digits or boundary -> 0
|
||||
cleaned = re.compile(r'(\d)[oO](\d|\b)').sub(r'\g<1>0\g<2>', cleaned)
|
||||
cleaned = re.compile(r'(\b|\d)[oO](\d)').sub(r'\g<1>0\g<2>', cleaned)
|
||||
|
||||
# letter I/i/l/| flanked by digits -> 1
|
||||
cleaned = re.compile(r'(\d)[Ii|l](\d|\b)').sub(r'\g<1>1\g<2>', cleaned)
|
||||
cleaned = re.compile(r'(\b|\d)[Ii|l](\d)').sub(r'\g<1>1\g<2>', cleaned)
|
||||
|
||||
# letter S/s flanked by digits -> 5
|
||||
cleaned = re.compile(r'(\d)[Ss](\d|\b)').sub(r'\g<1>5\g<2>', cleaned)
|
||||
cleaned = re.compile(r'(\b|\d)[Ss](\d)').sub(r'\g<1>5\g<2>', cleaned)
|
||||
|
||||
# letter Z/z flanked by digits -> 2
|
||||
cleaned = re.compile(r'(\d)[Zz](\d|\b)').sub(r'\g<1>2\g<2>', cleaned)
|
||||
cleaned = re.compile(r'(\b|\d)[Zz](\d)').sub(r'\g<1>2\g<2>', cleaned)
|
||||
|
||||
# letter B flanked by digits -> 8
|
||||
cleaned = re.compile(r'(\d)B(\d|\b)').sub(r'\g<1>8\g<2>', cleaned)
|
||||
cleaned = re.compile(r'(\b|\d)B(\d)').sub(r'\g<1>8\g<2>', cleaned)
|
||||
|
||||
return cleaned
|
||||
|
||||
test_cases = [
|
||||
"231)92026",
|
||||
"23o92026",
|
||||
"23O92026",
|
||||
"2309202l",
|
||||
"2309202I",
|
||||
"230920Z6",
|
||||
"230920s6",
|
||||
"2309202B",
|
||||
"BB 231)92026",
|
||||
"BB: 23()92026",
|
||||
"12010111",
|
||||
"B8021122027",
|
||||
"88021122027",
|
||||
"020122027",
|
||||
"BB 02/012/2027",
|
||||
"021122027",
|
||||
"BB 02/112/2027"
|
||||
]
|
||||
|
||||
for tc in test_cases:
|
||||
cleaned = clean_date_line(tc)
|
||||
print(f"Original: {tc:<18} -> Cleaned: {cleaned}")
|
||||
|
||||
import re
|
||||
|
||||
def clean_date_line(line: str) -> str:
|
||||
# 1) Replace "1)" with "0"
|
||||
cleaned = line.replace("1)", "0")
|
||||
|
||||
# 2) Replace "()" with "0"
|
||||
cleaned = cleaned.replace("()", "0")
|
||||
|
||||
# Clean BB misrecognitions (convert B8, 8B, 88 to BB when followed by digits)
|
||||
cleaned = re.sub(r'\b(?:B8|8B|88)(?=\d)', 'BB', cleaned)
|
||||
cleaned = re.sub(r'^(?:B8|8B|88)(?=\d)', 'BB', cleaned)
|
||||
|
||||
# Clean 012 month misrecognition (e.g. 020122027 -> 02022027)
|
||||
cleaned = re.sub(r'(?<!\d)(\d{1,2})012(20\d{2})(?!\d)', r'\g<1>02\g<2>', cleaned)
|
||||
cleaned = re.sub(r'(?<!\d)(\d{1,2})([-./\s]+)012([-./\s]+)(20\d{2})(?!\d)', r'\g<1>\g<2>02\g<3>\g<4>', cleaned)
|
||||
|
||||
# Clean 112 month misrecognition (e.g. 021122027 -> 02022027)
|
||||
cleaned = re.sub(r'(?<!\d)(\d{1,2})112(20\d{2})(?!\d)', r'\g<1>02\g<2>', cleaned)
|
||||
cleaned = re.sub(r'(?<!\d)(\d{1,2})([-./\s]+)112([-./\s]+)(20\d{2})(?!\d)', r'\g<1>\g<2>02\g<3>\g<4>', cleaned)
|
||||
|
||||
# Run contextual replacements
|
||||
for _ in range(3):
|
||||
# letter o/O flanked by digits or boundary -> 0
|
||||
cleaned = re.compile(r'(\d)[oO](\d|\b)').sub(r'\g<1>0\g<2>', cleaned)
|
||||
cleaned = re.compile(r'(\b|\d)[oO](\d)').sub(r'\g<1>0\g<2>', cleaned)
|
||||
|
||||
# letter I/i/l/| flanked by digits -> 1
|
||||
cleaned = re.compile(r'(\d)[Ii|l](\d|\b)').sub(r'\g<1>1\g<2>', cleaned)
|
||||
cleaned = re.compile(r'(\b|\d)[Ii|l](\d)').sub(r'\g<1>1\g<2>', cleaned)
|
||||
|
||||
# letter S/s flanked by digits -> 5
|
||||
cleaned = re.compile(r'(\d)[Ss](\d|\b)').sub(r'\g<1>5\g<2>', cleaned)
|
||||
cleaned = re.compile(r'(\b|\d)[Ss](\d)').sub(r'\g<1>5\g<2>', cleaned)
|
||||
|
||||
# letter Z/z flanked by digits -> 2
|
||||
cleaned = re.compile(r'(\d)[Zz](\d|\b)').sub(r'\g<1>2\g<2>', cleaned)
|
||||
cleaned = re.compile(r'(\b|\d)[Zz](\d)').sub(r'\g<1>2\g<2>', cleaned)
|
||||
|
||||
# letter B flanked by digits -> 8
|
||||
cleaned = re.compile(r'(\d)B(\d|\b)').sub(r'\g<1>8\g<2>', cleaned)
|
||||
cleaned = re.compile(r'(\b|\d)B(\d)').sub(r'\g<1>8\g<2>', cleaned)
|
||||
|
||||
return cleaned
|
||||
|
||||
test_cases = [
|
||||
"231)92026",
|
||||
"23o92026",
|
||||
"23O92026",
|
||||
"2309202l",
|
||||
"2309202I",
|
||||
"230920Z6",
|
||||
"230920s6",
|
||||
"2309202B",
|
||||
"BB 231)92026",
|
||||
"BB: 23()92026",
|
||||
"12010111",
|
||||
"B8021122027",
|
||||
"88021122027",
|
||||
"020122027",
|
||||
"BB 02/012/2027",
|
||||
"021122027",
|
||||
"BB 02/112/2027"
|
||||
]
|
||||
|
||||
for tc in test_cases:
|
||||
cleaned = clean_date_line(tc)
|
||||
print(f"Original: {tc:<18} -> Cleaned: {cleaned}")
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
# vLLM backend tuning for paddleocr genai_server
|
||||
# Docs: https://www.paddleocr.ai/latest/en/version3.x/pipeline_usage/PaddleOCR-VL.html#331-server-side-parameter-adjustment
|
||||
gpu-memory-utilization: 0.35
|
||||
max-num-seqs: 4
|
||||
enforce-eager: true
|
||||
max-model-len: 2048
|
||||
max-num-batched-tokens: 2048
|
||||
# vLLM backend tuning for paddleocr genai_server
|
||||
# Docs: https://www.paddleocr.ai/latest/en/version3.x/pipeline_usage/PaddleOCR-VL.html#331-server-side-parameter-adjustment
|
||||
gpu-memory-utilization: 0.35
|
||||
max-num-seqs: 4
|
||||
enforce-eager: true
|
||||
max-model-len: 2048
|
||||
max-num-batched-tokens: 2048
|
||||
@@ -1,8 +1,8 @@
|
||||
#!/bin/bash
|
||||
# Create the target directory inside the Next.js app
|
||||
mkdir -p pfm-web-app/public/do-pfm
|
||||
|
||||
# Copy DO-PFM images
|
||||
cp -v PaddleOCR-VL-1.6_Online_Demo/examples/do-pfm/*.jpg pfm-web-app/public/do-pfm/
|
||||
|
||||
echo "DO-PFM examples copied successfully!"
|
||||
#!/bin/bash
|
||||
# Create the target directory inside the Next.js app
|
||||
mkdir -p pfm-web-app/public/do-pfm
|
||||
|
||||
# Copy DO-PFM images
|
||||
cp -v PaddleOCR-VL-1.6_Online_Demo/examples/do-pfm/*.jpg pfm-web-app/public/do-pfm/
|
||||
|
||||
echo "DO-PFM examples copied successfully!"
|
||||
+110
-110
@@ -1,110 +1,110 @@
|
||||
# Database Entity Relationship Diagram (ERD)
|
||||
|
||||
This document describes the PostgreSQL database schema used to store OCR documents, parsed layout elements, inline cell edits, and row flagging status for the DO-PFM system.
|
||||
|
||||
## Relationship Diagram
|
||||
|
||||
```mermaid
|
||||
erDiagram
|
||||
documents {
|
||||
integer id PK "SERIAL"
|
||||
varchar filename UK "VARCHAR(255)"
|
||||
timestamp upload_time "TIMESTAMP"
|
||||
integer size "INTEGER"
|
||||
boolean parsed "BOOLEAN"
|
||||
jsonb metadata "JSONB"
|
||||
jsonb layout_parsing_result "JSONB"
|
||||
boolean is_sample "BOOLEAN"
|
||||
varchar file_hash "VARCHAR(64)"
|
||||
}
|
||||
|
||||
ocr_items {
|
||||
integer id PK "SERIAL"
|
||||
integer document_id FK "INTEGER"
|
||||
integer row_index "INTEGER"
|
||||
varchar kode_barang_original "VARCHAR(255)"
|
||||
varchar kode_barang "VARCHAR(255)"
|
||||
varchar nama_barang "VARCHAR(255)"
|
||||
varchar banyak_original "VARCHAR(255)"
|
||||
varchar banyak "VARCHAR(255)"
|
||||
varchar jumlah_original "VARCHAR(255)"
|
||||
varchar jumlah "VARCHAR(255)"
|
||||
boolean is_flagged "BOOLEAN"
|
||||
varchar remark "VARCHAR(1000)"
|
||||
}
|
||||
|
||||
documents ||--o{ ocr_items : "has"
|
||||
|
||||
vendors {
|
||||
integer id PK "SERIAL"
|
||||
varchar name UK "VARCHAR(255)"
|
||||
timestamp created_at "TIMESTAMP"
|
||||
}
|
||||
|
||||
customers {
|
||||
integer id PK "SERIAL"
|
||||
varchar name UK "VARCHAR(255)"
|
||||
timestamp created_at "TIMESTAMP"
|
||||
}
|
||||
```
|
||||
|
||||
## Schema Definitions
|
||||
|
||||
### 1. `documents` Table
|
||||
Stores parsed OCR files (both static sample pages and user-uploaded invoices/documents).
|
||||
|
||||
| Column | Type | Constraints | Description |
|
||||
|---|---|---|---|
|
||||
| `id` | `SERIAL` | `PRIMARY KEY` | Unique autoincrement ID of the document. |
|
||||
| `filename` | `VARCHAR(255)` | `UNIQUE`, `NOT NULL` | The unique name of the document file. |
|
||||
| `upload_time` | `TIMESTAMP` | `DEFAULT NOW()`, `NOT NULL` | The timestamp of the file upload. |
|
||||
| `size` | `INTEGER` | `DEFAULT 0`, `NOT NULL` | The file size in bytes. |
|
||||
| `parsed` | `BOOLEAN` | `DEFAULT FALSE`, `NOT NULL` | Indicates whether the document layout parsing has completed. |
|
||||
| `metadata` | `JSONB` | | Structured general metadata (Vendor, Customer, PO, SO, DO, etc.). |
|
||||
| `layout_parsing_result` | `JSONB` | | Raw layout parser response JSON from pipeline backend. |
|
||||
| `is_sample` | `BOOLEAN` | `DEFAULT FALSE`, `NOT NULL` | True if the file belongs to the pre-seeded static sample pages. |
|
||||
| `file_hash` | `VARCHAR(64)` | | SHA-256 hash of the document file contents. |
|
||||
|
||||
---
|
||||
|
||||
### 2. `ocr_items` Table
|
||||
Stores the extracted row items from tabular components of the document, supporting inline modifications and flagging details.
|
||||
|
||||
| Column | Type | Constraints | Description |
|
||||
|---|---|---|---|
|
||||
| `id` | `SERIAL` | `PRIMARY KEY` | Unique autoincrement ID of the item row. |
|
||||
| `document_id` | `INTEGER` | `REFERENCES documents(id) ON DELETE CASCADE`, `NOT NULL` | The associated document ID. |
|
||||
| `row_index` | `INTEGER` | `NOT NULL` | The index of the item row in the document table list (0-indexed). |
|
||||
| `kode_barang_original` | `VARCHAR(255)` | | The initial "Kode Barang" value extracted directly from OCR. |
|
||||
| `kode_barang` | `VARCHAR(255)` | | The edited/current "Kode Barang" value. |
|
||||
| `nama_barang` | `VARCHAR(255)` | | The "Nama Barang" value (read-only reference). |
|
||||
| `banyak_original` | `VARCHAR(255)` | | The initial "Banyak" value extracted from OCR. |
|
||||
| `banyak` | `VARCHAR(255)` | | The edited/current "Banyak" value. |
|
||||
| `jumlah_original` | `VARCHAR(255)` | | The initial "Jumlah" value extracted from OCR. |
|
||||
| `jumlah` | `VARCHAR(255)` | | The edited/current "Jumlah" value. |
|
||||
| `is_flagged` | `BOOLEAN` | `DEFAULT FALSE`, `NOT NULL` | True if the line item is flagged/strikethrough ("dicoret"). |
|
||||
| `remark` | `VARCHAR(1000)` | | Custom notes/remarks provided for flagging. |
|
||||
|
||||
* **Unique Constraints**: A unique index on `(document_id, row_index)` prevents duplicate indexes for the same page.
|
||||
|
||||
---
|
||||
|
||||
### 3. `vendors` Table
|
||||
Stores the Vendor Master registry.
|
||||
|
||||
| Column | Type | Constraints | Description |
|
||||
|---|---|---|---|
|
||||
| `id` | `SERIAL` | `PRIMARY KEY` | Unique autoincrement ID of the vendor. |
|
||||
| `name` | `VARCHAR(255)` | `UNIQUE`, `NOT NULL` | The unique name of the vendor (e.g. including kawasan/address). |
|
||||
| `created_at` | `TIMESTAMP` | `DEFAULT NOW()`, `NOT NULL` | The registration timestamp. |
|
||||
|
||||
---
|
||||
|
||||
### 4. `customers` Table
|
||||
Stores the Customer Master registry.
|
||||
|
||||
| Column | Type | Constraints | Description |
|
||||
|---|---|---|---|
|
||||
| `id` | `SERIAL` | `PRIMARY KEY` | Unique autoincrement ID of the customer. |
|
||||
| `name` | `VARCHAR(255)` | `UNIQUE`, `NOT NULL` | The unique name of the customer (e.g. including branch/address). |
|
||||
| `created_at` | `TIMESTAMP` | `DEFAULT NOW()`, `NOT NULL` | The registration timestamp. |
|
||||
# Database Entity Relationship Diagram (ERD)
|
||||
|
||||
This document describes the PostgreSQL database schema used to store OCR documents, parsed layout elements, inline cell edits, and row flagging status for the DO-PFM system.
|
||||
|
||||
## Relationship Diagram
|
||||
|
||||
```mermaid
|
||||
erDiagram
|
||||
documents {
|
||||
integer id PK "SERIAL"
|
||||
varchar filename UK "VARCHAR(255)"
|
||||
timestamp upload_time "TIMESTAMP"
|
||||
integer size "INTEGER"
|
||||
boolean parsed "BOOLEAN"
|
||||
jsonb metadata "JSONB"
|
||||
jsonb layout_parsing_result "JSONB"
|
||||
boolean is_sample "BOOLEAN"
|
||||
varchar file_hash "VARCHAR(64)"
|
||||
}
|
||||
|
||||
ocr_items {
|
||||
integer id PK "SERIAL"
|
||||
integer document_id FK "INTEGER"
|
||||
integer row_index "INTEGER"
|
||||
varchar kode_barang_original "VARCHAR(255)"
|
||||
varchar kode_barang "VARCHAR(255)"
|
||||
varchar nama_barang "VARCHAR(255)"
|
||||
varchar banyak_original "VARCHAR(255)"
|
||||
varchar banyak "VARCHAR(255)"
|
||||
varchar jumlah_original "VARCHAR(255)"
|
||||
varchar jumlah "VARCHAR(255)"
|
||||
boolean is_flagged "BOOLEAN"
|
||||
varchar remark "VARCHAR(1000)"
|
||||
}
|
||||
|
||||
documents ||--o{ ocr_items : "has"
|
||||
|
||||
vendors {
|
||||
integer id PK "SERIAL"
|
||||
varchar name UK "VARCHAR(255)"
|
||||
timestamp created_at "TIMESTAMP"
|
||||
}
|
||||
|
||||
customers {
|
||||
integer id PK "SERIAL"
|
||||
varchar name UK "VARCHAR(255)"
|
||||
timestamp created_at "TIMESTAMP"
|
||||
}
|
||||
```
|
||||
|
||||
## Schema Definitions
|
||||
|
||||
### 1. `documents` Table
|
||||
Stores parsed OCR files (both static sample pages and user-uploaded invoices/documents).
|
||||
|
||||
| Column | Type | Constraints | Description |
|
||||
|---|---|---|---|
|
||||
| `id` | `SERIAL` | `PRIMARY KEY` | Unique autoincrement ID of the document. |
|
||||
| `filename` | `VARCHAR(255)` | `UNIQUE`, `NOT NULL` | The unique name of the document file. |
|
||||
| `upload_time` | `TIMESTAMP` | `DEFAULT NOW()`, `NOT NULL` | The timestamp of the file upload. |
|
||||
| `size` | `INTEGER` | `DEFAULT 0`, `NOT NULL` | The file size in bytes. |
|
||||
| `parsed` | `BOOLEAN` | `DEFAULT FALSE`, `NOT NULL` | Indicates whether the document layout parsing has completed. |
|
||||
| `metadata` | `JSONB` | | Structured general metadata (Vendor, Customer, PO, SO, DO, etc.). |
|
||||
| `layout_parsing_result` | `JSONB` | | Raw layout parser response JSON from pipeline backend. |
|
||||
| `is_sample` | `BOOLEAN` | `DEFAULT FALSE`, `NOT NULL` | True if the file belongs to the pre-seeded static sample pages. |
|
||||
| `file_hash` | `VARCHAR(64)` | | SHA-256 hash of the document file contents. |
|
||||
|
||||
---
|
||||
|
||||
### 2. `ocr_items` Table
|
||||
Stores the extracted row items from tabular components of the document, supporting inline modifications and flagging details.
|
||||
|
||||
| Column | Type | Constraints | Description |
|
||||
|---|---|---|---|
|
||||
| `id` | `SERIAL` | `PRIMARY KEY` | Unique autoincrement ID of the item row. |
|
||||
| `document_id` | `INTEGER` | `REFERENCES documents(id) ON DELETE CASCADE`, `NOT NULL` | The associated document ID. |
|
||||
| `row_index` | `INTEGER` | `NOT NULL` | The index of the item row in the document table list (0-indexed). |
|
||||
| `kode_barang_original` | `VARCHAR(255)` | | The initial "Kode Barang" value extracted directly from OCR. |
|
||||
| `kode_barang` | `VARCHAR(255)` | | The edited/current "Kode Barang" value. |
|
||||
| `nama_barang` | `VARCHAR(255)` | | The "Nama Barang" value (read-only reference). |
|
||||
| `banyak_original` | `VARCHAR(255)` | | The initial "Banyak" value extracted from OCR. |
|
||||
| `banyak` | `VARCHAR(255)` | | The edited/current "Banyak" value. |
|
||||
| `jumlah_original` | `VARCHAR(255)` | | The initial "Jumlah" value extracted from OCR. |
|
||||
| `jumlah` | `VARCHAR(255)` | | The edited/current "Jumlah" value. |
|
||||
| `is_flagged` | `BOOLEAN` | `DEFAULT FALSE`, `NOT NULL` | True if the line item is flagged/strikethrough ("dicoret"). |
|
||||
| `remark` | `VARCHAR(1000)` | | Custom notes/remarks provided for flagging. |
|
||||
|
||||
* **Unique Constraints**: A unique index on `(document_id, row_index)` prevents duplicate indexes for the same page.
|
||||
|
||||
---
|
||||
|
||||
### 3. `vendors` Table
|
||||
Stores the Vendor Master registry.
|
||||
|
||||
| Column | Type | Constraints | Description |
|
||||
|---|---|---|---|
|
||||
| `id` | `SERIAL` | `PRIMARY KEY` | Unique autoincrement ID of the vendor. |
|
||||
| `name` | `VARCHAR(255)` | `UNIQUE`, `NOT NULL` | The unique name of the vendor (e.g. including kawasan/address). |
|
||||
| `created_at` | `TIMESTAMP` | `DEFAULT NOW()`, `NOT NULL` | The registration timestamp. |
|
||||
|
||||
---
|
||||
|
||||
### 4. `customers` Table
|
||||
Stores the Customer Master registry.
|
||||
|
||||
| Column | Type | Constraints | Description |
|
||||
|---|---|---|---|
|
||||
| `id` | `SERIAL` | `PRIMARY KEY` | Unique autoincrement ID of the customer. |
|
||||
| `name` | `VARCHAR(255)` | `UNIQUE`, `NOT NULL` | The unique name of the customer (e.g. including branch/address). |
|
||||
| `created_at` | `TIMESTAMP` | `DEFAULT NOW()`, `NOT NULL` | The registration timestamp. |
|
||||
@@ -1,29 +1,29 @@
|
||||
-- Migration: 001_init_schema
|
||||
-- Description: Initialize schema for documents and ocr_items
|
||||
|
||||
CREATE TABLE IF NOT EXISTS documents (
|
||||
id SERIAL PRIMARY KEY,
|
||||
filename VARCHAR(255) UNIQUE NOT NULL,
|
||||
upload_time TIMESTAMP NOT NULL DEFAULT NOW(),
|
||||
size INTEGER NOT NULL DEFAULT 0,
|
||||
parsed BOOLEAN NOT NULL DEFAULT FALSE,
|
||||
metadata JSONB,
|
||||
layout_parsing_result JSONB,
|
||||
is_sample BOOLEAN NOT NULL DEFAULT FALSE
|
||||
);
|
||||
|
||||
CREATE TABLE IF NOT EXISTS ocr_items (
|
||||
id SERIAL PRIMARY KEY,
|
||||
document_id INTEGER NOT NULL REFERENCES documents(id) ON DELETE CASCADE,
|
||||
row_index INTEGER NOT NULL,
|
||||
kode_barang_original VARCHAR(255),
|
||||
kode_barang VARCHAR(255),
|
||||
nama_barang VARCHAR(255),
|
||||
banyak_original VARCHAR(255),
|
||||
banyak VARCHAR(255),
|
||||
jumlah_original VARCHAR(255),
|
||||
jumlah VARCHAR(255),
|
||||
is_flagged BOOLEAN NOT NULL DEFAULT FALSE,
|
||||
remark VARCHAR(1000),
|
||||
UNIQUE(document_id, row_index)
|
||||
);
|
||||
-- Migration: 001_init_schema
|
||||
-- Description: Initialize schema for documents and ocr_items
|
||||
|
||||
CREATE TABLE IF NOT EXISTS documents (
|
||||
id SERIAL PRIMARY KEY,
|
||||
filename VARCHAR(255) UNIQUE NOT NULL,
|
||||
upload_time TIMESTAMP NOT NULL DEFAULT NOW(),
|
||||
size INTEGER NOT NULL DEFAULT 0,
|
||||
parsed BOOLEAN NOT NULL DEFAULT FALSE,
|
||||
metadata JSONB,
|
||||
layout_parsing_result JSONB,
|
||||
is_sample BOOLEAN NOT NULL DEFAULT FALSE
|
||||
);
|
||||
|
||||
CREATE TABLE IF NOT EXISTS ocr_items (
|
||||
id SERIAL PRIMARY KEY,
|
||||
document_id INTEGER NOT NULL REFERENCES documents(id) ON DELETE CASCADE,
|
||||
row_index INTEGER NOT NULL,
|
||||
kode_barang_original VARCHAR(255),
|
||||
kode_barang VARCHAR(255),
|
||||
nama_barang VARCHAR(255),
|
||||
banyak_original VARCHAR(255),
|
||||
banyak VARCHAR(255),
|
||||
jumlah_original VARCHAR(255),
|
||||
jumlah VARCHAR(255),
|
||||
is_flagged BOOLEAN NOT NULL DEFAULT FALSE,
|
||||
remark VARCHAR(1000),
|
||||
UNIQUE(document_id, row_index)
|
||||
);
|
||||
@@ -1,4 +1,4 @@
|
||||
-- Migration: 002_add_file_hash
|
||||
-- Description: Add file_hash column to documents table for duplicate content detection
|
||||
|
||||
ALTER TABLE documents ADD COLUMN IF NOT EXISTS file_hash VARCHAR(64);
|
||||
-- Migration: 002_add_file_hash
|
||||
-- Description: Add file_hash column to documents table for duplicate content detection
|
||||
|
||||
ALTER TABLE documents ADD COLUMN IF NOT EXISTS file_hash VARCHAR(64);
|
||||
@@ -1,12 +1,12 @@
|
||||
-- Migration: 003_create_vendor_master
|
||||
-- Description: Create vendors table and seed the initial vendor entry
|
||||
|
||||
CREATE TABLE IF NOT EXISTS vendors (
|
||||
id SERIAL PRIMARY KEY,
|
||||
name VARCHAR(255) UNIQUE NOT NULL,
|
||||
created_at TIMESTAMP NOT NULL DEFAULT NOW()
|
||||
);
|
||||
|
||||
INSERT INTO vendors (name)
|
||||
VALUES ('PT. CHAROEN POKPHAND INDONESIA Tbk KAWASAN INDUSTRI MODERN, BANTEN')
|
||||
ON CONFLICT (name) DO NOTHING;
|
||||
-- Migration: 003_create_vendor_master
|
||||
-- Description: Create vendors table and seed the initial vendor entry
|
||||
|
||||
CREATE TABLE IF NOT EXISTS vendors (
|
||||
id SERIAL PRIMARY KEY,
|
||||
name VARCHAR(255) UNIQUE NOT NULL,
|
||||
created_at TIMESTAMP NOT NULL DEFAULT NOW()
|
||||
);
|
||||
|
||||
INSERT INTO vendors (name)
|
||||
VALUES ('PT. CHAROEN POKPHAND INDONESIA Tbk KAWASAN INDUSTRI MODERN, BANTEN')
|
||||
ON CONFLICT (name) DO NOTHING;
|
||||
@@ -1,12 +1,12 @@
|
||||
-- Migration: 004_create_customer_master
|
||||
-- Description: Create customers table and seed the initial customer entry
|
||||
|
||||
CREATE TABLE IF NOT EXISTS customers (
|
||||
id SERIAL PRIMARY KEY,
|
||||
name VARCHAR(255) UNIQUE NOT NULL,
|
||||
created_at TIMESTAMP NOT NULL DEFAULT NOW()
|
||||
);
|
||||
|
||||
INSERT INTO customers (name)
|
||||
VALUES ('PT.PRIMAFOOD INTERNATIONAL, JL. ANCOL BARAT VIII/1, ANCOL, PADEMANGAN, JAKARTA UTARA, 14430')
|
||||
ON CONFLICT (name) DO NOTHING;
|
||||
-- Migration: 004_create_customer_master
|
||||
-- Description: Create customers table and seed the initial customer entry
|
||||
|
||||
CREATE TABLE IF NOT EXISTS customers (
|
||||
id SERIAL PRIMARY KEY,
|
||||
name VARCHAR(255) UNIQUE NOT NULL,
|
||||
created_at TIMESTAMP NOT NULL DEFAULT NOW()
|
||||
);
|
||||
|
||||
INSERT INTO customers (name)
|
||||
VALUES ('PT.PRIMAFOOD INTERNATIONAL, JL. ANCOL BARAT VIII/1, ANCOL, PADEMANGAN, JAKARTA UTARA, 14430')
|
||||
ON CONFLICT (name) DO NOTHING;
|
||||
@@ -1,244 +1,244 @@
|
||||
-- Migration: 005_create_sku_master
|
||||
-- Description: Create sku_master table and seed the initial SKU entries
|
||||
|
||||
CREATE TABLE IF NOT EXISTS sku_master (
|
||||
id SERIAL PRIMARY KEY,
|
||||
no_sku VARCHAR(255) UNIQUE NOT NULL,
|
||||
nama_item VARCHAR(255) NOT NULL,
|
||||
created_at TIMESTAMP NOT NULL DEFAULT NOW()
|
||||
);
|
||||
|
||||
INSERT INTO sku_master (no_sku, nama_item) VALUES
|
||||
('11048006', 'BEBEK PARTING-NEW(*)'),
|
||||
('11110059', 'CEKER BERKUKU FROZEN PACK 1 KG(*)'),
|
||||
('11110074', 'CEKER 1 KG FROZEN (20 PAC/KARUNG)(*)'),
|
||||
('11140051', 'AMPELA FROZEN PACK 1 KG(*)'),
|
||||
('11140062', 'AMPELA 1 KG FROZEN (20 PAC/KARUNG)(*)'),
|
||||
('11148002', 'AMPELA BEBEK FROZEN 1 KG/PACK (NEW)(*)'),
|
||||
('11150052', 'HATI FROZEN PACK 1 KG(*)'),
|
||||
('11150055', 'JANTUNG FROZEN PACK 1 KG(*)'),
|
||||
('11150064', 'HATI 1 KG FROZEN (20 PAC/KARUNG)(*)'),
|
||||
('11150065', 'JANTUNG 1 KG FROZEN (20 PAC/KARUNG)(*)'),
|
||||
('11310012', 'AYAM SIZE 0 (0.6-0.7)KG(*)'),
|
||||
('11310013', 'AYAM SIZE1 FROZEN (0.75-0.8) KG(*)'),
|
||||
('11310014', 'AYAM SIZE 2 FROZEN (0.8-0.9)KG(*)'),
|
||||
('11310016', 'AYAM SIZE Z PR FROZ(*)'),
|
||||
('11310017', 'AYAM SIZE 0 PR FROZEN(*)'),
|
||||
('11310018', 'AYAM SIZE 1 PR FROZEN(*)'),
|
||||
('11310019', 'AYAM SIZE 2 PR FROZEN(*)'),
|
||||
('11310021', 'AYAM SIZE BESAR (B) FROZ (1-1.1)KG/PC(*)'),
|
||||
('11310022', 'AYAM SIZE A PR (0.9-1)KG/PC(*)'),
|
||||
('11310024', 'AYAM SIZE A FROZEN (0.9-1)KG/PC(*)'),
|
||||
('11310025', 'AYAM SIZE SUPER (C) FROZ(1.1-1.2)KG/ PC(*)'),
|
||||
('11310026', 'AYAM SIZE JUMBO (D) FROZ (1.2- 1.3)KG/PC(*)'),
|
||||
('11318301', 'BEBEK MUDA-BD1(1.0-1.1 KG)-NEW(*)'),
|
||||
('11318306', 'CP DUCK PEKING 1.5-1.6 KG/PC(*)'),
|
||||
('11318308', 'BEBEK PEKING SPR BD5(1.7 -1.8 )Kg-NEW(*)'),
|
||||
('11410043', 'PARTING 10 SIZE D FRESH BENSU 1.25 KG/PAC(*)'),
|
||||
('11420055', 'PARTING 12 ALL SIZE FROZ/PAC(*)'),
|
||||
('11600053', 'BONELESS LEG FROZEN 1 KG(*)'),
|
||||
('11620056', 'SBL (FILLET PAHA) 1 KG(*)'),
|
||||
('11640053', 'PAHA UTUH (1 KG)(*)'),
|
||||
('11650053', 'PAHA ATAS 1 KG(*)'),
|
||||
('11660050', 'PAHA BAWAH (1 KG)(*)'),
|
||||
('11690053', 'SBB (FILLET DADA )1 KG(*)'),
|
||||
('11690081', 'SBB JUMBO FZ (2.0 - 2.2 KG/PAC)(*)'),
|
||||
('11710051', 'DADA UTUH (1 KG)(*)'),
|
||||
('11720055', 'FULL WING FROZ PACK 1 KG(*)'),
|
||||
('11730050', 'MIDDLE WING FROZ PACK 1 KG(*)'),
|
||||
('11750050', 'FILLET MITRA 1 KG(*)'),
|
||||
('11818300', 'CP-BEBEK GORENG 400GR/PAC'),
|
||||
('11840002', 'AYAM JANTAN BKKL SZ 0 (600-700) GR/PC(*)'),
|
||||
('11959937', 'SATE AYAM FRESHMART 360 GR (PAC)'),
|
||||
('12010111', 'FIESTA CRISPY BUBBLE 400 GR/PAC'),
|
||||
('12010112', 'FIESTA CHICKEN NUGGET 400 GR/PAC'),
|
||||
('12010113', 'FIESTA CHICKEN NUGGET 200 GR/PAC'),
|
||||
('12010115', 'FIESTA NUGGET ZOO 400 GR/PAC'),
|
||||
('12010116', 'FIESTA NUGGET DINO 400 GR/PAC'),
|
||||
('12010117', 'FIESTA NUGGET HAPPY STAR 400 GR/PAC'),
|
||||
('12010119', 'FIESTA NUGGET CHEESE 123 400 GR/PAC'),
|
||||
('12010121', 'FIESTA NUGGET PIZZABC 400 GR/PAC'),
|
||||
('12010122', 'FIESTA CHEESY LOVER 400 GR/PAC'),
|
||||
('12010123', 'FIESTA GARLIC CHEESE 400 GR/PAC'),
|
||||
('12010124', 'FIESTA CHEESY CHIC W/BROCCOLI 400 GR/PAC'),
|
||||
('12010127', 'FIESTA SPICY NUGGET 400 GR/PAC'),
|
||||
('12010128', 'FIESTA VOLCANO CHEESE 400 GR/PAC'),
|
||||
('12010129', 'FIESTA CHEESY BOMBS CHICKEN NUGGET 400 GR'),
|
||||
('12010402', 'GOLDEN FIESTA NUGGET W/PINEAPPLE SAUCE 500 GR'),
|
||||
('12010509', 'CHAMP CRUNCHY NUGGET 450 GR/PAC'),
|
||||
('12010510', 'CHAMP NUGGET AYAM 225 GR/PAC'),
|
||||
('12010511', 'CHAMP NUGGET AYAM 450 GR/PAC'),
|
||||
('12010512', 'CHAMP NUGGET AYAM 900 GR/PAC'),
|
||||
('12010513', 'CHAMP NUGGET ABC KOMBINASI 225 GR/PAC'),
|
||||
('12010514', 'CHAMP NUGGET ABC KOMBINASI 450 GR/PAC'),
|
||||
('12010515', 'CHAMP KOIN KOMBINASI 450 GR/PAC'),
|
||||
('12010516', 'CHAMP KOIN KOMBINASI 200 GR/PAC'),
|
||||
('12010517', 'CHAMP NUGGET STICK 225 GR/PAC'),
|
||||
('12010518', 'CHAMP NUGGET STICK 450 GR/PAC'),
|
||||
('12010519', 'CHAMP NUGGET STICK 900 GR/PAC'),
|
||||
('12010520', 'CHAMP CHICKEN NUGGET BENTUK 123 450 GR/PAC'),
|
||||
('12010521', 'CHAMP NUGGET HOTZZ LEVEL 5 450 GR/PAC'),
|
||||
('12010606', 'CHAMP CRUNCHY NUGGET 225 GR/PAC'),
|
||||
('12010707', 'CHAMP MITRA NUGGET COIN 200 GR (NEW)'),
|
||||
('12010801', 'OKEY NUGGET 500GR'),
|
||||
('12012201', 'ASIMO NUGGET KOMBINASI 500 GR/PAC'),
|
||||
('12012202', 'ASIMO NUGGET KOMBINASI 1 KG/PAC'),
|
||||
('12012203', 'ASIMO NUGGET KOMBINASI 250 GR/PAC'),
|
||||
('12012501', 'AKUMO CHICKEN NAGET 250 GR'),
|
||||
('12012502', 'AKUMO CHICKEN NUGGET 500 GR'),
|
||||
('12012503', 'AKUMO CHICKEN NUGGET 1000 GR'),
|
||||
('12012504', 'AKUMO COIN 200 GR/PAC'),
|
||||
('12012505', 'AKUMO KOIN 400 GR/PAC'),
|
||||
('12020102', 'FIESTA SPICY WING 400 GR/PAC'),
|
||||
('12020401', 'GOLDEN FIESTA SP WING 500 GR'),
|
||||
('12030101', 'FIESTA STIKIE 400 GR/PAC'),
|
||||
('12030102', 'FIESTA STIKIE 200 GR/PAC'),
|
||||
('12030403', 'GOLDEN FIESTA STIKIE W/ SWEET CHILLI SAUCE 500GR'),
|
||||
('12030801', 'OKEY STICK 1000 GR'),
|
||||
('12030802', 'OKEY STICK 500GR'),
|
||||
('12032201', 'ASIMO STICK KOMBINASI 500 GR/PAC'),
|
||||
('12032202', 'ASIMO STICK KOMBINASI 1000 GR/PAC'),
|
||||
('12032203', 'ASIMO STIK KOMBINASI 250 GR/PAC'),
|
||||
('12032501', 'AKUMO CHICKEN STICK 250 GR'),
|
||||
('12032502', 'AKUMO CHICKEN STIK 500 GR'),
|
||||
('12032503', 'AKUMO CHICKEN STICK 1000 GR'),
|
||||
('12040101', 'FIESTA SCHNITZEL 400 GR/PAC'),
|
||||
('12040102', 'FIESTA CRISPY BUBBLE KATSU 400 GR/PAC'),
|
||||
('12040404', 'GOLDEN FIESTA CORDON BLEU BBQ SAUCE 500 GR'),
|
||||
('12040406', 'GOLDEN FIESTA KATSU W/CHEESE SAUCE 500 GR/PAC'),
|
||||
('12050103', 'FIESTA FRIED CHICKEN 400 GR/PAC'),
|
||||
('12050104', 'FIESTA HOT & CRISPY FRIED CHICKEN 400 GR/PAC'),
|
||||
('12050401', 'GOLDEN FIESTA CRISPY WING W/SP GLAZ SC 500 GR/PAC'),
|
||||
('12060103', 'FIESTA KARAGE 200 GR/PAC'),
|
||||
('12060104', 'FIESTA KARAGE 400 GR/PAC'),
|
||||
('12060105', 'FIESTA SPICY KARAGE 400 GR/PAC'),
|
||||
('12060402', 'GOLDEN FIESTA KARAGE CHILI SAUCE 500GR'),
|
||||
('12070101', 'FIESTA POK-POK 400 GR/PAC (NEW)'),
|
||||
('12080101', 'FIESTA SPICY CHICK 400 GR/PAC'),
|
||||
('12130102', 'FIESTA CRISPY BURGER 360 GR (NEW)'),
|
||||
('12130504', 'CHAMP BURGER 315 GR (NEW)'),
|
||||
('12140105', 'FIESTA CHICK TOFU 400 GR/PAC'),
|
||||
('12150201', 'FIESTA DS CRISPY CRUNCH 300 GR/PAC'),
|
||||
('12150501', 'CHAMP CRUNCHY HOTZZ 300 GR/PAC'),
|
||||
('12190103', 'FIESTA DELISTRIPE 400 GR/PAC'),
|
||||
('12240102', 'FIESTA CHEESY ITALIAN R/BITES 400 GR/PAC'),
|
||||
('12240103', 'FIESTA YAKINIKU R/BITES 400 GR/PAC'),
|
||||
('13010101', 'FIESTA CHICK SSG 300 GR'),
|
||||
('13010102', 'FIESTA CHICK SSG 500 GR'),
|
||||
('13010103', 'FIESTA CHICK SSG 200 GR/PAC'),
|
||||
('13010111', 'FIESTA SOSIS BRATWURST 300 GR'),
|
||||
('13010112', 'FIESTA CHEESE SSG 300 GR'),
|
||||
('13010113', 'FIESTA SOSIS CURRYWURST 300 GR'),
|
||||
('13010114', 'FIESTA SSG BOCKWURST 300GR'),
|
||||
('13010115', 'FIESTA SSG WIENER 300GR'),
|
||||
('13010116', 'FIESTA SSG ORIGINAL 300 GR'),
|
||||
('13010117', 'FIESTA SSG FRANKFURTER 300GR'),
|
||||
('13010118', 'FIESTA RTG SSG 65 GR/PAC'),
|
||||
('13010119', 'FIESTA RTG C/SPICY KOREAN 60 GR/PAC'),
|
||||
('13010120', 'FIESTA RTG C/CHEESY MELTS 65 GR/PAC'),
|
||||
('13010122', 'FIESTA RTG SAUSAGE WITH HOT LAVA 60G'),
|
||||
('13010123', 'FIESTA RTG SAUSAGE WITH CHEESE LAVA 60G'),
|
||||
('13010124', 'FIESTA RTG SAUSAGE HICKORY SAUCE 60GR'),
|
||||
('13010125', 'FIESTA RTG SAUSAGE WITH MENTAI LAVA 60GR'),
|
||||
('13010510', 'CHAMP CHICK SSG 75 GR'),
|
||||
('13010513', 'CHAMP CHICK SSG 375 GR'),
|
||||
('13010514', 'CHAMP CHICK SSG 1000 GR'),
|
||||
('13010518', 'CHAMP SSG JUMBO BAKAR 500 GR/PAC'),
|
||||
('13010519', 'CHAMP SSG BAKAR MINI 500 GR/PAC-INACT'),
|
||||
('13010521', 'CHAMP CHICK SSG 150 GR/PAC (NEW)'),
|
||||
('13010523', 'CHAMP CHICK SSG AYAM MADU 300 GR/PAC'),
|
||||
('13010524', 'CHAMP SSG JUMBO BAKAR 500 GR/PAC (NEW)'),
|
||||
('13010525', 'CHAMP SSG BAKAR MINI 500 GR/PAC (NEW)'),
|
||||
('13010809', 'OKEY CHICK SSG 500GR-INACT'),
|
||||
('13010815', 'OKEY SSG BAKAR JUMBO 500 GR/PAC (NEW)'),
|
||||
('13010816', 'OKEY SSG BAKAR MINI 500 GR/PAC (NEW)'),
|
||||
('13010817', 'OKEY SSG CHICK KOMBINASI 500 GR (EXTRA1)'),
|
||||
('13010818', 'OKEY SSG CHICK KOMBINASI 1 KG (EXTR$A2)'),
|
||||
('13012205', 'ASIMO SOSIS AYAM KOMBINASI 375 GR (PAC)'),
|
||||
('13012206', 'ASIMO SOSIS AYAM KOMBINASI 500 GR'),
|
||||
('13012207', 'ASIMO SOSIS AYAM KOMBINASI 750 GR'),
|
||||
('13012208', 'ASIMO SOSIS AYAM KOMBINASI 1000 GR'),
|
||||
('13012209', 'ASIMO SSG CHICK KOMBINASI 500 GR (EXTRA1)'),
|
||||
('13012210', 'ASIMO SSG CHICK KOMBINASI 1 KG (EXTR$A2)'),
|
||||
('13030101', 'FIESTA CHICK MEAT BALL 300 GR'),
|
||||
('13030102', 'FIESTA CHICK MEATBALL 500 GR'),
|
||||
('13030501', 'CHAMP CHICK MEATBALL 200 GR'),
|
||||
('13030502', 'CHAMP CHICK MEATBALL 500 GR'),
|
||||
('13050101', 'FIESTA SCB 250 GR'),
|
||||
('13050105', 'FIESTA CHICKEN SLICE 300 GR'),
|
||||
('13050106', 'FIESTA BEEF SLICE 300 GR'),
|
||||
('13070501', 'CHAMP BEEF SSG SERBAGUNA 150 GR'),
|
||||
('13070502', 'CHAMP BEEF SSG SERBAGUNA 375GR'),
|
||||
('13070505', 'CHAMP BEEF SSG GORENG 375 GR'),
|
||||
('13070506', 'CHAMP FRANKFURTER SSG 375GR'),
|
||||
('13100512', 'CHAMP CHICK SSG S/SANTAP ORIG 546GR (CAN)'),
|
||||
('13110504', 'CHAMP BEEF BALL 500GR'),
|
||||
('13170510', 'CHAMP BEEF BBQ SSG S/SANTAP 546GR (CAN)'),
|
||||
('15010101', 'FIESTA SHOESTRING 500 GR'),
|
||||
('15010102', 'FIESTA SHOESTRING 1000 GR'),
|
||||
('15010107', 'FIESTA FRENCH F SHOESTRING INSTITUSI 2KG'),
|
||||
('15020101', 'FIESTA STRAIGHT CUT 500 GR'),
|
||||
('15020102', 'FIESTA STRAIGHT CUT 1000 GR'),
|
||||
('15030101', 'FIESTA CRINKLE CUT 500 GR'),
|
||||
('15030102', 'FIESTA CRINKLE CUT 1000 GR'),
|
||||
('15040101', 'FIESTA BATTER COATED 500 GR'),
|
||||
('15040102', 'FIESTA BATTER COATED 1000 GR'),
|
||||
('16060103', 'FIESTA CHICK SIOMAY 900GR'),
|
||||
('16060113', 'FIESTA CHICK SIOMAY 180GR (NEW)'),
|
||||
('16060114', 'FIESTA GYOZA 180 GR (NEW)'),
|
||||
('16060119', 'FIESTA RTG SIOMAY 54 GR/PAC'),
|
||||
('16060120', 'FIESTA KEECHO 400 GR/PAC'),
|
||||
('16060121', 'FIESTA CHICKEN TOFU 400 GR/PAC (NEW)'),
|
||||
('16060503', 'CHAMP CHICK&FISH SIOMAY 180 GR (NEW)'),
|
||||
('17200109', 'FIESTA RTS C/TERIYAKI 300GR/PAC'),
|
||||
('17200110', 'FIESTA RTS C/RENDANG 300GR/PAC'),
|
||||
('17200111', 'FIESTA RTS C/W RUJAK SC 300GR/PAC'),
|
||||
('17200112', 'FIESTA RTS C/W S/MATAH 300GR/PAC'),
|
||||
('17210106', 'FIESTA RTS B/YAKINIKU 300GR/PAC'),
|
||||
('17210107', 'FIESTA RTS B/RENDANG 300GR/PAC'),
|
||||
('17210108', 'FIESTA RTS B/BLACKPEPPER 300GR/PAC'),
|
||||
('17210109', 'FIESTA RTS B/BULGOGI 300GR/PAC'),
|
||||
('18050102', 'FIESTA RTG BAKSO KEJU 60 GR/PAC'),
|
||||
('18050103', 'FIESTA RTG BAKSO BAKAR BBQ 60 GR/PAC'),
|
||||
('18050104', 'FIESTA RTG BEEF BALL WITH MENTAI LAVA 55GR'),
|
||||
('18050105', 'FIESTA RTG BEEF BALL WITH CHEESE LAVA 55GR'),
|
||||
('20010101', 'FIESTA CRISPY CRUMBS 200 GR'),
|
||||
('20010102', 'FIESTA TP ROTI PUTIH 200 GR'),
|
||||
('20040101', 'FIESTA RAMEN BEKU 570 GR/PAC'),
|
||||
('20120102', 'FIESTA T/B AYAM GORENG 80 GR'),
|
||||
('20120105', 'FIESTA T/B SERBAGUNA 80 GR'),
|
||||
('20120106', 'FIESTA T/B KREMES 80 GR'),
|
||||
('20120115', 'FIESTA RACIK AYAM GORENG 20 GR/PAC'),
|
||||
('20120116', 'FIESTA RACIK NASI GORENG 20 GR/PAC'),
|
||||
('21000123', 'FIESTA RICE W/GEPREK CHICKEN 320GR/PAC'),
|
||||
('21000124', 'FIESTA RICE W/CHICK RUJAK 320 GR/PAC'),
|
||||
('21000125', 'FIESTA RICE W/KOREAN BBQ CHICK 320 GR/PAC'),
|
||||
('21000126', 'NEW FIESTA CHICK RENDANG W RICE 320GR (PAC)'),
|
||||
('21000127', 'NEW FIESTA CHICK TERIYAKI W RICE 320GR (PAC)'),
|
||||
('21000128', 'NEW FIESTA CHICK TANDORI W RICE 320GR (PAC)'),
|
||||
('21000129', 'NEW FIESTA RICE W KARAGE&SSS 320GR (PAC)'),
|
||||
('21000130', 'NEW FIESTA RICE W/C CHEESE BULDAK 320GR (PAC)'),
|
||||
('21000131', 'NEW FIESTA RICE W CHIC CURRY 320GR (PAC)'),
|
||||
('21000132', 'NEW FIESTA RICE W CHICK DONBURI 320GR (PAC)'),
|
||||
('21000133', 'NEW FIESTA RICE W CHICK SATAY 320GR (PAC)'),
|
||||
('21000134', 'NEW FIESTA COCONUT RICE W SPICY CHICK 320GR (PAC)'),
|
||||
('21000135', 'NEW FIESTA RICE W POPBITES S/MATAH 320GR (PAC)'),
|
||||
('21000136', 'NEW FIESTA TUMERIC W POPBITES 320GR (PAC)'),
|
||||
('21000137', 'FIESTA HAINAMESE CHICKEN RICE 320GR (PAC)'),
|
||||
('21010101', 'FIESTA TRUFFLE GYUDON 320 GR/PAC'),
|
||||
('21010102', 'NEW FIESTA BEEF YAKINIKU W RICE 320GR (PAC)'),
|
||||
('21010103', 'NEW FIESTA BEEF BULGOGI W RICE 320GR (PAC)'),
|
||||
('21010104', 'NEW FIESTA BEEF RENDANG W RICE 320GR (PAC)'),
|
||||
('21010105', 'NEW FIESTA RICE W BEEF BLACKPEPPER 320GR (PAC)'),
|
||||
('21200107', 'NEW FIESTA SPAGHETTI CARBONARA 300GR (PAC)'),
|
||||
('21200108', 'NEW FIESTA SPAGHETTI CHIC BOLOGNESE 320GR (PAC)'),
|
||||
('21200109', 'NEW FIESTA ITALIAN MEATBALL SPAGHETTI 320GR (PAC)'),
|
||||
('21310103', 'NEW FIESTA SCB&S/SSG FRIED RICE 320GR (PAC)'),
|
||||
('21500101', 'FIESTA CHICK SSG & C. BALL PIZZA 230GR/PAC'),
|
||||
('21500102', 'FIESTA CHEESY BEEF PIZZA 230GR/PAC'),
|
||||
('91000012', 'PHOTOCARD RTG'),
|
||||
('1188002W', 'PAHA ATAS 25-30 G FZ (*)'),
|
||||
('1195008A', 'RTC CHICKEN KALASAN 400 GR (PAC)'),
|
||||
('1195008E', 'RTC CHICKEN TERIYAKI 400 GR (PAC)'),
|
||||
('1195008X', 'RTC CHICKEN SPICY 400 GR (PAC)')
|
||||
ON CONFLICT (no_sku) DO NOTHING;
|
||||
-- Migration: 005_create_sku_master
|
||||
-- Description: Create sku_master table and seed the initial SKU entries
|
||||
|
||||
CREATE TABLE IF NOT EXISTS sku_master (
|
||||
id SERIAL PRIMARY KEY,
|
||||
no_sku VARCHAR(255) UNIQUE NOT NULL,
|
||||
nama_item VARCHAR(255) NOT NULL,
|
||||
created_at TIMESTAMP NOT NULL DEFAULT NOW()
|
||||
);
|
||||
|
||||
INSERT INTO sku_master (no_sku, nama_item) VALUES
|
||||
('11048006', 'BEBEK PARTING-NEW(*)'),
|
||||
('11110059', 'CEKER BERKUKU FROZEN PACK 1 KG(*)'),
|
||||
('11110074', 'CEKER 1 KG FROZEN (20 PAC/KARUNG)(*)'),
|
||||
('11140051', 'AMPELA FROZEN PACK 1 KG(*)'),
|
||||
('11140062', 'AMPELA 1 KG FROZEN (20 PAC/KARUNG)(*)'),
|
||||
('11148002', 'AMPELA BEBEK FROZEN 1 KG/PACK (NEW)(*)'),
|
||||
('11150052', 'HATI FROZEN PACK 1 KG(*)'),
|
||||
('11150055', 'JANTUNG FROZEN PACK 1 KG(*)'),
|
||||
('11150064', 'HATI 1 KG FROZEN (20 PAC/KARUNG)(*)'),
|
||||
('11150065', 'JANTUNG 1 KG FROZEN (20 PAC/KARUNG)(*)'),
|
||||
('11310012', 'AYAM SIZE 0 (0.6-0.7)KG(*)'),
|
||||
('11310013', 'AYAM SIZE1 FROZEN (0.75-0.8) KG(*)'),
|
||||
('11310014', 'AYAM SIZE 2 FROZEN (0.8-0.9)KG(*)'),
|
||||
('11310016', 'AYAM SIZE Z PR FROZ(*)'),
|
||||
('11310017', 'AYAM SIZE 0 PR FROZEN(*)'),
|
||||
('11310018', 'AYAM SIZE 1 PR FROZEN(*)'),
|
||||
('11310019', 'AYAM SIZE 2 PR FROZEN(*)'),
|
||||
('11310021', 'AYAM SIZE BESAR (B) FROZ (1-1.1)KG/PC(*)'),
|
||||
('11310022', 'AYAM SIZE A PR (0.9-1)KG/PC(*)'),
|
||||
('11310024', 'AYAM SIZE A FROZEN (0.9-1)KG/PC(*)'),
|
||||
('11310025', 'AYAM SIZE SUPER (C) FROZ(1.1-1.2)KG/ PC(*)'),
|
||||
('11310026', 'AYAM SIZE JUMBO (D) FROZ (1.2- 1.3)KG/PC(*)'),
|
||||
('11318301', 'BEBEK MUDA-BD1(1.0-1.1 KG)-NEW(*)'),
|
||||
('11318306', 'CP DUCK PEKING 1.5-1.6 KG/PC(*)'),
|
||||
('11318308', 'BEBEK PEKING SPR BD5(1.7 -1.8 )Kg-NEW(*)'),
|
||||
('11410043', 'PARTING 10 SIZE D FRESH BENSU 1.25 KG/PAC(*)'),
|
||||
('11420055', 'PARTING 12 ALL SIZE FROZ/PAC(*)'),
|
||||
('11600053', 'BONELESS LEG FROZEN 1 KG(*)'),
|
||||
('11620056', 'SBL (FILLET PAHA) 1 KG(*)'),
|
||||
('11640053', 'PAHA UTUH (1 KG)(*)'),
|
||||
('11650053', 'PAHA ATAS 1 KG(*)'),
|
||||
('11660050', 'PAHA BAWAH (1 KG)(*)'),
|
||||
('11690053', 'SBB (FILLET DADA )1 KG(*)'),
|
||||
('11690081', 'SBB JUMBO FZ (2.0 - 2.2 KG/PAC)(*)'),
|
||||
('11710051', 'DADA UTUH (1 KG)(*)'),
|
||||
('11720055', 'FULL WING FROZ PACK 1 KG(*)'),
|
||||
('11730050', 'MIDDLE WING FROZ PACK 1 KG(*)'),
|
||||
('11750050', 'FILLET MITRA 1 KG(*)'),
|
||||
('11818300', 'CP-BEBEK GORENG 400GR/PAC'),
|
||||
('11840002', 'AYAM JANTAN BKKL SZ 0 (600-700) GR/PC(*)'),
|
||||
('11959937', 'SATE AYAM FRESHMART 360 GR (PAC)'),
|
||||
('12010111', 'FIESTA CRISPY BUBBLE 400 GR/PAC'),
|
||||
('12010112', 'FIESTA CHICKEN NUGGET 400 GR/PAC'),
|
||||
('12010113', 'FIESTA CHICKEN NUGGET 200 GR/PAC'),
|
||||
('12010115', 'FIESTA NUGGET ZOO 400 GR/PAC'),
|
||||
('12010116', 'FIESTA NUGGET DINO 400 GR/PAC'),
|
||||
('12010117', 'FIESTA NUGGET HAPPY STAR 400 GR/PAC'),
|
||||
('12010119', 'FIESTA NUGGET CHEESE 123 400 GR/PAC'),
|
||||
('12010121', 'FIESTA NUGGET PIZZABC 400 GR/PAC'),
|
||||
('12010122', 'FIESTA CHEESY LOVER 400 GR/PAC'),
|
||||
('12010123', 'FIESTA GARLIC CHEESE 400 GR/PAC'),
|
||||
('12010124', 'FIESTA CHEESY CHIC W/BROCCOLI 400 GR/PAC'),
|
||||
('12010127', 'FIESTA SPICY NUGGET 400 GR/PAC'),
|
||||
('12010128', 'FIESTA VOLCANO CHEESE 400 GR/PAC'),
|
||||
('12010129', 'FIESTA CHEESY BOMBS CHICKEN NUGGET 400 GR'),
|
||||
('12010402', 'GOLDEN FIESTA NUGGET W/PINEAPPLE SAUCE 500 GR'),
|
||||
('12010509', 'CHAMP CRUNCHY NUGGET 450 GR/PAC'),
|
||||
('12010510', 'CHAMP NUGGET AYAM 225 GR/PAC'),
|
||||
('12010511', 'CHAMP NUGGET AYAM 450 GR/PAC'),
|
||||
('12010512', 'CHAMP NUGGET AYAM 900 GR/PAC'),
|
||||
('12010513', 'CHAMP NUGGET ABC KOMBINASI 225 GR/PAC'),
|
||||
('12010514', 'CHAMP NUGGET ABC KOMBINASI 450 GR/PAC'),
|
||||
('12010515', 'CHAMP KOIN KOMBINASI 450 GR/PAC'),
|
||||
('12010516', 'CHAMP KOIN KOMBINASI 200 GR/PAC'),
|
||||
('12010517', 'CHAMP NUGGET STICK 225 GR/PAC'),
|
||||
('12010518', 'CHAMP NUGGET STICK 450 GR/PAC'),
|
||||
('12010519', 'CHAMP NUGGET STICK 900 GR/PAC'),
|
||||
('12010520', 'CHAMP CHICKEN NUGGET BENTUK 123 450 GR/PAC'),
|
||||
('12010521', 'CHAMP NUGGET HOTZZ LEVEL 5 450 GR/PAC'),
|
||||
('12010606', 'CHAMP CRUNCHY NUGGET 225 GR/PAC'),
|
||||
('12010707', 'CHAMP MITRA NUGGET COIN 200 GR (NEW)'),
|
||||
('12010801', 'OKEY NUGGET 500GR'),
|
||||
('12012201', 'ASIMO NUGGET KOMBINASI 500 GR/PAC'),
|
||||
('12012202', 'ASIMO NUGGET KOMBINASI 1 KG/PAC'),
|
||||
('12012203', 'ASIMO NUGGET KOMBINASI 250 GR/PAC'),
|
||||
('12012501', 'AKUMO CHICKEN NAGET 250 GR'),
|
||||
('12012502', 'AKUMO CHICKEN NUGGET 500 GR'),
|
||||
('12012503', 'AKUMO CHICKEN NUGGET 1000 GR'),
|
||||
('12012504', 'AKUMO COIN 200 GR/PAC'),
|
||||
('12012505', 'AKUMO KOIN 400 GR/PAC'),
|
||||
('12020102', 'FIESTA SPICY WING 400 GR/PAC'),
|
||||
('12020401', 'GOLDEN FIESTA SP WING 500 GR'),
|
||||
('12030101', 'FIESTA STIKIE 400 GR/PAC'),
|
||||
('12030102', 'FIESTA STIKIE 200 GR/PAC'),
|
||||
('12030403', 'GOLDEN FIESTA STIKIE W/ SWEET CHILLI SAUCE 500GR'),
|
||||
('12030801', 'OKEY STICK 1000 GR'),
|
||||
('12030802', 'OKEY STICK 500GR'),
|
||||
('12032201', 'ASIMO STICK KOMBINASI 500 GR/PAC'),
|
||||
('12032202', 'ASIMO STICK KOMBINASI 1000 GR/PAC'),
|
||||
('12032203', 'ASIMO STIK KOMBINASI 250 GR/PAC'),
|
||||
('12032501', 'AKUMO CHICKEN STICK 250 GR'),
|
||||
('12032502', 'AKUMO CHICKEN STIK 500 GR'),
|
||||
('12032503', 'AKUMO CHICKEN STICK 1000 GR'),
|
||||
('12040101', 'FIESTA SCHNITZEL 400 GR/PAC'),
|
||||
('12040102', 'FIESTA CRISPY BUBBLE KATSU 400 GR/PAC'),
|
||||
('12040404', 'GOLDEN FIESTA CORDON BLEU BBQ SAUCE 500 GR'),
|
||||
('12040406', 'GOLDEN FIESTA KATSU W/CHEESE SAUCE 500 GR/PAC'),
|
||||
('12050103', 'FIESTA FRIED CHICKEN 400 GR/PAC'),
|
||||
('12050104', 'FIESTA HOT & CRISPY FRIED CHICKEN 400 GR/PAC'),
|
||||
('12050401', 'GOLDEN FIESTA CRISPY WING W/SP GLAZ SC 500 GR/PAC'),
|
||||
('12060103', 'FIESTA KARAGE 200 GR/PAC'),
|
||||
('12060104', 'FIESTA KARAGE 400 GR/PAC'),
|
||||
('12060105', 'FIESTA SPICY KARAGE 400 GR/PAC'),
|
||||
('12060402', 'GOLDEN FIESTA KARAGE CHILI SAUCE 500GR'),
|
||||
('12070101', 'FIESTA POK-POK 400 GR/PAC (NEW)'),
|
||||
('12080101', 'FIESTA SPICY CHICK 400 GR/PAC'),
|
||||
('12130102', 'FIESTA CRISPY BURGER 360 GR (NEW)'),
|
||||
('12130504', 'CHAMP BURGER 315 GR (NEW)'),
|
||||
('12140105', 'FIESTA CHICK TOFU 400 GR/PAC'),
|
||||
('12150201', 'FIESTA DS CRISPY CRUNCH 300 GR/PAC'),
|
||||
('12150501', 'CHAMP CRUNCHY HOTZZ 300 GR/PAC'),
|
||||
('12190103', 'FIESTA DELISTRIPE 400 GR/PAC'),
|
||||
('12240102', 'FIESTA CHEESY ITALIAN R/BITES 400 GR/PAC'),
|
||||
('12240103', 'FIESTA YAKINIKU R/BITES 400 GR/PAC'),
|
||||
('13010101', 'FIESTA CHICK SSG 300 GR'),
|
||||
('13010102', 'FIESTA CHICK SSG 500 GR'),
|
||||
('13010103', 'FIESTA CHICK SSG 200 GR/PAC'),
|
||||
('13010111', 'FIESTA SOSIS BRATWURST 300 GR'),
|
||||
('13010112', 'FIESTA CHEESE SSG 300 GR'),
|
||||
('13010113', 'FIESTA SOSIS CURRYWURST 300 GR'),
|
||||
('13010114', 'FIESTA SSG BOCKWURST 300GR'),
|
||||
('13010115', 'FIESTA SSG WIENER 300GR'),
|
||||
('13010116', 'FIESTA SSG ORIGINAL 300 GR'),
|
||||
('13010117', 'FIESTA SSG FRANKFURTER 300GR'),
|
||||
('13010118', 'FIESTA RTG SSG 65 GR/PAC'),
|
||||
('13010119', 'FIESTA RTG C/SPICY KOREAN 60 GR/PAC'),
|
||||
('13010120', 'FIESTA RTG C/CHEESY MELTS 65 GR/PAC'),
|
||||
('13010122', 'FIESTA RTG SAUSAGE WITH HOT LAVA 60G'),
|
||||
('13010123', 'FIESTA RTG SAUSAGE WITH CHEESE LAVA 60G'),
|
||||
('13010124', 'FIESTA RTG SAUSAGE HICKORY SAUCE 60GR'),
|
||||
('13010125', 'FIESTA RTG SAUSAGE WITH MENTAI LAVA 60GR'),
|
||||
('13010510', 'CHAMP CHICK SSG 75 GR'),
|
||||
('13010513', 'CHAMP CHICK SSG 375 GR'),
|
||||
('13010514', 'CHAMP CHICK SSG 1000 GR'),
|
||||
('13010518', 'CHAMP SSG JUMBO BAKAR 500 GR/PAC'),
|
||||
('13010519', 'CHAMP SSG BAKAR MINI 500 GR/PAC-INACT'),
|
||||
('13010521', 'CHAMP CHICK SSG 150 GR/PAC (NEW)'),
|
||||
('13010523', 'CHAMP CHICK SSG AYAM MADU 300 GR/PAC'),
|
||||
('13010524', 'CHAMP SSG JUMBO BAKAR 500 GR/PAC (NEW)'),
|
||||
('13010525', 'CHAMP SSG BAKAR MINI 500 GR/PAC (NEW)'),
|
||||
('13010809', 'OKEY CHICK SSG 500GR-INACT'),
|
||||
('13010815', 'OKEY SSG BAKAR JUMBO 500 GR/PAC (NEW)'),
|
||||
('13010816', 'OKEY SSG BAKAR MINI 500 GR/PAC (NEW)'),
|
||||
('13010817', 'OKEY SSG CHICK KOMBINASI 500 GR (EXTRA1)'),
|
||||
('13010818', 'OKEY SSG CHICK KOMBINASI 1 KG (EXTR$A2)'),
|
||||
('13012205', 'ASIMO SOSIS AYAM KOMBINASI 375 GR (PAC)'),
|
||||
('13012206', 'ASIMO SOSIS AYAM KOMBINASI 500 GR'),
|
||||
('13012207', 'ASIMO SOSIS AYAM KOMBINASI 750 GR'),
|
||||
('13012208', 'ASIMO SOSIS AYAM KOMBINASI 1000 GR'),
|
||||
('13012209', 'ASIMO SSG CHICK KOMBINASI 500 GR (EXTRA1)'),
|
||||
('13012210', 'ASIMO SSG CHICK KOMBINASI 1 KG (EXTR$A2)'),
|
||||
('13030101', 'FIESTA CHICK MEAT BALL 300 GR'),
|
||||
('13030102', 'FIESTA CHICK MEATBALL 500 GR'),
|
||||
('13030501', 'CHAMP CHICK MEATBALL 200 GR'),
|
||||
('13030502', 'CHAMP CHICK MEATBALL 500 GR'),
|
||||
('13050101', 'FIESTA SCB 250 GR'),
|
||||
('13050105', 'FIESTA CHICKEN SLICE 300 GR'),
|
||||
('13050106', 'FIESTA BEEF SLICE 300 GR'),
|
||||
('13070501', 'CHAMP BEEF SSG SERBAGUNA 150 GR'),
|
||||
('13070502', 'CHAMP BEEF SSG SERBAGUNA 375GR'),
|
||||
('13070505', 'CHAMP BEEF SSG GORENG 375 GR'),
|
||||
('13070506', 'CHAMP FRANKFURTER SSG 375GR'),
|
||||
('13100512', 'CHAMP CHICK SSG S/SANTAP ORIG 546GR (CAN)'),
|
||||
('13110504', 'CHAMP BEEF BALL 500GR'),
|
||||
('13170510', 'CHAMP BEEF BBQ SSG S/SANTAP 546GR (CAN)'),
|
||||
('15010101', 'FIESTA SHOESTRING 500 GR'),
|
||||
('15010102', 'FIESTA SHOESTRING 1000 GR'),
|
||||
('15010107', 'FIESTA FRENCH F SHOESTRING INSTITUSI 2KG'),
|
||||
('15020101', 'FIESTA STRAIGHT CUT 500 GR'),
|
||||
('15020102', 'FIESTA STRAIGHT CUT 1000 GR'),
|
||||
('15030101', 'FIESTA CRINKLE CUT 500 GR'),
|
||||
('15030102', 'FIESTA CRINKLE CUT 1000 GR'),
|
||||
('15040101', 'FIESTA BATTER COATED 500 GR'),
|
||||
('15040102', 'FIESTA BATTER COATED 1000 GR'),
|
||||
('16060103', 'FIESTA CHICK SIOMAY 900GR'),
|
||||
('16060113', 'FIESTA CHICK SIOMAY 180GR (NEW)'),
|
||||
('16060114', 'FIESTA GYOZA 180 GR (NEW)'),
|
||||
('16060119', 'FIESTA RTG SIOMAY 54 GR/PAC'),
|
||||
('16060120', 'FIESTA KEECHO 400 GR/PAC'),
|
||||
('16060121', 'FIESTA CHICKEN TOFU 400 GR/PAC (NEW)'),
|
||||
('16060503', 'CHAMP CHICK&FISH SIOMAY 180 GR (NEW)'),
|
||||
('17200109', 'FIESTA RTS C/TERIYAKI 300GR/PAC'),
|
||||
('17200110', 'FIESTA RTS C/RENDANG 300GR/PAC'),
|
||||
('17200111', 'FIESTA RTS C/W RUJAK SC 300GR/PAC'),
|
||||
('17200112', 'FIESTA RTS C/W S/MATAH 300GR/PAC'),
|
||||
('17210106', 'FIESTA RTS B/YAKINIKU 300GR/PAC'),
|
||||
('17210107', 'FIESTA RTS B/RENDANG 300GR/PAC'),
|
||||
('17210108', 'FIESTA RTS B/BLACKPEPPER 300GR/PAC'),
|
||||
('17210109', 'FIESTA RTS B/BULGOGI 300GR/PAC'),
|
||||
('18050102', 'FIESTA RTG BAKSO KEJU 60 GR/PAC'),
|
||||
('18050103', 'FIESTA RTG BAKSO BAKAR BBQ 60 GR/PAC'),
|
||||
('18050104', 'FIESTA RTG BEEF BALL WITH MENTAI LAVA 55GR'),
|
||||
('18050105', 'FIESTA RTG BEEF BALL WITH CHEESE LAVA 55GR'),
|
||||
('20010101', 'FIESTA CRISPY CRUMBS 200 GR'),
|
||||
('20010102', 'FIESTA TP ROTI PUTIH 200 GR'),
|
||||
('20040101', 'FIESTA RAMEN BEKU 570 GR/PAC'),
|
||||
('20120102', 'FIESTA T/B AYAM GORENG 80 GR'),
|
||||
('20120105', 'FIESTA T/B SERBAGUNA 80 GR'),
|
||||
('20120106', 'FIESTA T/B KREMES 80 GR'),
|
||||
('20120115', 'FIESTA RACIK AYAM GORENG 20 GR/PAC'),
|
||||
('20120116', 'FIESTA RACIK NASI GORENG 20 GR/PAC'),
|
||||
('21000123', 'FIESTA RICE W/GEPREK CHICKEN 320GR/PAC'),
|
||||
('21000124', 'FIESTA RICE W/CHICK RUJAK 320 GR/PAC'),
|
||||
('21000125', 'FIESTA RICE W/KOREAN BBQ CHICK 320 GR/PAC'),
|
||||
('21000126', 'NEW FIESTA CHICK RENDANG W RICE 320GR (PAC)'),
|
||||
('21000127', 'NEW FIESTA CHICK TERIYAKI W RICE 320GR (PAC)'),
|
||||
('21000128', 'NEW FIESTA CHICK TANDORI W RICE 320GR (PAC)'),
|
||||
('21000129', 'NEW FIESTA RICE W KARAGE&SSS 320GR (PAC)'),
|
||||
('21000130', 'NEW FIESTA RICE W/C CHEESE BULDAK 320GR (PAC)'),
|
||||
('21000131', 'NEW FIESTA RICE W CHIC CURRY 320GR (PAC)'),
|
||||
('21000132', 'NEW FIESTA RICE W CHICK DONBURI 320GR (PAC)'),
|
||||
('21000133', 'NEW FIESTA RICE W CHICK SATAY 320GR (PAC)'),
|
||||
('21000134', 'NEW FIESTA COCONUT RICE W SPICY CHICK 320GR (PAC)'),
|
||||
('21000135', 'NEW FIESTA RICE W POPBITES S/MATAH 320GR (PAC)'),
|
||||
('21000136', 'NEW FIESTA TUMERIC W POPBITES 320GR (PAC)'),
|
||||
('21000137', 'FIESTA HAINAMESE CHICKEN RICE 320GR (PAC)'),
|
||||
('21010101', 'FIESTA TRUFFLE GYUDON 320 GR/PAC'),
|
||||
('21010102', 'NEW FIESTA BEEF YAKINIKU W RICE 320GR (PAC)'),
|
||||
('21010103', 'NEW FIESTA BEEF BULGOGI W RICE 320GR (PAC)'),
|
||||
('21010104', 'NEW FIESTA BEEF RENDANG W RICE 320GR (PAC)'),
|
||||
('21010105', 'NEW FIESTA RICE W BEEF BLACKPEPPER 320GR (PAC)'),
|
||||
('21200107', 'NEW FIESTA SPAGHETTI CARBONARA 300GR (PAC)'),
|
||||
('21200108', 'NEW FIESTA SPAGHETTI CHIC BOLOGNESE 320GR (PAC)'),
|
||||
('21200109', 'NEW FIESTA ITALIAN MEATBALL SPAGHETTI 320GR (PAC)'),
|
||||
('21310103', 'NEW FIESTA SCB&S/SSG FRIED RICE 320GR (PAC)'),
|
||||
('21500101', 'FIESTA CHICK SSG & C. BALL PIZZA 230GR/PAC'),
|
||||
('21500102', 'FIESTA CHEESY BEEF PIZZA 230GR/PAC'),
|
||||
('91000012', 'PHOTOCARD RTG'),
|
||||
('1188002W', 'PAHA ATAS 25-30 G FZ (*)'),
|
||||
('1195008A', 'RTC CHICKEN KALASAN 400 GR (PAC)'),
|
||||
('1195008E', 'RTC CHICKEN TERIYAKI 400 GR (PAC)'),
|
||||
('1195008X', 'RTC CHICKEN SPICY 400 GR (PAC)')
|
||||
ON CONFLICT (no_sku) DO NOTHING;
|
||||
@@ -1,9 +1,9 @@
|
||||
services:
|
||||
db:
|
||||
ports:
|
||||
- "5432:5432"
|
||||
pipeline-api:
|
||||
ports:
|
||||
- "8090:8090"
|
||||
environment:
|
||||
- VLLM_SERVER_URL=http://paddleocr-vllm-server:8118/v1
|
||||
services:
|
||||
db:
|
||||
ports:
|
||||
- "5432:5432"
|
||||
pipeline-api:
|
||||
ports:
|
||||
- "8090:8090"
|
||||
environment:
|
||||
- VLLM_SERVER_URL=http://paddleocr-vllm-server:8118/v1
|
||||
+146
-146
@@ -1,146 +1,146 @@
|
||||
name: ai-ocr-pfm-2026
|
||||
|
||||
services:
|
||||
nginx:
|
||||
image: nginx:alpine
|
||||
container_name: paddleocr-nginx
|
||||
ports:
|
||||
- "${APP_PORT:-8000}:80"
|
||||
volumes:
|
||||
- ./nginx.conf:/etc/nginx/nginx.conf:ro
|
||||
depends_on:
|
||||
- vllm-server
|
||||
- pipeline-api
|
||||
- gradio-ui
|
||||
- pfm-web-app
|
||||
restart: unless-stopped
|
||||
|
||||
vllm-server:
|
||||
build:
|
||||
context: .
|
||||
target: vllm-server
|
||||
container_name: paddleocr-vllm-server
|
||||
image: paddleocr-vllm-server:latest
|
||||
environment:
|
||||
- GENAI_HOST=0.0.0.0
|
||||
- GENAI_PORT=8118
|
||||
- GENAI_MODEL=${GENAI_MODEL:-PaddleOCR-VL-1.6-0.9B}
|
||||
- GENAI_BACKEND=${GENAI_BACKEND:-vllm}
|
||||
- VLLM_CONFIG=${VLLM_CONFIG:-config/vllm_config.yaml}
|
||||
- CUDA_VISIBLE_DEVICES=${CUDA_VISIBLE_DEVICES:-0}
|
||||
- PADDLE_PDX_DISABLE_MODEL_SOURCE_CHECK=True
|
||||
# No exposed ports; internal only
|
||||
deploy:
|
||||
resources:
|
||||
reservations:
|
||||
devices:
|
||||
- driver: nvidia
|
||||
count: all
|
||||
capabilities: [gpu]
|
||||
volumes:
|
||||
- hf_cache:/root/.cache/huggingface
|
||||
- paddle_cache:/root/.paddleocr
|
||||
- paddlex_cache:/root/.paddlex
|
||||
- ./config:/app/config
|
||||
- ./.env:/app/.env:ro
|
||||
restart: unless-stopped
|
||||
|
||||
pipeline-api:
|
||||
build:
|
||||
context: .
|
||||
target: pipeline-api
|
||||
container_name: paddleocr-pipeline-api-v10
|
||||
image: paddleocr-pipeline-api:latest
|
||||
environment:
|
||||
- PIPELINE_CONFIG=${PIPELINE_CONFIG:-config/pipeline_config_vllm.yaml}
|
||||
- PIPELINE_HOST=0.0.0.0
|
||||
- PIPELINE_PORT=8090
|
||||
- PIPELINE_DEVICE=gpu:0
|
||||
- CUDA_VISIBLE_DEVICES=${CUDA_VISIBLE_DEVICES:-0}
|
||||
- VLLM_SERVER_URL=http://paddleocr-pfm-web-app:3000/api/vllm-proxy/v1
|
||||
- PADDLE_PDX_DISABLE_MODEL_SOURCE_CHECK=True
|
||||
# No exposed ports; internal only
|
||||
deploy:
|
||||
resources:
|
||||
reservations:
|
||||
devices:
|
||||
- driver: nvidia
|
||||
count: all
|
||||
capabilities: [gpu]
|
||||
volumes:
|
||||
- paddle_cache:/root/.paddleocr
|
||||
- paddlex_cache:/root/.paddlex
|
||||
- ./config:/app/config
|
||||
- ./pfm-web-app/public/produk-pfm/models:/app/pfm-web-app/public/produk-pfm/models:ro
|
||||
- ./.env:/app/.env:ro
|
||||
depends_on:
|
||||
- vllm-server
|
||||
restart: unless-stopped
|
||||
|
||||
gradio-ui:
|
||||
build:
|
||||
context: .
|
||||
target: gradio-ui
|
||||
container_name: paddleocr-gradio-ui
|
||||
image: paddleocr-gradio-ui:latest
|
||||
environment:
|
||||
- API_URL=http://paddleocr-pipeline-api-v10:8090/layout-parsing
|
||||
- GRADIO_PORT=7870
|
||||
- GRADIO_MCP_SERVER=True
|
||||
# No exposed ports; internal only
|
||||
depends_on:
|
||||
- pipeline-api
|
||||
restart: unless-stopped
|
||||
|
||||
pfm-web-app:
|
||||
build:
|
||||
context: .
|
||||
target: pfm-web-app
|
||||
container_name: paddleocr-pfm-web-app
|
||||
image: paddleocr-pfm-web-app:latest
|
||||
command: npm run dev
|
||||
environment:
|
||||
- PIPELINE_URL=http://paddleocr-pipeline-api-v10:8090/layout-parsing
|
||||
- NODE_ENV=development
|
||||
- PGHOST=paddleocr-db
|
||||
- PGPORT=5432
|
||||
- PGUSER=postgres
|
||||
- PGPASSWORD=postgres
|
||||
- PGDATABASE=dopfm
|
||||
- JWT_SECRET=${JWT_SECRET:-dev-only-insecure-secret-change-me}
|
||||
pid: "host"
|
||||
volumes:
|
||||
- ./pfm-web-app:/app
|
||||
- /app/node_modules
|
||||
- /app/.next
|
||||
- ./uploads:/uploads
|
||||
- /var/run/docker.sock:/var/run/docker.sock
|
||||
# No exposed ports; internal only
|
||||
depends_on:
|
||||
- pipeline-api
|
||||
- db
|
||||
extra_hosts:
|
||||
- "host.docker.internal:host-gateway"
|
||||
restart: unless-stopped
|
||||
|
||||
db:
|
||||
image: postgres:15-alpine
|
||||
container_name: paddleocr-db
|
||||
environment:
|
||||
- POSTGRES_USER=postgres
|
||||
- POSTGRES_PASSWORD=postgres
|
||||
- POSTGRES_DB=dopfm
|
||||
volumes:
|
||||
- pgdata:/var/lib/postgresql/data
|
||||
- ./db/migrations:/docker-entrypoint-initdb.d:ro
|
||||
restart: unless-stopped
|
||||
|
||||
volumes:
|
||||
hf_cache:
|
||||
name: paddleocr_hf_cache
|
||||
paddle_cache:
|
||||
name: paddleocr_paddle_cache
|
||||
paddlex_cache:
|
||||
name: paddleocr_paddlex_cache
|
||||
pgdata:
|
||||
name: paddleocr_pgdata
|
||||
name: ai-ocr-pfm-2026
|
||||
|
||||
services:
|
||||
nginx:
|
||||
image: nginx:alpine
|
||||
container_name: paddleocr-nginx
|
||||
ports:
|
||||
- "${APP_PORT:-8000}:80"
|
||||
volumes:
|
||||
- ./nginx.conf:/etc/nginx/nginx.conf:ro
|
||||
depends_on:
|
||||
- vllm-server
|
||||
- pipeline-api
|
||||
- gradio-ui
|
||||
- pfm-web-app
|
||||
restart: unless-stopped
|
||||
|
||||
vllm-server:
|
||||
build:
|
||||
context: .
|
||||
target: vllm-server
|
||||
container_name: paddleocr-vllm-server
|
||||
image: paddleocr-vllm-server:latest
|
||||
environment:
|
||||
- GENAI_HOST=0.0.0.0
|
||||
- GENAI_PORT=8118
|
||||
- GENAI_MODEL=${GENAI_MODEL:-PaddleOCR-VL-1.6-0.9B}
|
||||
- GENAI_BACKEND=${GENAI_BACKEND:-vllm}
|
||||
- VLLM_CONFIG=${VLLM_CONFIG:-config/vllm_config.yaml}
|
||||
- CUDA_VISIBLE_DEVICES=${CUDA_VISIBLE_DEVICES:-0}
|
||||
- PADDLE_PDX_DISABLE_MODEL_SOURCE_CHECK=True
|
||||
# No exposed ports; internal only
|
||||
deploy:
|
||||
resources:
|
||||
reservations:
|
||||
devices:
|
||||
- driver: nvidia
|
||||
count: all
|
||||
capabilities: [gpu]
|
||||
volumes:
|
||||
- hf_cache:/root/.cache/huggingface
|
||||
- paddle_cache:/root/.paddleocr
|
||||
- paddlex_cache:/root/.paddlex
|
||||
- ./config:/app/config
|
||||
- ./.env:/app/.env:ro
|
||||
restart: unless-stopped
|
||||
|
||||
pipeline-api:
|
||||
build:
|
||||
context: .
|
||||
target: pipeline-api
|
||||
container_name: paddleocr-pipeline-api-v10
|
||||
image: paddleocr-pipeline-api:latest
|
||||
environment:
|
||||
- PIPELINE_CONFIG=${PIPELINE_CONFIG:-config/pipeline_config_vllm.yaml}
|
||||
- PIPELINE_HOST=0.0.0.0
|
||||
- PIPELINE_PORT=8090
|
||||
- PIPELINE_DEVICE=gpu:0
|
||||
- CUDA_VISIBLE_DEVICES=${CUDA_VISIBLE_DEVICES:-0}
|
||||
- VLLM_SERVER_URL=http://paddleocr-pfm-web-app:3000/api/vllm-proxy/v1
|
||||
- PADDLE_PDX_DISABLE_MODEL_SOURCE_CHECK=True
|
||||
# No exposed ports; internal only
|
||||
deploy:
|
||||
resources:
|
||||
reservations:
|
||||
devices:
|
||||
- driver: nvidia
|
||||
count: all
|
||||
capabilities: [gpu]
|
||||
volumes:
|
||||
- paddle_cache:/root/.paddleocr
|
||||
- paddlex_cache:/root/.paddlex
|
||||
- ./config:/app/config
|
||||
- ./pfm-web-app/public/produk-pfm/models:/app/pfm-web-app/public/produk-pfm/models:ro
|
||||
- ./.env:/app/.env:ro
|
||||
depends_on:
|
||||
- vllm-server
|
||||
restart: unless-stopped
|
||||
|
||||
gradio-ui:
|
||||
build:
|
||||
context: .
|
||||
target: gradio-ui
|
||||
container_name: paddleocr-gradio-ui
|
||||
image: paddleocr-gradio-ui:latest
|
||||
environment:
|
||||
- API_URL=http://paddleocr-pipeline-api-v10:8090/layout-parsing
|
||||
- GRADIO_PORT=7870
|
||||
- GRADIO_MCP_SERVER=True
|
||||
# No exposed ports; internal only
|
||||
depends_on:
|
||||
- pipeline-api
|
||||
restart: unless-stopped
|
||||
|
||||
pfm-web-app:
|
||||
build:
|
||||
context: .
|
||||
target: pfm-web-app
|
||||
container_name: paddleocr-pfm-web-app
|
||||
image: paddleocr-pfm-web-app:latest
|
||||
command: npm run dev
|
||||
environment:
|
||||
- PIPELINE_URL=http://paddleocr-pipeline-api-v10:8090/layout-parsing
|
||||
- NODE_ENV=development
|
||||
- PGHOST=paddleocr-db
|
||||
- PGPORT=5432
|
||||
- PGUSER=postgres
|
||||
- PGPASSWORD=postgres
|
||||
- PGDATABASE=dopfm
|
||||
- JWT_SECRET=${JWT_SECRET:-dev-only-insecure-secret-change-me}
|
||||
pid: "host"
|
||||
volumes:
|
||||
- ./pfm-web-app:/app
|
||||
- /app/node_modules
|
||||
- /app/.next
|
||||
- ./uploads:/uploads
|
||||
- /var/run/docker.sock:/var/run/docker.sock
|
||||
# No exposed ports; internal only
|
||||
depends_on:
|
||||
- pipeline-api
|
||||
- db
|
||||
extra_hosts:
|
||||
- "host.docker.internal:host-gateway"
|
||||
restart: unless-stopped
|
||||
|
||||
db:
|
||||
image: postgres:15-alpine
|
||||
container_name: paddleocr-db
|
||||
environment:
|
||||
- POSTGRES_USER=postgres
|
||||
- POSTGRES_PASSWORD=postgres
|
||||
- POSTGRES_DB=dopfm
|
||||
volumes:
|
||||
- pgdata:/var/lib/postgresql/data
|
||||
- ./db/migrations:/docker-entrypoint-initdb.d:ro
|
||||
restart: unless-stopped
|
||||
|
||||
volumes:
|
||||
hf_cache:
|
||||
name: paddleocr_hf_cache
|
||||
paddle_cache:
|
||||
name: paddleocr_paddle_cache
|
||||
paddlex_cache:
|
||||
name: paddleocr_paddlex_cache
|
||||
pgdata:
|
||||
name: paddleocr_pgdata
|
||||
@@ -1,95 +1,95 @@
|
||||
# Feature List (backend)
|
||||
|
||||
Structured log of shipped backend features, updated by the `n`/`next` workflow (see
|
||||
[AGENTS.md](../AGENTS.md) Part B) whenever a task in
|
||||
[plans/next-enhancements.md](../plans/next-enhancements.md) is marked `[DONE]`.
|
||||
Split out 2026-07-08 from root `docs/feature-list.md`'s backend sections — this file
|
||||
is the sole home for backend feature history going forward.
|
||||
|
||||
## Format
|
||||
|
||||
```
|
||||
## <Section / Module Name>
|
||||
|
||||
- **<task number>** <feature description> — shipped <date>
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## Existing Features (pre-kit)
|
||||
|
||||
Backfilled 2026-07-08 during adoption of this kit — these predate the `e`/`n`
|
||||
workflow and have no task numbers; see `git log` for real dates/history.
|
||||
|
||||
### Backend — Next.js API Gateway
|
||||
- Upload/parse/documents CRUD routes, GPU status endpoint, vLLM proxy, manual-label review tool.
|
||||
|
||||
### Backend — OCR Pipeline & Accuracy
|
||||
- PaddleOCR + vLLM classification pipeline with DB layout caching, table column-shift correction, date normalization, and an accuracy regression harness (`pfm-web-app/scripts/accuracy-check.mts`) — **95.10% overall as of 2026-07-08** (target 95% met; see task 2.3 below for the investigation and `CLAUDE.md`).
|
||||
|
||||
### Backend — Postgres Data Layer
|
||||
- Schema/init in `pfm-web-app/src/db/init.ts`, served via the canonical root `docker-compose.yml` stack.
|
||||
- **3.3** Added a standard `INDEX` on `documents(file_hash)` in `db/init.ts` to accelerate the upload deduplication queries without strictly enforcing uniqueness across different stores. Correspondingly updated the dedup query in `api/v1/documents/upload/route.ts` to scope duplicate detection by `kode_toko`. This fixes a conflict where one store could be incorrectly linked to another store's duplicate receipt image — shipped 2026-07-08.
|
||||
|
||||
### DevOps — Docker & Dev Tunnel
|
||||
- **4.1 Docker Compose Policy Documented**: Formalized the execution policy in `README.md` and `CLAUDE.md`, explicitly requiring the use of the `docker-compose.demo.yml` override (production build) for all client demonstrations and field testing to bypass the Next.js dev server bottleneck — shipped 2026-07-08.
|
||||
- **Docker Compose Dependency Gates**: Added strict Docker `healthcheck` gates (`Task 4.2`) blocking the `pfm-web-app` (Next.js) from starting until PostgreSQL and the VLLM models are initialized and fully healthy.
|
||||
- **Secure Tunnel Ingress**: Restructured `nginx.conf` and `start-dev-tunnel.ps1` (`Tasks 4.3, 4.5`) to expose a dedicated, restricted port (`8001`) that exclusively routes to `/api/v1/*`. This perfectly secures the development UI (`/scan-pfm`) and legacy routes from public exposure.
|
||||
- **Dead Config Pruning**: Stripped deprecated and redundant proxy blocks from the Nginx edge router (`Task 4.4`).
|
||||
|
||||
*(New features shipped via `n`/`next` go below, organized the same way, with task numbers.)*
|
||||
|
||||
## Backend — Next.js API Gateway
|
||||
|
||||
- **1.4** Enforced real 401 auth on `/api/v1/documents/*` (list, PUT-by-id, upload) — the actual production API surface, already fully supported by the Flutter client (real login + `Authorization: Bearer` on every request). Previously none of these three routes rejected a missing/invalid token; upload only optionally read it. Added the pre-existing `getAccountFromAuthHeader()` helper (`utils/auth.ts`) + a 401 guard to all three; `OPTIONS` (CORS preflight) untouched. The original task 1.3 (auth on the *classic* routes) was cancelled instead — those routes are dev-only web UI surface with no login flow, going away in production. Verified via `curl`: 401 with no token, success with a real token from `/api/v1/auth/login` — shipped 2026-07-08.
|
||||
- **1.5** Implemented per-store data scoping on `/api/v1/documents/*`. Added `kode_toko` column to `documents` table via `db/init.ts` migration. The upload route now binds `kode_toko` to documents upon creation. `GET /api/v1/documents` and `PUT /api/v1/documents/:id` enforce ownership checks (`kode_toko` matching) for `store` role accounts, while `admin` retains global access including legacy unassigned documents — shipped 2026-07-08.
|
||||
- **1.6** `GET /api/v1/health` Endpoint: Unauthenticated health probe verifying both PostgreSQL connectivity and Pipeline API HTTP reachability. Returns `HTTP 503` if any core dependency is down — shipped 2026-07-08.
|
||||
- **Ad-hoc** Connected `/scan-pfm` page with `/api/scan-pfm` route and enabled auto-trigger scanning on custom file upload, sample selection, thumbnail change, and canvas rotation. Supported both `image` and `image_base64` payload keys — shipped 2026-07-09.
|
||||
- **9.1** Added `GET /api/v1/documents/:id` (same 401/403 scoping as `PUT`), returning a single document — including still-unparsed rows — with a new `parseStatus: "pending"|"done"|"failed"` field, so the Flutter poller can move off scanning the entire list every 2s. Added `scan_mode`/`parse_error` columns to `documents` (`db/init.ts`, migrated via `ALTER TABLE ... ADD COLUMN IF NOT EXISTS` for already-running DBs). `scan_mode` is now persisted on upload (`v1/documents/upload/route.ts`) and on the classic `/api/parse` route's upserts (`COALESCE`, same pattern as `kode_toko`), and surfaced as `docType` on every GET response (`utils/document-mapper.ts`, a new shared helper extracted from the list route's inline mapping so list/by-id/dedup all agree) — falling back to the legacy `order_untuk == "PRODUCT SCAN"` sentinel for pre-existing rows with no `scan_mode`. `parse_error` is now recorded when the upload route's *internal* call to `/api/parse` itself fails to complete (network error or the 210s abort firing) — previously this was silently swallowed and the document stayed `parsed=false` forever with no signal, burning the client's full 260s timeout; `/api/parse`'s own existing pipeline-error fallback (`parsed=true` + "Not Found" placeholder) was already fine and is unchanged. Also fixed the dedup branch (a repeat upload of an already-seen file) to return the original document's real current state via the same mapper instead of a hardcoded empty stub. Verified via `docker compose up -d --build` + `curl`: schema migration applied cleanly to the live DB (confirmed via `psql`), DO and Product uploads both correctly persist `scan_mode` and surface it as `docType`, a dedup retry returns real header/items instead of an empty stub, `GET /:id` returns 401 (no token) / 403 (wrong store) / 404 (nonexistent id) / 200 (admin or owning store), and the list endpoint's existing filter/scoping is unchanged — shipped 2026-07-10.
|
||||
- **9.3** Added authenticated `POST /api/v1/scan-product`, the v1 equivalent of the classic dev-only `/api/scan-pfm` (unauthenticated, and unreachable off-LAN since task 4.5 restricted the public tunnel to `/api/v1/*`). Extracted the shared classify-and-match logic (Python classifier call + Levenshtein SKU matching against `sku_master`, top-5 scoring) out of `api/scan-pfm/route.ts` into a new `utils/product-scan.ts` (`classifyAndMatchProduct`, plus a `ClassifierError` class that preserves forwarding the classifier's own HTTP status instead of collapsing every failure to 500) so the classic route and the new v1 route share one implementation instead of duplicating it — the classic route's response shape, auth-free behavior, and desktop-only layout-parsing visualization are otherwise unchanged. The new route accepts **either** multipart (`image`/`file` field, matching the v1 upload route's convention) or a JSON `{image_base64}` body, is open to any authenticated account (not admin-gated, since this is what the mobile app itself calls), and wraps the result in the standard `{status, data}` envelope with `classification`, `ocr` (including `extracted_expired_date`), and `possibleMatches`. Verified via `curl` against the live stack with a real product photo: multipart upload and JSON-body variants both return identical, correct top-5 matches; no-token request returns 401; the classic `/api/scan-pfm` route's response (including `layoutParsingResult`) is unchanged post-refactor — shipped 2026-07-10.
|
||||
- **9.2** Relaxed `GET /api/v1/master/skus` (`master/skus/route.ts`) so any authenticated account can read the SKU master list, not just `admin` — the Flutter product editor needs this and previously had to string-hack its base URL to call the unauthenticated classic `GET /api/skus`, which task 4.5 had already removed from the public tunnel, breaking product scans off-LAN. Changed the guard from a combined `!account || role !== 'admin'` check (403 for both "no token" and "wrong role") to `!account` (correct 401) followed by an unconditional pass-through for any valid account; `POST` (SKU creation) is untouched, still admin-only, per the user's explicit choice between the two options this task flagged as undecided. No response-shape change. Verified via `curl` against the live stack with a real non-admin (`store` role) account's token: `GET` → 200 with real data; no token → 401 (was incorrectly 403 before this fix); the same non-admin token against `POST` → still 403; admin `GET` → still 200. Along the way, hit and resolved a dev-loop issue: the container had the edited file on disk but Turbopack's file watcher wasn't detecting the change over the Windows bind mount, requiring `docker restart paddleocr-pfm-web-app` to pick it up — noted in case it recurs for future edits. With 9.1-9.3 all shipped, Flutter root task 7.1 (moving the product editor onto the v1 surface) is now fully unblocked — shipped 2026-07-10.
|
||||
|
||||
## Backend — OCR Pipeline & Accuracy
|
||||
|
||||
- **2.1** Built the Product/SKU scan classifier's model artifacts: `models/dinov2_index.pkl` (118/118 reference photos indexed across 16 SKU classes) and `models/produk-pfm-classifier-26n-100e-2026-07-08.pt` (+ `.onnx` export) — a YOLO classifier fine-tuned 100 epochs, 83.3% top-1 / 90% top-5 validation accuracy on the current (thin, 2-16 photos/class) dataset. Built via a one-off `docker run` from a freshly-rebuilt `pipeline-api` image (bare-metal training isn't viable on Windows — `paddlepaddle-gpu`'s wheel index is Linux-only). `pipeline-api` restarted and confirmed loading both models from logs. Also fixed `scripts/install-pipeline.sh`, which was missing `ultralytics`/`torch` — shipped 2026-07-08.
|
||||
- **2.1 (verification pass)** Ran a full browser walkthrough of `/scan-pfm` (classification, top-5, OCR expiry extraction + crop, SKU-master matching, Visual/Spotting Grid, Raw Response — all confirmed working with real data). Found and fixed a real bug: "Save Ground Truth" was returning success but silently writing into the `pfm-web-app` container's ephemeral filesystem instead of the host, because `/sources` wasn't a bind-mounted path in root `docker-compose.yml`. Added `./backend/sources:/sources` to the `pfm-web-app` service, recovered an orphaned entry via `docker cp`, and re-verified the save now persists to `backend/sources/product_manual_labels.json` on the host (confirmed the DO-flow's `manual_labels.json` save was fixed by the same change too) — shipped 2026-07-08.
|
||||
- **2.3** Ran the accuracy regression harness and discovered `sources/accuracy_report.md` was badly stale (claimed 75.04%; real current baseline is **95.10% overall, already at/above the 95% target** — added a staleness banner to that file). Root-caused every remaining mismatch by pulling raw OCR text from Postgres (`documents.layout_parsing_result`): the worst field, `plat` (67.6%), is almost entirely the license-plate region being classified as an image/seal by the layout model rather than OCR'd as text — not fixable in `parser.ts`. Found and fixed one genuine parser logic bug along the way: the "global pattern scanning fallback" could duplicate an already-correctly-extracted `noDO` value into a still-missing `noSO` field; fixed by excluding already-assigned values from that fallback's candidate pool (`pfm-web-app/src/utils/parser.ts`). Doesn't change the aggregate score (a wrong value and "Not Found" score the same) but stops a fabricated-looking wrong number from silently reaching the database. All 48 parser unit tests still pass — shipped 2026-07-08.
|
||||
- **Ad-hoc** Built custom expiry-date-based auto-rotation algorithm in Python classifier server (`classify_ocr_server.py`). The algorithm calculates the slant angle of the Expiry Date / Batch text line bounding box, automatically rotates the image to make it horizontal, and re-runs YOLO classification + PaddleOCR for maximum accuracy. Enhanced SKU matching database lookup to prioritize exact SKU matches with a score of 1.0, pinning them as the Best Match — shipped 2026-07-09.
|
||||
- **2.5** Retrained the Product/SKU scan classifier's model artifacts against the full current dataset, which had grown to 81 SKU classes / 2,493 photos (up from the original 16 classes / 118 photos the deployed model dated 2026-07-08 was actually trained on — the other 65 classes had photos but no trained weights). Rebuilt `models/dinov2_index.pkl` (now 2,493/2,493 photos indexed) and retrained the YOLO classifier 100 epochs on an RTX 2060 (real elapsed time 54m21s), publishing `models/produk-pfm-classifier-26n-100e-2026-07-14.pt`/`.onnx` at **85.8% top-1 / 94.4% top-5** validation accuracy across all 81 classes (up from 83.3%/90% on the old 16-class model). Along the way, fixed a real train/val split bug in `train_classifier.py`: `split_dataset()` previously shuffled and split individual image files, letting an augmented copy (`photo_aug_2.jpeg`) land in validation while its near-duplicate source stayed in training — inflating val accuracy with memorization instead of measuring generalization; now groups by source photo (stripping `_aug_N`) before shuffling and splitting 80/20. Verified via `docker compose up -d pipeline-api` + `docker logs`: "DINOv2 index loaded with 2493 reference images", "Using classifier weights: .../produk-pfm-classifier-26n-100e-2026-07-14.pt", "YOLO model loaded successfully" — the live service is confirmed serving the new 81-class model, not assumed from the newest-file-by-date fallback logic. Remaining gap toward the program's ±230-SKU target is dataset growth, not a pipeline limitation — shipped 2026-07-14.
|
||||
|
||||
## Backend — Postgres Data Layer
|
||||
|
||||
- **3.1** Wrapped the `ocr_items` delete-then-reinsert in `/api/parse` and `/api/v1/documents/[id]` PUT inside a DB transaction (`withTransaction` helper, `pfm-web-app/src/db/index.ts`) — a mid-loop insert failure now rolls back to the previous item set instead of leaving a document with a correct header but partial/missing items — shipped 2026-07-08. (Renumbered from root's `7.1` when this file split from root `docs/feature-list.md`.)
|
||||
- **3.2** Hashed `accounts.password` with `bcryptjs` (pure-JS, no native compile step — the `pfm-web-app` Docker image has no build toolchain). `db/init.ts` hashes the seed and idempotently migrates any pre-existing plaintext rows on every startup; `api/v1/auth/login/route.ts` now compares with `bcrypt.compareSync` and cleanly rejects missing credentials with a 401 instead of risking a raw-query edge case. Verified via `psql` (hash format) and `curl` (correct login succeeds, wrong/missing password returns 401) — shipped 2026-07-08.
|
||||
|
||||
## Docs & Workflow Integrity
|
||||
|
||||
- **5.1** Fixed stale doc claims in `SKILLS.md` (accuracy baseline pointer) and `CLAUDE.md` (Flutter auth claim and API base URL fallback) — shipped 2026-07-08.
|
||||
- **5.2** Refactored `plans/next-enhancements.md` to archive verbose `[DONE]` and `[CANCELLED]` task bodies into one-line stubs. Reduced the file size significantly, strictly enforcing the 256-line threshold rule for maintainability — shipped 2026-07-08.
|
||||
- **5.3** Amended `AGENTS.md` completion checklist with a doc-sync step to ensure architecture changes are synced back to documentation — shipped 2026-07-08.
|
||||
|
||||
## Product Scan — Ground Truth Annotation & Accuracy
|
||||
|
||||
- **6.1** Built standalone annotation page `manual-label-scan/page.tsx` for ground truth editing. Includes image browser, editable fields (`no_sku`, `nama_item`, `expiry_date`, `notes`), and a "Scan with AI" fill-blanks feature — shipped 2026-07-08.
|
||||
- **6.2** API + storage groundwork for scan annotation. Extended `api/manual-label-scan` with `GET` list mode and `DELETE`. Persisted uploaded scan photos as base64 images into `sources/product-test-images/`. Made the `scan-pfm` quick-save honest by allowing manual correction before save — shipped 2026-07-08.
|
||||
- **6.3** Built `backend/scripts/accuracy-check-scan.mts` mirroring the DO-harness architecture, measuring overall match rate plus per-field breakdown (`no_sku`, `expiry_date`) against the new stable labels — shipped 2026-07-08.
|
||||
- **6.4** Ported the DO-harness's auto-diff-vs-previous-run reporting into `accuracy-check-scan.mts`: every run now prints a Δ column per field per split (Training/Validation) vs the last `product_accuracy_history.jsonl` entry, and calls out field- and image-level regressions/improvements explicitly. Added classifier method (`dinov2_similarity`/`yolo_classifier`) distribution and average confidence as informational (non-scoring) context. Created the previously-missing `sources/product-test-images/README.md` documenting the validation-photo drop workflow — shipped 2026-07-13, user-directed `n` request to make algorithm tuning self-verifying.
|
||||
|
||||
### Master Data Management
|
||||
- **8.1 & 8.3 CRUD APIs and Web UI**: Created `/api/v1/master/stores` and `/api/v1/master/skus` endpoints alongside a Next.js Admin page (`/admin/master-data`) to visually manage the core reference data used by the OCR matching engine — shipped 2026-07-08.
|
||||
- **8.2 Auto-Provisioning Store Accounts**: Store creation now automatically securely hashes a default password ("123") and creates a paired login account, keeping store configuration perfectly in sync with the `accounts` table — shipped 2026-07-08.
|
||||
|
||||
## Auth — Store Accounts & Profile-Sourced Metadata
|
||||
|
||||
- **7.1** Seeded one account per store in `db/init.ts` during initialization by assigning `username = kode_toko` and a bcrypt-hashed default password `"123"`. Included `role` and `is_active` schema additions — shipped 2026-07-08.
|
||||
- **7.2** Enhanced authentication routing by modifying `POST /api/v1/auth/login` to perform a `LEFT JOIN` on `store_master`, returning the extended store profile alongside the token. Added a guard to reject login if `is_active = false`. Implemented a new `GET /api/v1/auth/me` endpoint to cleanly re-fetch the profile via token — shipped 2026-07-08.
|
||||
- **7.3** Created a reproducible `store_master` bootstrap logic in `db/init.ts` that reads from `sources/toko_aktif.json` idempotently on startup. Also correctly seeded the `WH_JOFFICE` head office to resolve the admin account foreign-key setup constraint — shipped 2026-07-08.
|
||||
|
||||
## Backend — Document Confirmation Gate & Data Hygiene
|
||||
|
||||
- **10.1** Added a `confirmed BOOLEAN NOT NULL DEFAULT TRUE` column to `documents` (`db/init.ts`, `ALTER TABLE ... ADD COLUMN IF NOT EXISTS` — grandfathers every pre-existing row so today's history didn't go empty after migration) and used it to separate "OCR finished" from "user confirmed": previously `GET /api/v1/documents` filtered only on `parsed = true`, which the backend sets synchronously right after upload — before the mobile user ever taps "Simpan & Konfirmasi" in the editor — so a scan captured, previewed, then backed out of (never confirmed) was already sitting in every entitled account's document list with blank/placeholder fields (root cause of `document_card.dart`'s "Staff Toko" fallback text on the Flutter side). `v1/documents/upload/route.ts` now explicitly inserts `confirmed = false` on every new upload; `v1/documents/[id]/route.ts`'s `PUT` handler is the *only* place that flips it to `true` (literally "the user confirmed"); `v1/documents/route.ts` (list) now filters `AND confirmed = true` unconditionally for every account including `admin` (no role special-casing, per explicit user decision); `v1/documents/[id]/route.ts`'s `GET`-by-id handler is deliberately untouched by the new filter so the mobile poller can keep seeing pending/unconfirmed documents mid-flow. `utils/document-mapper.ts`'s shared `DocumentRow`/`mapDocumentRow()` now carries `confirmed` through to all three call sites (list, GET-by-id, upload's dedup-hit branch) from one place. `parse/route.ts`'s own `INSERT ... ON CONFLICT (filename) DO UPDATE` statements (both DO and Product branches) were deliberately left untouched for `confirmed` — in the real mobile flow the upload route's INSERT always runs first, so this upsert always hits the `ON CONFLICT` branch, and since its `SET` clause doesn't mention `confirmed`, Postgres correctly leaves the existing value alone (verified this is correct, not an oversight). Verified live against the running Docker stack: uploaded a real DO photo as a store account without confirming it — absent from that store's list (and from `admin`'s) while `GET /documents/:id` still reported the correct `parseStatus`; `PUT` (confirm) made it appear immediately with the real submitted data; all 13 pre-existing rows carried `confirmed = true` after the migration ran — shipped 2026-07-10.
|
||||
- **10.2** Removed the fabricated Product Scan placeholder values `noPO: "PO-PRODUCT-001"`, `noSO: "1002003004"`, `noDO: "DO-PRODUCT-999"` (both the flat keys and the mirrored `header.no_po`/`no_so`/`no_do` sub-object) from `parse/route.ts`'s Product-scan branch, replacing them with empty strings — these are DO-specific concepts that don't apply to a product verification scan, and were never actually read by anything: `pdf_service.dart`'s Product receipt branch never prints them, and `product_editor_submit_logic.dart`'s `_submit()` builds its own `noPo`/`noSo`/`noDo` from the user's PO-link dropdown and batch selection, ignoring the stored values entirely. Same class of issue as the earlier G7 fix (fabricated data presented as if real) — low risk to remove since nothing meaningfully depended on the old values. Scope stayed narrow to exactly these three fields; `nama_driver`/`nama_penerima`'s "PRODUCT SCAN"/"STORE STAFF" placeholders were left alone as a deliberate fixed convention, not a fabricated document number. Verified via `curl`: a freshly-uploaded, unconfirmed Product Scan document's raw `GET /documents/:id` response now returns `no_po`/`no_so`/`no_do` as empty strings instead of the old fake values — shipped 2026-07-10.
|
||||
|
||||
## Backend — Single-Pass Product Classification
|
||||
|
||||
- **11.1** Eliminated the duplicate GPU classification pass on Product Scan (gap G3), sourced from user feedback that the review screen took noticeably longer to open than DO Scan's. `api/parse/route.ts`'s Product branch previously had its own separate, poorer inline classify call (kept only `top1_name`/`extracted_sku`), forcing the Flutter editor to re-run the entire classify+OCR pipeline a second time via `POST /api/v1/scan-product` just to get the top-5 candidate list and OCR-extracted expiry date. Now calls the same shared `classifyAndMatchProduct()` (`utils/product-scan.ts`) already used by that v1 route — one GPU call, richer result — and persists it under a new `metadata.productScan` JSONB key (no schema migration), surfaced by `document-mapper.ts` as a top-level `productScan` field on every GET response. Caught and fixed a real regression along the way: delegating to the shared function silently dropped the 90s pipeline timeout the old inline fetch had; added the same bound (`PIPELINE_TIMEOUT_MS`) directly inside `classifyAndMatchProduct()` so both callers — this route and the live `POST /api/v1/scan-product` (which never had the bound either) — are protected. Verified via `curl` with a genuinely fresh image/store combination (proving a real classify pass, not a dedup hit): took 9s, and the immediate `GET /documents/:id` response already contained 5 real `possibleMatches` and the extracted expiry date, before any editor interaction — shipped 2026-07-10.
|
||||
# Feature List (backend)
|
||||
|
||||
Structured log of shipped backend features, updated by the `n`/`next` workflow (see
|
||||
[AGENTS.md](../AGENTS.md) Part B) whenever a task in
|
||||
[plans/next-enhancements.md](../plans/next-enhancements.md) is marked `[DONE]`.
|
||||
Split out 2026-07-08 from root `docs/feature-list.md`'s backend sections — this file
|
||||
is the sole home for backend feature history going forward.
|
||||
|
||||
## Format
|
||||
|
||||
```
|
||||
## <Section / Module Name>
|
||||
|
||||
- **<task number>** <feature description> — shipped <date>
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## Existing Features (pre-kit)
|
||||
|
||||
Backfilled 2026-07-08 during adoption of this kit — these predate the `e`/`n`
|
||||
workflow and have no task numbers; see `git log` for real dates/history.
|
||||
|
||||
### Backend — Next.js API Gateway
|
||||
- Upload/parse/documents CRUD routes, GPU status endpoint, vLLM proxy, manual-label review tool.
|
||||
|
||||
### Backend — OCR Pipeline & Accuracy
|
||||
- PaddleOCR + vLLM classification pipeline with DB layout caching, table column-shift correction, date normalization, and an accuracy regression harness (`pfm-web-app/scripts/accuracy-check.mts`) — **95.10% overall as of 2026-07-08** (target 95% met; see task 2.3 below for the investigation and `CLAUDE.md`).
|
||||
|
||||
### Backend — Postgres Data Layer
|
||||
- Schema/init in `pfm-web-app/src/db/init.ts`, served via the canonical root `docker-compose.yml` stack.
|
||||
- **3.3** Added a standard `INDEX` on `documents(file_hash)` in `db/init.ts` to accelerate the upload deduplication queries without strictly enforcing uniqueness across different stores. Correspondingly updated the dedup query in `api/v1/documents/upload/route.ts` to scope duplicate detection by `kode_toko`. This fixes a conflict where one store could be incorrectly linked to another store's duplicate receipt image — shipped 2026-07-08.
|
||||
|
||||
### DevOps — Docker & Dev Tunnel
|
||||
- **4.1 Docker Compose Policy Documented**: Formalized the execution policy in `README.md` and `CLAUDE.md`, explicitly requiring the use of the `docker-compose.demo.yml` override (production build) for all client demonstrations and field testing to bypass the Next.js dev server bottleneck — shipped 2026-07-08.
|
||||
- **Docker Compose Dependency Gates**: Added strict Docker `healthcheck` gates (`Task 4.2`) blocking the `pfm-web-app` (Next.js) from starting until PostgreSQL and the VLLM models are initialized and fully healthy.
|
||||
- **Secure Tunnel Ingress**: Restructured `nginx.conf` and `start-dev-tunnel.ps1` (`Tasks 4.3, 4.5`) to expose a dedicated, restricted port (`8001`) that exclusively routes to `/api/v1/*`. This perfectly secures the development UI (`/scan-pfm`) and legacy routes from public exposure.
|
||||
- **Dead Config Pruning**: Stripped deprecated and redundant proxy blocks from the Nginx edge router (`Task 4.4`).
|
||||
|
||||
*(New features shipped via `n`/`next` go below, organized the same way, with task numbers.)*
|
||||
|
||||
## Backend — Next.js API Gateway
|
||||
|
||||
- **1.4** Enforced real 401 auth on `/api/v1/documents/*` (list, PUT-by-id, upload) — the actual production API surface, already fully supported by the Flutter client (real login + `Authorization: Bearer` on every request). Previously none of these three routes rejected a missing/invalid token; upload only optionally read it. Added the pre-existing `getAccountFromAuthHeader()` helper (`utils/auth.ts`) + a 401 guard to all three; `OPTIONS` (CORS preflight) untouched. The original task 1.3 (auth on the *classic* routes) was cancelled instead — those routes are dev-only web UI surface with no login flow, going away in production. Verified via `curl`: 401 with no token, success with a real token from `/api/v1/auth/login` — shipped 2026-07-08.
|
||||
- **1.5** Implemented per-store data scoping on `/api/v1/documents/*`. Added `kode_toko` column to `documents` table via `db/init.ts` migration. The upload route now binds `kode_toko` to documents upon creation. `GET /api/v1/documents` and `PUT /api/v1/documents/:id` enforce ownership checks (`kode_toko` matching) for `store` role accounts, while `admin` retains global access including legacy unassigned documents — shipped 2026-07-08.
|
||||
- **1.6** `GET /api/v1/health` Endpoint: Unauthenticated health probe verifying both PostgreSQL connectivity and Pipeline API HTTP reachability. Returns `HTTP 503` if any core dependency is down — shipped 2026-07-08.
|
||||
- **Ad-hoc** Connected `/scan-pfm` page with `/api/scan-pfm` route and enabled auto-trigger scanning on custom file upload, sample selection, thumbnail change, and canvas rotation. Supported both `image` and `image_base64` payload keys — shipped 2026-07-09.
|
||||
- **9.1** Added `GET /api/v1/documents/:id` (same 401/403 scoping as `PUT`), returning a single document — including still-unparsed rows — with a new `parseStatus: "pending"|"done"|"failed"` field, so the Flutter poller can move off scanning the entire list every 2s. Added `scan_mode`/`parse_error` columns to `documents` (`db/init.ts`, migrated via `ALTER TABLE ... ADD COLUMN IF NOT EXISTS` for already-running DBs). `scan_mode` is now persisted on upload (`v1/documents/upload/route.ts`) and on the classic `/api/parse` route's upserts (`COALESCE`, same pattern as `kode_toko`), and surfaced as `docType` on every GET response (`utils/document-mapper.ts`, a new shared helper extracted from the list route's inline mapping so list/by-id/dedup all agree) — falling back to the legacy `order_untuk == "PRODUCT SCAN"` sentinel for pre-existing rows with no `scan_mode`. `parse_error` is now recorded when the upload route's *internal* call to `/api/parse` itself fails to complete (network error or the 210s abort firing) — previously this was silently swallowed and the document stayed `parsed=false` forever with no signal, burning the client's full 260s timeout; `/api/parse`'s own existing pipeline-error fallback (`parsed=true` + "Not Found" placeholder) was already fine and is unchanged. Also fixed the dedup branch (a repeat upload of an already-seen file) to return the original document's real current state via the same mapper instead of a hardcoded empty stub. Verified via `docker compose up -d --build` + `curl`: schema migration applied cleanly to the live DB (confirmed via `psql`), DO and Product uploads both correctly persist `scan_mode` and surface it as `docType`, a dedup retry returns real header/items instead of an empty stub, `GET /:id` returns 401 (no token) / 403 (wrong store) / 404 (nonexistent id) / 200 (admin or owning store), and the list endpoint's existing filter/scoping is unchanged — shipped 2026-07-10.
|
||||
- **9.3** Added authenticated `POST /api/v1/scan-product`, the v1 equivalent of the classic dev-only `/api/scan-pfm` (unauthenticated, and unreachable off-LAN since task 4.5 restricted the public tunnel to `/api/v1/*`). Extracted the shared classify-and-match logic (Python classifier call + Levenshtein SKU matching against `sku_master`, top-5 scoring) out of `api/scan-pfm/route.ts` into a new `utils/product-scan.ts` (`classifyAndMatchProduct`, plus a `ClassifierError` class that preserves forwarding the classifier's own HTTP status instead of collapsing every failure to 500) so the classic route and the new v1 route share one implementation instead of duplicating it — the classic route's response shape, auth-free behavior, and desktop-only layout-parsing visualization are otherwise unchanged. The new route accepts **either** multipart (`image`/`file` field, matching the v1 upload route's convention) or a JSON `{image_base64}` body, is open to any authenticated account (not admin-gated, since this is what the mobile app itself calls), and wraps the result in the standard `{status, data}` envelope with `classification`, `ocr` (including `extracted_expired_date`), and `possibleMatches`. Verified via `curl` against the live stack with a real product photo: multipart upload and JSON-body variants both return identical, correct top-5 matches; no-token request returns 401; the classic `/api/scan-pfm` route's response (including `layoutParsingResult`) is unchanged post-refactor — shipped 2026-07-10.
|
||||
- **9.2** Relaxed `GET /api/v1/master/skus` (`master/skus/route.ts`) so any authenticated account can read the SKU master list, not just `admin` — the Flutter product editor needs this and previously had to string-hack its base URL to call the unauthenticated classic `GET /api/skus`, which task 4.5 had already removed from the public tunnel, breaking product scans off-LAN. Changed the guard from a combined `!account || role !== 'admin'` check (403 for both "no token" and "wrong role") to `!account` (correct 401) followed by an unconditional pass-through for any valid account; `POST` (SKU creation) is untouched, still admin-only, per the user's explicit choice between the two options this task flagged as undecided. No response-shape change. Verified via `curl` against the live stack with a real non-admin (`store` role) account's token: `GET` → 200 with real data; no token → 401 (was incorrectly 403 before this fix); the same non-admin token against `POST` → still 403; admin `GET` → still 200. Along the way, hit and resolved a dev-loop issue: the container had the edited file on disk but Turbopack's file watcher wasn't detecting the change over the Windows bind mount, requiring `docker restart paddleocr-pfm-web-app` to pick it up — noted in case it recurs for future edits. With 9.1-9.3 all shipped, Flutter root task 7.1 (moving the product editor onto the v1 surface) is now fully unblocked — shipped 2026-07-10.
|
||||
|
||||
## Backend — OCR Pipeline & Accuracy
|
||||
|
||||
- **2.1** Built the Product/SKU scan classifier's model artifacts: `models/dinov2_index.pkl` (118/118 reference photos indexed across 16 SKU classes) and `models/produk-pfm-classifier-26n-100e-2026-07-08.pt` (+ `.onnx` export) — a YOLO classifier fine-tuned 100 epochs, 83.3% top-1 / 90% top-5 validation accuracy on the current (thin, 2-16 photos/class) dataset. Built via a one-off `docker run` from a freshly-rebuilt `pipeline-api` image (bare-metal training isn't viable on Windows — `paddlepaddle-gpu`'s wheel index is Linux-only). `pipeline-api` restarted and confirmed loading both models from logs. Also fixed `scripts/install-pipeline.sh`, which was missing `ultralytics`/`torch` — shipped 2026-07-08.
|
||||
- **2.1 (verification pass)** Ran a full browser walkthrough of `/scan-pfm` (classification, top-5, OCR expiry extraction + crop, SKU-master matching, Visual/Spotting Grid, Raw Response — all confirmed working with real data). Found and fixed a real bug: "Save Ground Truth" was returning success but silently writing into the `pfm-web-app` container's ephemeral filesystem instead of the host, because `/sources` wasn't a bind-mounted path in root `docker-compose.yml`. Added `./backend/sources:/sources` to the `pfm-web-app` service, recovered an orphaned entry via `docker cp`, and re-verified the save now persists to `backend/sources/product_manual_labels.json` on the host (confirmed the DO-flow's `manual_labels.json` save was fixed by the same change too) — shipped 2026-07-08.
|
||||
- **2.3** Ran the accuracy regression harness and discovered `sources/accuracy_report.md` was badly stale (claimed 75.04%; real current baseline is **95.10% overall, already at/above the 95% target** — added a staleness banner to that file). Root-caused every remaining mismatch by pulling raw OCR text from Postgres (`documents.layout_parsing_result`): the worst field, `plat` (67.6%), is almost entirely the license-plate region being classified as an image/seal by the layout model rather than OCR'd as text — not fixable in `parser.ts`. Found and fixed one genuine parser logic bug along the way: the "global pattern scanning fallback" could duplicate an already-correctly-extracted `noDO` value into a still-missing `noSO` field; fixed by excluding already-assigned values from that fallback's candidate pool (`pfm-web-app/src/utils/parser.ts`). Doesn't change the aggregate score (a wrong value and "Not Found" score the same) but stops a fabricated-looking wrong number from silently reaching the database. All 48 parser unit tests still pass — shipped 2026-07-08.
|
||||
- **Ad-hoc** Built custom expiry-date-based auto-rotation algorithm in Python classifier server (`classify_ocr_server.py`). The algorithm calculates the slant angle of the Expiry Date / Batch text line bounding box, automatically rotates the image to make it horizontal, and re-runs YOLO classification + PaddleOCR for maximum accuracy. Enhanced SKU matching database lookup to prioritize exact SKU matches with a score of 1.0, pinning them as the Best Match — shipped 2026-07-09.
|
||||
- **2.5** Retrained the Product/SKU scan classifier's model artifacts against the full current dataset, which had grown to 81 SKU classes / 2,493 photos (up from the original 16 classes / 118 photos the deployed model dated 2026-07-08 was actually trained on — the other 65 classes had photos but no trained weights). Rebuilt `models/dinov2_index.pkl` (now 2,493/2,493 photos indexed) and retrained the YOLO classifier 100 epochs on an RTX 2060 (real elapsed time 54m21s), publishing `models/produk-pfm-classifier-26n-100e-2026-07-14.pt`/`.onnx` at **85.8% top-1 / 94.4% top-5** validation accuracy across all 81 classes (up from 83.3%/90% on the old 16-class model). Along the way, fixed a real train/val split bug in `train_classifier.py`: `split_dataset()` previously shuffled and split individual image files, letting an augmented copy (`photo_aug_2.jpeg`) land in validation while its near-duplicate source stayed in training — inflating val accuracy with memorization instead of measuring generalization; now groups by source photo (stripping `_aug_N`) before shuffling and splitting 80/20. Verified via `docker compose up -d pipeline-api` + `docker logs`: "DINOv2 index loaded with 2493 reference images", "Using classifier weights: .../produk-pfm-classifier-26n-100e-2026-07-14.pt", "YOLO model loaded successfully" — the live service is confirmed serving the new 81-class model, not assumed from the newest-file-by-date fallback logic. Remaining gap toward the program's ±230-SKU target is dataset growth, not a pipeline limitation — shipped 2026-07-14.
|
||||
|
||||
## Backend — Postgres Data Layer
|
||||
|
||||
- **3.1** Wrapped the `ocr_items` delete-then-reinsert in `/api/parse` and `/api/v1/documents/[id]` PUT inside a DB transaction (`withTransaction` helper, `pfm-web-app/src/db/index.ts`) — a mid-loop insert failure now rolls back to the previous item set instead of leaving a document with a correct header but partial/missing items — shipped 2026-07-08. (Renumbered from root's `7.1` when this file split from root `docs/feature-list.md`.)
|
||||
- **3.2** Hashed `accounts.password` with `bcryptjs` (pure-JS, no native compile step — the `pfm-web-app` Docker image has no build toolchain). `db/init.ts` hashes the seed and idempotently migrates any pre-existing plaintext rows on every startup; `api/v1/auth/login/route.ts` now compares with `bcrypt.compareSync` and cleanly rejects missing credentials with a 401 instead of risking a raw-query edge case. Verified via `psql` (hash format) and `curl` (correct login succeeds, wrong/missing password returns 401) — shipped 2026-07-08.
|
||||
|
||||
## Docs & Workflow Integrity
|
||||
|
||||
- **5.1** Fixed stale doc claims in `SKILLS.md` (accuracy baseline pointer) and `CLAUDE.md` (Flutter auth claim and API base URL fallback) — shipped 2026-07-08.
|
||||
- **5.2** Refactored `plans/next-enhancements.md` to archive verbose `[DONE]` and `[CANCELLED]` task bodies into one-line stubs. Reduced the file size significantly, strictly enforcing the 256-line threshold rule for maintainability — shipped 2026-07-08.
|
||||
- **5.3** Amended `AGENTS.md` completion checklist with a doc-sync step to ensure architecture changes are synced back to documentation — shipped 2026-07-08.
|
||||
|
||||
## Product Scan — Ground Truth Annotation & Accuracy
|
||||
|
||||
- **6.1** Built standalone annotation page `manual-label-scan/page.tsx` for ground truth editing. Includes image browser, editable fields (`no_sku`, `nama_item`, `expiry_date`, `notes`), and a "Scan with AI" fill-blanks feature — shipped 2026-07-08.
|
||||
- **6.2** API + storage groundwork for scan annotation. Extended `api/manual-label-scan` with `GET` list mode and `DELETE`. Persisted uploaded scan photos as base64 images into `sources/product-test-images/`. Made the `scan-pfm` quick-save honest by allowing manual correction before save — shipped 2026-07-08.
|
||||
- **6.3** Built `backend/scripts/accuracy-check-scan.mts` mirroring the DO-harness architecture, measuring overall match rate plus per-field breakdown (`no_sku`, `expiry_date`) against the new stable labels — shipped 2026-07-08.
|
||||
- **6.4** Ported the DO-harness's auto-diff-vs-previous-run reporting into `accuracy-check-scan.mts`: every run now prints a Δ column per field per split (Training/Validation) vs the last `product_accuracy_history.jsonl` entry, and calls out field- and image-level regressions/improvements explicitly. Added classifier method (`dinov2_similarity`/`yolo_classifier`) distribution and average confidence as informational (non-scoring) context. Created the previously-missing `sources/product-test-images/README.md` documenting the validation-photo drop workflow — shipped 2026-07-13, user-directed `n` request to make algorithm tuning self-verifying.
|
||||
|
||||
### Master Data Management
|
||||
- **8.1 & 8.3 CRUD APIs and Web UI**: Created `/api/v1/master/stores` and `/api/v1/master/skus` endpoints alongside a Next.js Admin page (`/admin/master-data`) to visually manage the core reference data used by the OCR matching engine — shipped 2026-07-08.
|
||||
- **8.2 Auto-Provisioning Store Accounts**: Store creation now automatically securely hashes a default password ("123") and creates a paired login account, keeping store configuration perfectly in sync with the `accounts` table — shipped 2026-07-08.
|
||||
|
||||
## Auth — Store Accounts & Profile-Sourced Metadata
|
||||
|
||||
- **7.1** Seeded one account per store in `db/init.ts` during initialization by assigning `username = kode_toko` and a bcrypt-hashed default password `"123"`. Included `role` and `is_active` schema additions — shipped 2026-07-08.
|
||||
- **7.2** Enhanced authentication routing by modifying `POST /api/v1/auth/login` to perform a `LEFT JOIN` on `store_master`, returning the extended store profile alongside the token. Added a guard to reject login if `is_active = false`. Implemented a new `GET /api/v1/auth/me` endpoint to cleanly re-fetch the profile via token — shipped 2026-07-08.
|
||||
- **7.3** Created a reproducible `store_master` bootstrap logic in `db/init.ts` that reads from `sources/toko_aktif.json` idempotently on startup. Also correctly seeded the `WH_JOFFICE` head office to resolve the admin account foreign-key setup constraint — shipped 2026-07-08.
|
||||
|
||||
## Backend — Document Confirmation Gate & Data Hygiene
|
||||
|
||||
- **10.1** Added a `confirmed BOOLEAN NOT NULL DEFAULT TRUE` column to `documents` (`db/init.ts`, `ALTER TABLE ... ADD COLUMN IF NOT EXISTS` — grandfathers every pre-existing row so today's history didn't go empty after migration) and used it to separate "OCR finished" from "user confirmed": previously `GET /api/v1/documents` filtered only on `parsed = true`, which the backend sets synchronously right after upload — before the mobile user ever taps "Simpan & Konfirmasi" in the editor — so a scan captured, previewed, then backed out of (never confirmed) was already sitting in every entitled account's document list with blank/placeholder fields (root cause of `document_card.dart`'s "Staff Toko" fallback text on the Flutter side). `v1/documents/upload/route.ts` now explicitly inserts `confirmed = false` on every new upload; `v1/documents/[id]/route.ts`'s `PUT` handler is the *only* place that flips it to `true` (literally "the user confirmed"); `v1/documents/route.ts` (list) now filters `AND confirmed = true` unconditionally for every account including `admin` (no role special-casing, per explicit user decision); `v1/documents/[id]/route.ts`'s `GET`-by-id handler is deliberately untouched by the new filter so the mobile poller can keep seeing pending/unconfirmed documents mid-flow. `utils/document-mapper.ts`'s shared `DocumentRow`/`mapDocumentRow()` now carries `confirmed` through to all three call sites (list, GET-by-id, upload's dedup-hit branch) from one place. `parse/route.ts`'s own `INSERT ... ON CONFLICT (filename) DO UPDATE` statements (both DO and Product branches) were deliberately left untouched for `confirmed` — in the real mobile flow the upload route's INSERT always runs first, so this upsert always hits the `ON CONFLICT` branch, and since its `SET` clause doesn't mention `confirmed`, Postgres correctly leaves the existing value alone (verified this is correct, not an oversight). Verified live against the running Docker stack: uploaded a real DO photo as a store account without confirming it — absent from that store's list (and from `admin`'s) while `GET /documents/:id` still reported the correct `parseStatus`; `PUT` (confirm) made it appear immediately with the real submitted data; all 13 pre-existing rows carried `confirmed = true` after the migration ran — shipped 2026-07-10.
|
||||
- **10.2** Removed the fabricated Product Scan placeholder values `noPO: "PO-PRODUCT-001"`, `noSO: "1002003004"`, `noDO: "DO-PRODUCT-999"` (both the flat keys and the mirrored `header.no_po`/`no_so`/`no_do` sub-object) from `parse/route.ts`'s Product-scan branch, replacing them with empty strings — these are DO-specific concepts that don't apply to a product verification scan, and were never actually read by anything: `pdf_service.dart`'s Product receipt branch never prints them, and `product_editor_submit_logic.dart`'s `_submit()` builds its own `noPo`/`noSo`/`noDo` from the user's PO-link dropdown and batch selection, ignoring the stored values entirely. Same class of issue as the earlier G7 fix (fabricated data presented as if real) — low risk to remove since nothing meaningfully depended on the old values. Scope stayed narrow to exactly these three fields; `nama_driver`/`nama_penerima`'s "PRODUCT SCAN"/"STORE STAFF" placeholders were left alone as a deliberate fixed convention, not a fabricated document number. Verified via `curl`: a freshly-uploaded, unconfirmed Product Scan document's raw `GET /documents/:id` response now returns `no_po`/`no_so`/`no_do` as empty strings instead of the old fake values — shipped 2026-07-10.
|
||||
|
||||
## Backend — Single-Pass Product Classification
|
||||
|
||||
- **11.1** Eliminated the duplicate GPU classification pass on Product Scan (gap G3), sourced from user feedback that the review screen took noticeably longer to open than DO Scan's. `api/parse/route.ts`'s Product branch previously had its own separate, poorer inline classify call (kept only `top1_name`/`extracted_sku`), forcing the Flutter editor to re-run the entire classify+OCR pipeline a second time via `POST /api/v1/scan-product` just to get the top-5 candidate list and OCR-extracted expiry date. Now calls the same shared `classifyAndMatchProduct()` (`utils/product-scan.ts`) already used by that v1 route — one GPU call, richer result — and persists it under a new `metadata.productScan` JSONB key (no schema migration), surfaced by `document-mapper.ts` as a top-level `productScan` field on every GET response. Caught and fixed a real regression along the way: delegating to the shared function silently dropped the 90s pipeline timeout the old inline fetch had; added the same bound (`PIPELINE_TIMEOUT_MS`) directly inside `classifyAndMatchProduct()` so both callers — this route and the live `POST /api/v1/scan-product` (which never had the bound either) — are protected. Verified via `curl` with a genuinely fresh image/store combination (proving a real classify pass, not a dedup hit): took 9s, and the immediate `GET /documents/:id` response already contained 5 real `possibleMatches` and the extracted expiry date, before any editor interaction — shipped 2026-07-10.
|
||||
+579
-579
File diff suppressed because it is too large.
Load diff
+214
-214
@@ -1,214 +1,214 @@
|
||||
# Product Scan (scan-pfm) — How It Works
|
||||
|
||||
End-to-end reference for the Product/SKU scanning feature: a photo of a Primafood
|
||||
product package goes in; the SKU class, product name, expiry date, and a ranked
|
||||
SKU-master match list come out. Written 2026-07-08 against the live code. Related:
|
||||
`plans/next-enhancements.md` §2 (build history) and §6 (ground-truth roadmap);
|
||||
`docs/feature-list.md` tasks 2.1/2.3.
|
||||
|
||||
## High-level flow
|
||||
|
||||
```mermaid
|
||||
flowchart LR
|
||||
A[Browser: /scan-pfm page] -->|"POST /api/scan-pfm {image_base64}"| B[Next.js gateway<br/>pfm-web-app :3000]
|
||||
B -->|"POST :8120/classify-ocr"| C[classify_ocr_server.py<br/>FastAPI, in pipeline-api]
|
||||
C --> C1[1. DINOv2 similarity<br/>fallback: YOLO classifier]
|
||||
C --> C2[2. PaddleOCR + regex<br/>SKU / expiry / name]
|
||||
C -->|"POST localhost:8090/layout-parsing<br/>promptLabel: spotting"| D[PaddleX pipeline<br/>same container]
|
||||
B -->|"POST :8090/layout-parsing"| D
|
||||
B -->|"SELECT sku_master"| E[(Postgres)]
|
||||
B -->|Levenshtein ranking| A
|
||||
```
|
||||
|
||||
Two processes live in the `paddleocr-pipeline-api` container, both started by
|
||||
`scripts/serve-pipeline.sh`: the PaddleX layout-parsing pipeline on **:8090**
|
||||
(shared with the DO flow; VL recognition goes out to the vLLM server on :8118) and
|
||||
`config/classify_ocr_server.py` on **:8120** (product scan only). The gateway
|
||||
reaches them via Docker DNS (`CLASSIFIER_SERVER_URL`, `PIPELINE_URL` in root
|
||||
`docker-compose.yml:87-88`); nginx (:8000) proxies `/scan-pfm` to the Next.js app.
|
||||
|
||||
## Request walkthrough
|
||||
|
||||
1. **Page** (`pfm-web-app/src/app/scan-pfm/page.tsx`, desktop-only test UI): pick a
|
||||
sample from the gallery (`GET /api/produk-pfm`) or upload/rotate a photo (rotation
|
||||
is done client-side on a canvas), then send it as a base64 data-URL.
|
||||
2. **Gateway** (`api/scan-pfm/route.ts`):
|
||||
- forwards `{image_base64}` to the classifier server (`/classify-ocr`);
|
||||
- separately calls the layout-parsing pipeline with `useLayoutDetection: true`
|
||||
for the Visual Grid tab's output images (failure here is non-fatal — logged,
|
||||
`layoutParsingResult` returns `null`);
|
||||
- loads the full `sku_master` table and ranks every SKU by **Levenshtein
|
||||
similarity between `nama_item` and the classifier's `top1_name`**
|
||||
(lowercased, alphanumerics only). Top 5 with score > 0.1 are returned;
|
||||
rank 1 gets `isBestMatch: true`. Note: `ocr.extracted_sku` and
|
||||
`ocr.extracted_product_name` are read but **not used** in this ranking —
|
||||
see Future recommendations.
|
||||
3. **Classifier server** (`config/classify_ocr_server.py`) does classification,
|
||||
OCR extraction, and visualization — detailed below — and returns
|
||||
`{classification, ocr}`.
|
||||
4. **Page renders** four tabs: Summary (classification card + top-5 override
|
||||
"Use" buttons + OCR fields + SKU matches), Visual Grid, Spotting Grid, Raw
|
||||
Response (JSON). "Save Ground Truth" posts to `/api/manual-label-scan`.
|
||||
|
||||
## Stage 1 — classification (which product is this?)
|
||||
|
||||
**Primary: DINOv2 similarity search** (`method: "dinov2_similarity"`). At startup
|
||||
the server loads `dinov2_vits14` **from `torch.hub` (network fetch on first run)**
|
||||
plus `models/dinov2_index.pkl` — precomputed L2-normalized 384-dim embeddings of
|
||||
all 118 reference photos across 16 SKU class folders. Per request: embed the query
|
||||
image (resize 224², ImageNet normalization), dot-product against all reference
|
||||
embeddings (= cosine similarity), then aggregate **per class = max similarity of
|
||||
any reference photo in that class**. Classes sorted by similarity become
|
||||
`all_probabilities`. Caveat: these "confidences" are cosine similarities, **not
|
||||
probabilities** — they don't sum to 1 and are typically all high (0.4–0.9);
|
||||
compare relatively, not against an absolute threshold.
|
||||
|
||||
**Fallback: YOLO classifier** (`method: "yolo_classifier"`) — only when DINOv2 is
|
||||
unavailable (no index/model) or throws. A fine-tuned `yolo26n-cls` checkpoint;
|
||||
its `all_probabilities` are real softmax probabilities. Weights are
|
||||
**auto-discovered**: `CLASSIFIER_MODEL_PATH` env wins; otherwise the newest
|
||||
`produk-pfm-classifier-26n-*e-*.pt` in `models/` by (date-in-filename, mtime) —
|
||||
so retraining just drops a new dated file, no config change.
|
||||
|
||||
If both are unavailable, `classification` carries an `error` field instead.
|
||||
|
||||
## Stage 2 — OCR extraction (SKU, expiry date, product name)
|
||||
|
||||
PaddleOCR (`lang='en'`, textline orientation on) produces `rec_texts` lines +
|
||||
`rec_polys` boxes. Three extractors run over the lines:
|
||||
|
||||
- **SKU** (`extract_sku`): first 8-digit number anywhere; else first 7–9 digit
|
||||
number. (Primafood SKUs are 8 digits, printed near the label top.)
|
||||
- **Expiry date** (`extract_expired_date`): each line is first noise-cleaned
|
||||
(`clean_date_line`: `1)`→`0`, `()`→`0`, `B8/8B/88`→`BB` before digits, o→0,
|
||||
I/l/|→1, S→5, Z→2, B→8 when digit-flanked, plus `012`/`112` month-misread
|
||||
repairs), then a **6-level priority cascade** runs: (1) BB/EXP-keyword line
|
||||
with compact `DDMMYYYY`; (2) keyword line with spaced `DD MM YYYY`; (3)
|
||||
keyword + 6–8 digit run; (3.5) keyword line, lenient noisy match; (4) any line
|
||||
spaced date; (5) any line compact `DDMMYYYY` — skipping lines that look like a
|
||||
SKU-on-product-name; (6) legacy formats (slashes, `05 MAR 2027`). Recognized
|
||||
keywords: `EXP`, `EXPIRED`, `TGL`, `EXPIRY`, `BBD`, `BEST BEFORE`, `BB`,
|
||||
`BAIK DIGUNAKAN`. Output normalized to `DD/MM/YYYY`.
|
||||
- **Product name** (`extract_product_name`): longest line containing a brand/
|
||||
product keyword (FIESTA, CHAMP, OKEY, AKUMO, ASIMO, NUGGET, SOSIS, …) after
|
||||
stripping SKU digits and date fragments; falls back to the classifier's
|
||||
`top1_name`, then the longest non-numeric line, then `"Unknown Product"`.
|
||||
|
||||
Visualization artifacts built server-side: `vis_image_base64` (all OCR boxes
|
||||
drawn teal `TEXT`, the expiry line amber `EXP`, on the orientation-corrected
|
||||
image so boxes align), `expired_date_crop_base64` (padded crop of the expiry
|
||||
line for eyeball verification — `find_expired_crop_index` prefers the box whose
|
||||
digits actually contain the date), and `spotting_image_base64` (a second
|
||||
pipeline call with `promptLabel: "spotting"`, no layout detection).
|
||||
|
||||
## Endpoint reference
|
||||
|
||||
| Endpoint | Where | Purpose |
|
||||
|---|---|---|
|
||||
| `POST /api/scan-pfm` | gateway | Main scan. Body `{image_base64}` (data-URL ok). Returns `{classification, ocr, possibleMatches[], layoutParsingResult}` |
|
||||
| `POST http://paddleocr-pipeline-api:8120/classify-ocr` | classifier server | Internal. Body `{image_base64}`. Returns `{classification: {top1_name, top1_confidence, all_probabilities[], method}, ocr: {text_lines[], extracted_product_name, extracted_sku, extracted_expired_date, expired_line_index, expired_source_line, expired_date_crop_base64, vis_image_base64, spotting_image_base64}}` |
|
||||
| `GET /api/produk-pfm` | gateway | Gallery: SKU folders under `public/produk-pfm/foto-kemasan-v2/` with image + thumb URLs |
|
||||
| `GET/POST /api/manual-label-scan` | gateway | Ground-truth read/upsert to `sources/product_manual_labels.json` (host-visible via the `./backend/sources:/sources` mount) |
|
||||
| `POST :8090/layout-parsing` | pipeline | Shared PaddleX pipeline; used here for Visual Grid images and (with `promptLabel: "spotting"`) the Spotting Grid |
|
||||
| `/scan-pfm` | nginx :8000 | Proxies the page to Next.js :3000 |
|
||||
|
||||
`possibleMatches[]` items: `{no_sku, nama_item, score, yoloSimilarity, isBestMatch}` —
|
||||
`score` currently equals `yoloSimilarity` (name-vs-name Levenshtein, 0..1).
|
||||
|
||||
## Model artifacts & retraining
|
||||
|
||||
| File (`pfm-web-app/public/produk-pfm/`) | What |
|
||||
|---|---|
|
||||
| `foto-kemasan-v2/<SKU or class>/…` | Reference photo dataset — 81 classes, 2,493 photos (target ~230 SKU) |
|
||||
| `models/dinov2_index.pkl` | DINOv2 embeddings + metadata (rebuild after adding photos) — currently indexes all 2,493 photos across 81 classes |
|
||||
| `models/produk-pfm-classifier-26n-100e-2026-07-14.pt` / `.onnx` | Fine-tuned YOLO classifier (85.8% top-1 / 94.4% top-5 val across all 81 classes; retrained 2026-07-14, 54m21s on an RTX 2060, up from the prior 2026-07-08 model's 83.3%/90% on only 16 classes) |
|
||||
| `index_dinov2.py` | Rebuilds the pickle index from `foto-kemasan-v2/` |
|
||||
| `train_classifier.py` | Splits 80/20 into `yolo_dataset/`, fine-tunes `yolo26n-cls.pt` (default 100 epochs, `--imgsz 224`), writes a dated checkpoint |
|
||||
|
||||
**Retraining procedure (Windows host — bare-metal doesn't work here,
|
||||
`paddlepaddle-gpu` wheels are Linux-only):** add photos to `foto-kemasan-v2/`,
|
||||
`docker compose build pipeline-api` from the **repo root**, run a one-off
|
||||
`docker run --gpus all` from that image with `models/` mounted **writable** (the
|
||||
live service mounts it `:ro`), run `index_dinov2.py` then
|
||||
`train_classifier.py train --imgsz 224`, then `docker compose restart
|
||||
pipeline-api`. From Git Bash prefix `MSYS_NO_PATHCONV=1` or `/app/...` arguments
|
||||
get mangled. Verify in `docker logs`: "DINOv2 index loaded with N reference
|
||||
images", "Using classifier weights: <new dated file>". Full worked example:
|
||||
`plans/next-enhancements.md` task 2.1.
|
||||
|
||||
## Accuracy regression harness
|
||||
|
||||
`backend/scripts/accuracy-check-scan.mts` — mirrors the DO-flow's
|
||||
`pfm-web-app/scripts/accuracy-check.mts`. Hits the live `/api/scan-pfm` for
|
||||
every labeled image in `sources/product_manual_labels.json`, checks 3 fields
|
||||
(`no_sku`, `nama_item`, `expiry_date`) against ground truth, and splits into:
|
||||
- **Training Set** — gallery photos under `foto-kemasan-v2/` (the classifier's
|
||||
own reference images; scores here measure memorization, not generalization).
|
||||
- **Validation Set** — flat filenames, scored from the frozen
|
||||
`sources/product-test-images-fixed/` snapshot (renamed `<index> <no_sku>.<ext>`,
|
||||
built by `scripts/freeze-validation-set.mjs`) so a rerun always grades the
|
||||
same 79 images regardless of what's since been dropped into the live-intake
|
||||
`sources/product-test-images/` folder. See each folder's `README.md` — the
|
||||
live folder documents the drop-photo → label → re-run-freeze-script workflow
|
||||
via `/manual-label-scan`; the fixed folder documents the freeze/promote step
|
||||
and flags 5 SKUs (12010801, 12012504, 12130504, 13050101, 15040102) whose
|
||||
only available photo was already used to train the classifier, so their
|
||||
scores aren't a clean held-out result.
|
||||
|
||||
Every run appends to `sources/product_accuracy_history.jsonl` and **auto-diffs
|
||||
against the previous run**: the printed summary shows a Δ column per field per
|
||||
split, flags field/image-level regressions and improvements, and reports
|
||||
classifier method (`dinov2_similarity`/`yolo_classifier`) distribution +
|
||||
average confidence as informational context (not scored pass/fail, since
|
||||
DINOv2's "confidence" is a raw cosine similarity, not a calibrated
|
||||
probability — see Stage 1 above). This is what makes it safe to tune
|
||||
`classify_ocr_server.py` and immediately see whether a change helped or hurt.
|
||||
|
||||
```bash
|
||||
node scripts/accuracy-check-scan.mts # from backend/
|
||||
```
|
||||
|
||||
## Operational notes
|
||||
|
||||
- **Env vars**: `CLASSIFIER_SERVER_URL`, `PIPELINE_URL` (gateway, set in compose);
|
||||
`CLASSIFIER_MODELS_DIR`, `CLASSIFIER_MODEL_PATH` (classifier server overrides).
|
||||
The gateway's in-code default `PIPELINE_URL` (`localhost:7871`) is stale — the
|
||||
compose env always overrides it in Docker.
|
||||
- **Startup order/health**: the classifier server loads DINOv2 (torch.hub →
|
||||
needs network/cache), YOLO, and PaddleOCR at import time; until done, :8120
|
||||
refuses connections and `/api/scan-pfm` 500s. No healthcheck exists yet (plan
|
||||
task 4.2 / 1.6).
|
||||
- **GPU**: DINOv2 + YOLO + PaddleOCR share the container/GPU with the PaddleX
|
||||
pipeline; all are small (ViT-S/14, nano YOLO) next to the vLLM server's
|
||||
footprint, but they do add VRAM on the same `PIPELINE_DEVICE`.
|
||||
- **Failure isolation**: layout-vis and spotting calls are best-effort
|
||||
(`null`/absent on failure); classification and OCR errors surface as `error`
|
||||
fields inside their sections rather than failing the whole scan.
|
||||
|
||||
## Known gaps & future recommendations
|
||||
|
||||
Tracked ones (see `plans/next-enhancements.md`):
|
||||
- **Dataset thinness**: 2–16 photos/class caps both classifiers; every new real
|
||||
photo (especially non-studio, in-warehouse shots) matters. The harness above
|
||||
already reports gallery (training) vs. held-out (validation) accuracy
|
||||
separately, and as of 2026-07-14 the Validation Set has 79 labeled images
|
||||
(74 genuinely held out, 5 flagged trained-on — see above) — the first real
|
||||
(non-zero) Validation Set numbers.
|
||||
|
||||
Additional recommendations (not yet tasks — promote via `e`/`n` when wanted):
|
||||
1. ~~Use `extracted_sku` in match ranking.~~ **Done** — `product-scan.ts`'s
|
||||
`classifyAndMatchProduct` already pins rank 1 to an exact `no_sku` match
|
||||
(score forced to 1.0) before falling back to name similarity.
|
||||
2. **Fuse DINOv2 and YOLO instead of primary/fallback** (e.g. agreement boosts
|
||||
confidence; disagreement flags for review) — cheap, both already load.
|
||||
3. **"Not a known product" handling**: DINOv2 always returns *some* class; add a
|
||||
minimum-similarity threshold below which the response says unknown rather
|
||||
than confidently misclassifying a foreign package.
|
||||
4. **Pin the DINOv2 backbone offline** (vendor the weights or pre-bake the
|
||||
torch.hub cache into the image) — startup currently depends on an internet
|
||||
fetch on cold cache, bad for on-prem deploys.
|
||||
5. **Batch/lot number extraction** — explicitly out of scope so far (plan §2
|
||||
note); if requested, follow the expiry-date regex-cascade pattern.
|
||||
6. **Mobile**: no web mobile page by design (task 2.2 cancelled) — real mobile
|
||||
scanning should go through the Flutter app calling `POST /api/scan-pfm`
|
||||
(would need an authenticated `/api/v1` variant; the classic route has no auth).
|
||||
# Product Scan (scan-pfm) — How It Works
|
||||
|
||||
End-to-end reference for the Product/SKU scanning feature: a photo of a Primafood
|
||||
product package goes in; the SKU class, product name, expiry date, and a ranked
|
||||
SKU-master match list come out. Written 2026-07-08 against the live code. Related:
|
||||
`plans/next-enhancements.md` §2 (build history) and §6 (ground-truth roadmap);
|
||||
`docs/feature-list.md` tasks 2.1/2.3.
|
||||
|
||||
## High-level flow
|
||||
|
||||
```mermaid
|
||||
flowchart LR
|
||||
A[Browser: /scan-pfm page] -->|"POST /api/scan-pfm {image_base64}"| B[Next.js gateway<br/>pfm-web-app :3000]
|
||||
B -->|"POST :8120/classify-ocr"| C[classify_ocr_server.py<br/>FastAPI, in pipeline-api]
|
||||
C --> C1[1. DINOv2 similarity<br/>fallback: YOLO classifier]
|
||||
C --> C2[2. PaddleOCR + regex<br/>SKU / expiry / name]
|
||||
C -->|"POST localhost:8090/layout-parsing<br/>promptLabel: spotting"| D[PaddleX pipeline<br/>same container]
|
||||
B -->|"POST :8090/layout-parsing"| D
|
||||
B -->|"SELECT sku_master"| E[(Postgres)]
|
||||
B -->|Levenshtein ranking| A
|
||||
```
|
||||
|
||||
Two processes live in the `paddleocr-pipeline-api` container, both started by
|
||||
`scripts/serve-pipeline.sh`: the PaddleX layout-parsing pipeline on **:8090**
|
||||
(shared with the DO flow; VL recognition goes out to the vLLM server on :8118) and
|
||||
`config/classify_ocr_server.py` on **:8120** (product scan only). The gateway
|
||||
reaches them via Docker DNS (`CLASSIFIER_SERVER_URL`, `PIPELINE_URL` in root
|
||||
`docker-compose.yml:87-88`); nginx (:8000) proxies `/scan-pfm` to the Next.js app.
|
||||
|
||||
## Request walkthrough
|
||||
|
||||
1. **Page** (`pfm-web-app/src/app/scan-pfm/page.tsx`, desktop-only test UI): pick a
|
||||
sample from the gallery (`GET /api/produk-pfm`) or upload/rotate a photo (rotation
|
||||
is done client-side on a canvas), then send it as a base64 data-URL.
|
||||
2. **Gateway** (`api/scan-pfm/route.ts`):
|
||||
- forwards `{image_base64}` to the classifier server (`/classify-ocr`);
|
||||
- separately calls the layout-parsing pipeline with `useLayoutDetection: true`
|
||||
for the Visual Grid tab's output images (failure here is non-fatal — logged,
|
||||
`layoutParsingResult` returns `null`);
|
||||
- loads the full `sku_master` table and ranks every SKU by **Levenshtein
|
||||
similarity between `nama_item` and the classifier's `top1_name`**
|
||||
(lowercased, alphanumerics only). Top 5 with score > 0.1 are returned;
|
||||
rank 1 gets `isBestMatch: true`. Note: `ocr.extracted_sku` and
|
||||
`ocr.extracted_product_name` are read but **not used** in this ranking —
|
||||
see Future recommendations.
|
||||
3. **Classifier server** (`config/classify_ocr_server.py`) does classification,
|
||||
OCR extraction, and visualization — detailed below — and returns
|
||||
`{classification, ocr}`.
|
||||
4. **Page renders** four tabs: Summary (classification card + top-5 override
|
||||
"Use" buttons + OCR fields + SKU matches), Visual Grid, Spotting Grid, Raw
|
||||
Response (JSON). "Save Ground Truth" posts to `/api/manual-label-scan`.
|
||||
|
||||
## Stage 1 — classification (which product is this?)
|
||||
|
||||
**Primary: DINOv2 similarity search** (`method: "dinov2_similarity"`). At startup
|
||||
the server loads `dinov2_vits14` **from `torch.hub` (network fetch on first run)**
|
||||
plus `models/dinov2_index.pkl` — precomputed L2-normalized 384-dim embeddings of
|
||||
all 118 reference photos across 16 SKU class folders. Per request: embed the query
|
||||
image (resize 224², ImageNet normalization), dot-product against all reference
|
||||
embeddings (= cosine similarity), then aggregate **per class = max similarity of
|
||||
any reference photo in that class**. Classes sorted by similarity become
|
||||
`all_probabilities`. Caveat: these "confidences" are cosine similarities, **not
|
||||
probabilities** — they don't sum to 1 and are typically all high (0.4–0.9);
|
||||
compare relatively, not against an absolute threshold.
|
||||
|
||||
**Fallback: YOLO classifier** (`method: "yolo_classifier"`) — only when DINOv2 is
|
||||
unavailable (no index/model) or throws. A fine-tuned `yolo26n-cls` checkpoint;
|
||||
its `all_probabilities` are real softmax probabilities. Weights are
|
||||
**auto-discovered**: `CLASSIFIER_MODEL_PATH` env wins; otherwise the newest
|
||||
`produk-pfm-classifier-26n-*e-*.pt` in `models/` by (date-in-filename, mtime) —
|
||||
so retraining just drops a new dated file, no config change.
|
||||
|
||||
If both are unavailable, `classification` carries an `error` field instead.
|
||||
|
||||
## Stage 2 — OCR extraction (SKU, expiry date, product name)
|
||||
|
||||
PaddleOCR (`lang='en'`, textline orientation on) produces `rec_texts` lines +
|
||||
`rec_polys` boxes. Three extractors run over the lines:
|
||||
|
||||
- **SKU** (`extract_sku`): first 8-digit number anywhere; else first 7–9 digit
|
||||
number. (Primafood SKUs are 8 digits, printed near the label top.)
|
||||
- **Expiry date** (`extract_expired_date`): each line is first noise-cleaned
|
||||
(`clean_date_line`: `1)`→`0`, `()`→`0`, `B8/8B/88`→`BB` before digits, o→0,
|
||||
I/l/|→1, S→5, Z→2, B→8 when digit-flanked, plus `012`/`112` month-misread
|
||||
repairs), then a **6-level priority cascade** runs: (1) BB/EXP-keyword line
|
||||
with compact `DDMMYYYY`; (2) keyword line with spaced `DD MM YYYY`; (3)
|
||||
keyword + 6–8 digit run; (3.5) keyword line, lenient noisy match; (4) any line
|
||||
spaced date; (5) any line compact `DDMMYYYY` — skipping lines that look like a
|
||||
SKU-on-product-name; (6) legacy formats (slashes, `05 MAR 2027`). Recognized
|
||||
keywords: `EXP`, `EXPIRED`, `TGL`, `EXPIRY`, `BBD`, `BEST BEFORE`, `BB`,
|
||||
`BAIK DIGUNAKAN`. Output normalized to `DD/MM/YYYY`.
|
||||
- **Product name** (`extract_product_name`): longest line containing a brand/
|
||||
product keyword (FIESTA, CHAMP, OKEY, AKUMO, ASIMO, NUGGET, SOSIS, …) after
|
||||
stripping SKU digits and date fragments; falls back to the classifier's
|
||||
`top1_name`, then the longest non-numeric line, then `"Unknown Product"`.
|
||||
|
||||
Visualization artifacts built server-side: `vis_image_base64` (all OCR boxes
|
||||
drawn teal `TEXT`, the expiry line amber `EXP`, on the orientation-corrected
|
||||
image so boxes align), `expired_date_crop_base64` (padded crop of the expiry
|
||||
line for eyeball verification — `find_expired_crop_index` prefers the box whose
|
||||
digits actually contain the date), and `spotting_image_base64` (a second
|
||||
pipeline call with `promptLabel: "spotting"`, no layout detection).
|
||||
|
||||
## Endpoint reference
|
||||
|
||||
| Endpoint | Where | Purpose |
|
||||
|---|---|---|
|
||||
| `POST /api/scan-pfm` | gateway | Main scan. Body `{image_base64}` (data-URL ok). Returns `{classification, ocr, possibleMatches[], layoutParsingResult}` |
|
||||
| `POST http://paddleocr-pipeline-api:8120/classify-ocr` | classifier server | Internal. Body `{image_base64}`. Returns `{classification: {top1_name, top1_confidence, all_probabilities[], method}, ocr: {text_lines[], extracted_product_name, extracted_sku, extracted_expired_date, expired_line_index, expired_source_line, expired_date_crop_base64, vis_image_base64, spotting_image_base64}}` |
|
||||
| `GET /api/produk-pfm` | gateway | Gallery: SKU folders under `public/produk-pfm/foto-kemasan-v2/` with image + thumb URLs |
|
||||
| `GET/POST /api/manual-label-scan` | gateway | Ground-truth read/upsert to `sources/product_manual_labels.json` (host-visible via the `./backend/sources:/sources` mount) |
|
||||
| `POST :8090/layout-parsing` | pipeline | Shared PaddleX pipeline; used here for Visual Grid images and (with `promptLabel: "spotting"`) the Spotting Grid |
|
||||
| `/scan-pfm` | nginx :8000 | Proxies the page to Next.js :3000 |
|
||||
|
||||
`possibleMatches[]` items: `{no_sku, nama_item, score, yoloSimilarity, isBestMatch}` —
|
||||
`score` currently equals `yoloSimilarity` (name-vs-name Levenshtein, 0..1).
|
||||
|
||||
## Model artifacts & retraining
|
||||
|
||||
| File (`pfm-web-app/public/produk-pfm/`) | What |
|
||||
|---|---|
|
||||
| `foto-kemasan-v2/<SKU or class>/…` | Reference photo dataset — 81 classes, 2,493 photos (target ~230 SKU) |
|
||||
| `models/dinov2_index.pkl` | DINOv2 embeddings + metadata (rebuild after adding photos) — currently indexes all 2,493 photos across 81 classes |
|
||||
| `models/produk-pfm-classifier-26n-100e-2026-07-14.pt` / `.onnx` | Fine-tuned YOLO classifier (85.8% top-1 / 94.4% top-5 val across all 81 classes; retrained 2026-07-14, 54m21s on an RTX 2060, up from the prior 2026-07-08 model's 83.3%/90% on only 16 classes) |
|
||||
| `index_dinov2.py` | Rebuilds the pickle index from `foto-kemasan-v2/` |
|
||||
| `train_classifier.py` | Splits 80/20 into `yolo_dataset/`, fine-tunes `yolo26n-cls.pt` (default 100 epochs, `--imgsz 224`), writes a dated checkpoint |
|
||||
|
||||
**Retraining procedure (Windows host — bare-metal doesn't work here,
|
||||
`paddlepaddle-gpu` wheels are Linux-only):** add photos to `foto-kemasan-v2/`,
|
||||
`docker compose build pipeline-api` from the **repo root**, run a one-off
|
||||
`docker run --gpus all` from that image with `models/` mounted **writable** (the
|
||||
live service mounts it `:ro`), run `index_dinov2.py` then
|
||||
`train_classifier.py train --imgsz 224`, then `docker compose restart
|
||||
pipeline-api`. From Git Bash prefix `MSYS_NO_PATHCONV=1` or `/app/...` arguments
|
||||
get mangled. Verify in `docker logs`: "DINOv2 index loaded with N reference
|
||||
images", "Using classifier weights: <new dated file>". Full worked example:
|
||||
`plans/next-enhancements.md` task 2.1.
|
||||
|
||||
## Accuracy regression harness
|
||||
|
||||
`backend/scripts/accuracy-check-scan.mts` — mirrors the DO-flow's
|
||||
`pfm-web-app/scripts/accuracy-check.mts`. Hits the live `/api/scan-pfm` for
|
||||
every labeled image in `sources/product_manual_labels.json`, checks 3 fields
|
||||
(`no_sku`, `nama_item`, `expiry_date`) against ground truth, and splits into:
|
||||
- **Training Set** — gallery photos under `foto-kemasan-v2/` (the classifier's
|
||||
own reference images; scores here measure memorization, not generalization).
|
||||
- **Validation Set** — flat filenames, scored from the frozen
|
||||
`sources/product-test-images-fixed/` snapshot (renamed `<index> <no_sku>.<ext>`,
|
||||
built by `scripts/freeze-validation-set.mjs`) so a rerun always grades the
|
||||
same 79 images regardless of what's since been dropped into the live-intake
|
||||
`sources/product-test-images/` folder. See each folder's `README.md` — the
|
||||
live folder documents the drop-photo → label → re-run-freeze-script workflow
|
||||
via `/manual-label-scan`; the fixed folder documents the freeze/promote step
|
||||
and flags 5 SKUs (12010801, 12012504, 12130504, 13050101, 15040102) whose
|
||||
only available photo was already used to train the classifier, so their
|
||||
scores aren't a clean held-out result.
|
||||
|
||||
Every run appends to `sources/product_accuracy_history.jsonl` and **auto-diffs
|
||||
against the previous run**: the printed summary shows a Δ column per field per
|
||||
split, flags field/image-level regressions and improvements, and reports
|
||||
classifier method (`dinov2_similarity`/`yolo_classifier`) distribution +
|
||||
average confidence as informational context (not scored pass/fail, since
|
||||
DINOv2's "confidence" is a raw cosine similarity, not a calibrated
|
||||
probability — see Stage 1 above). This is what makes it safe to tune
|
||||
`classify_ocr_server.py` and immediately see whether a change helped or hurt.
|
||||
|
||||
```bash
|
||||
node scripts/accuracy-check-scan.mts # from backend/
|
||||
```
|
||||
|
||||
## Operational notes
|
||||
|
||||
- **Env vars**: `CLASSIFIER_SERVER_URL`, `PIPELINE_URL` (gateway, set in compose);
|
||||
`CLASSIFIER_MODELS_DIR`, `CLASSIFIER_MODEL_PATH` (classifier server overrides).
|
||||
The gateway's in-code default `PIPELINE_URL` (`localhost:7871`) is stale — the
|
||||
compose env always overrides it in Docker.
|
||||
- **Startup order/health**: the classifier server loads DINOv2 (torch.hub →
|
||||
needs network/cache), YOLO, and PaddleOCR at import time; until done, :8120
|
||||
refuses connections and `/api/scan-pfm` 500s. No healthcheck exists yet (plan
|
||||
task 4.2 / 1.6).
|
||||
- **GPU**: DINOv2 + YOLO + PaddleOCR share the container/GPU with the PaddleX
|
||||
pipeline; all are small (ViT-S/14, nano YOLO) next to the vLLM server's
|
||||
footprint, but they do add VRAM on the same `PIPELINE_DEVICE`.
|
||||
- **Failure isolation**: layout-vis and spotting calls are best-effort
|
||||
(`null`/absent on failure); classification and OCR errors surface as `error`
|
||||
fields inside their sections rather than failing the whole scan.
|
||||
|
||||
## Known gaps & future recommendations
|
||||
|
||||
Tracked ones (see `plans/next-enhancements.md`):
|
||||
- **Dataset thinness**: 2–16 photos/class caps both classifiers; every new real
|
||||
photo (especially non-studio, in-warehouse shots) matters. The harness above
|
||||
already reports gallery (training) vs. held-out (validation) accuracy
|
||||
separately, and as of 2026-07-14 the Validation Set has 79 labeled images
|
||||
(74 genuinely held out, 5 flagged trained-on — see above) — the first real
|
||||
(non-zero) Validation Set numbers.
|
||||
|
||||
Additional recommendations (not yet tasks — promote via `e`/`n` when wanted):
|
||||
1. ~~Use `extracted_sku` in match ranking.~~ **Done** — `product-scan.ts`'s
|
||||
`classifyAndMatchProduct` already pins rank 1 to an exact `no_sku` match
|
||||
(score forced to 1.0) before falling back to name similarity.
|
||||
2. **Fuse DINOv2 and YOLO instead of primary/fallback** (e.g. agreement boosts
|
||||
confidence; disagreement flags for review) — cheap, both already load.
|
||||
3. **"Not a known product" handling**: DINOv2 always returns *some* class; add a
|
||||
minimum-similarity threshold below which the response says unknown rather
|
||||
than confidently misclassifying a foreign package.
|
||||
4. **Pin the DINOv2 backbone offline** (vendor the weights or pre-bake the
|
||||
torch.hub cache into the image) — startup currently depends on an internet
|
||||
fetch on cold cache, bad for on-prem deploys.
|
||||
5. **Batch/lot number extraction** — explicitly out of scope so far (plan §2
|
||||
note); if requested, follow the expiry-date regex-cascade pattern.
|
||||
6. **Mobile**: no web mobile page by design (task 2.2 cancelled) — real mobile
|
||||
scanning should go through the Flutter app calling `POST /api/scan-pfm`
|
||||
(would need an authenticated `/api/v1` variant; the classic route has no auth).
|
||||
+138
-138
@@ -1,138 +1,138 @@
|
||||
# vLLM Service — Full Reference
|
||||
|
||||
Detail split out of `../AGENTS.md` (2026-07-08, to keep that file under the
|
||||
Agents Settings Kit's 256-line threshold once the `e`/`n` workflow was appended
|
||||
to it). `AGENTS.md` keeps the short version — architecture, quick start, the
|
||||
env var table, file map — and links here for everything else.
|
||||
|
||||
## Issue recording — naming and template
|
||||
|
||||
```
|
||||
issues/{NN}-{slug}.md
|
||||
```
|
||||
|
||||
| Part | Rule | Example |
|
||||
|------|------|---------|
|
||||
| `{NN}` | Two-digit running number (`01`, `02`, …). Increment from the highest existing file. | `03` |
|
||||
| `{slug}` | Lowercase kebab-case summary of the problem | `gpu-memory-startup-failure` |
|
||||
|
||||
Full example: `issues/04-gpu-memory-startup-failure.md`
|
||||
|
||||
### File template
|
||||
|
||||
```markdown
|
||||
# Issue {NN}: {Short title}
|
||||
|
||||
## Problem
|
||||
What failed, with exact error message or symptom.
|
||||
|
||||
## Context
|
||||
Environment, command run, relevant config (`.env`, `config/vllm_config.yaml`).
|
||||
|
||||
## Solution
|
||||
What fixed it, or current workaround / open status.
|
||||
|
||||
## References
|
||||
Links, related issue files, or AGENTS.md sections.
|
||||
```
|
||||
|
||||
Check `issues/` for the next number:
|
||||
|
||||
```bash
|
||||
ls issues/*.md 2>/dev/null | sort
|
||||
```
|
||||
|
||||
## Client usage
|
||||
|
||||
After the server is running:
|
||||
|
||||
```bash
|
||||
# CLI
|
||||
uv run paddleocr doc_parser \
|
||||
--input https://paddle-model-ecology.bj.bcebos.com/paddlex/imgs/demo_image/paddleocr_vl_demo.png \
|
||||
--vl_rec_backend vllm-server \
|
||||
--vl_rec_server_url http://localhost:8118/v1
|
||||
```
|
||||
|
||||
```python
|
||||
from paddleocr import PaddleOCRVL
|
||||
|
||||
pipeline = PaddleOCRVL(
|
||||
vl_rec_backend="vllm-server",
|
||||
vl_rec_server_url="http://127.0.0.1:8118/v1",
|
||||
)
|
||||
output = pipeline.predict("path/to/image.png")
|
||||
```
|
||||
|
||||
Note: The full PaddleOCR-VL client should run in a **separate** environment if it needs PaddlePaddle GPU + Transformers. This repo is the isolated vLLM server only.
|
||||
|
||||
## Tuning vLLM
|
||||
|
||||
Edit `config/vllm_config.yaml`:
|
||||
|
||||
```yaml
|
||||
gpu-memory-utilization: 0.8
|
||||
max-num-seqs: 128
|
||||
```
|
||||
|
||||
Reference: [PaddleOCR-VL vLLM parameter tuning](https://www.paddleocr.ai/latest/en/version3.x/pipeline_usage/PaddleOCR-VL.html#331-server-side-parameter-adjustment)
|
||||
|
||||
## Troubleshooting
|
||||
|
||||
See `issues/` for full write-ups. Quick pointers:
|
||||
|
||||
| Symptom | Issue file |
|
||||
|---------|------------|
|
||||
| `paddleocr install_genai_server_deps` / `No module named pip` | [01-genai-server-deps-pip-in-uv-venv.md](../issues/01-genai-server-deps-pip-in-uv-venv.md) |
|
||||
| flash-attn wheel incompatible with Python version | [02-flash-attn-wheel-python-version-mismatch.md](../issues/02-flash-attn-wheel-python-version-mismatch.md) |
|
||||
| `uv pip` targets wrong venv from another project | [03-active-virtual-env-from-other-project.md](../issues/03-active-virtual-env-from-other-project.md) |
|
||||
| Free memory below `gpu-memory-utilization` on startup | [04-gpu-memory-startup-failure.md](../issues/04-gpu-memory-startup-failure.md) |
|
||||
| `TokenizersBackend has no attribute all_special_tokens_extended` | [05-transformers-tokenizers-incompatibility.md](../issues/05-transformers-tokenizers-incompatibility.md) |
|
||||
| Extracted images not shown in Gradio demo (raw base64 in markdown) | [06-extracted-images-raw-base64-not-displayed.md](../issues/06-extracted-images-raw-base64-not-displayed.md) |
|
||||
|
||||
### flash-attn build failures
|
||||
|
||||
Install the prebuilt wheel after `uv sync` (see `scripts/install.sh`):
|
||||
|
||||
```bash
|
||||
FLASH_ATTN_WHEEL="https://github.com/mjun0812/flash-attention-prebuild-wheels/releases/download/v0.3.14/flash_attn-2.8.2+cu128torch2.8-cp312-cp312-linux_x86_64.whl" \
|
||||
./scripts/install.sh
|
||||
```
|
||||
|
||||
Pick the wheel matching your Python and CUDA versions from [flash-attention prebuild wheels](https://mjunya.com/flash-attention-prebuild-wheels/). Details: [02-flash-attn-wheel-python-version-mismatch.md](../issues/02-flash-attn-wheel-python-version-mismatch.md).
|
||||
|
||||
Note: `paddleocr install_genai_server_deps` uses `pip` internally and is incompatible with uv-managed venvs. See [01-genai-server-deps-pip-in-uv-venv.md](../issues/01-genai-server-deps-pip-in-uv-venv.md). This repo installs the vLLM stack via `uv sync` + `uv pip`.
|
||||
|
||||
### `TokenizersBackend has no attribute all_special_tokens_extended`
|
||||
|
||||
Pin transformers (already in `pyproject.toml`):
|
||||
|
||||
```bash
|
||||
uv pip install "transformers==4.57.6"
|
||||
```
|
||||
|
||||
See [05-transformers-tokenizers-incompatibility.md](../issues/05-transformers-tokenizers-incompatibility.md).
|
||||
|
||||
### Do not install `paddlepaddle-gpu` in this venv
|
||||
|
||||
vLLM and PaddlePaddle GPU conflict. This server env uses `paddleocr[doc-parser]` without Paddle GPU.
|
||||
|
||||
### GPU memory on startup
|
||||
|
||||
If vLLM reports free memory below `gpu-memory-utilization`, either:
|
||||
|
||||
- Set `CUDA_VISIBLE_DEVICES` to a less-busy GPU
|
||||
- Lower `gpu-memory-utilization` in `config/vllm_config.yaml` (e.g. `0.75` or `0.7`)
|
||||
|
||||
See [04-gpu-memory-startup-failure.md](../issues/04-gpu-memory-startup-failure.md).
|
||||
|
||||
### Health check
|
||||
|
||||
```bash
|
||||
curl -s http://localhost:8118/v1/models | jq .
|
||||
```
|
||||
|
||||
## References
|
||||
|
||||
- [PaddleOCR-VL usage tutorial](https://www.paddleocr.ai/latest/en/version3.x/pipeline_usage/PaddleOCR-VL.html)
|
||||
- [PaddleOCR genai_server FAQ](https://github.com/PaddlePaddle/PaddleOCR/discussions/16822)
|
||||
# vLLM Service — Full Reference
|
||||
|
||||
Detail split out of `../AGENTS.md` (2026-07-08, to keep that file under the
|
||||
Agents Settings Kit's 256-line threshold once the `e`/`n` workflow was appended
|
||||
to it). `AGENTS.md` keeps the short version — architecture, quick start, the
|
||||
env var table, file map — and links here for everything else.
|
||||
|
||||
## Issue recording — naming and template
|
||||
|
||||
```
|
||||
issues/{NN}-{slug}.md
|
||||
```
|
||||
|
||||
| Part | Rule | Example |
|
||||
|------|------|---------|
|
||||
| `{NN}` | Two-digit running number (`01`, `02`, …). Increment from the highest existing file. | `03` |
|
||||
| `{slug}` | Lowercase kebab-case summary of the problem | `gpu-memory-startup-failure` |
|
||||
|
||||
Full example: `issues/04-gpu-memory-startup-failure.md`
|
||||
|
||||
### File template
|
||||
|
||||
```markdown
|
||||
# Issue {NN}: {Short title}
|
||||
|
||||
## Problem
|
||||
What failed, with exact error message or symptom.
|
||||
|
||||
## Context
|
||||
Environment, command run, relevant config (`.env`, `config/vllm_config.yaml`).
|
||||
|
||||
## Solution
|
||||
What fixed it, or current workaround / open status.
|
||||
|
||||
## References
|
||||
Links, related issue files, or AGENTS.md sections.
|
||||
```
|
||||
|
||||
Check `issues/` for the next number:
|
||||
|
||||
```bash
|
||||
ls issues/*.md 2>/dev/null | sort
|
||||
```
|
||||
|
||||
## Client usage
|
||||
|
||||
After the server is running:
|
||||
|
||||
```bash
|
||||
# CLI
|
||||
uv run paddleocr doc_parser \
|
||||
--input https://paddle-model-ecology.bj.bcebos.com/paddlex/imgs/demo_image/paddleocr_vl_demo.png \
|
||||
--vl_rec_backend vllm-server \
|
||||
--vl_rec_server_url http://localhost:8118/v1
|
||||
```
|
||||
|
||||
```python
|
||||
from paddleocr import PaddleOCRVL
|
||||
|
||||
pipeline = PaddleOCRVL(
|
||||
vl_rec_backend="vllm-server",
|
||||
vl_rec_server_url="http://127.0.0.1:8118/v1",
|
||||
)
|
||||
output = pipeline.predict("path/to/image.png")
|
||||
```
|
||||
|
||||
Note: The full PaddleOCR-VL client should run in a **separate** environment if it needs PaddlePaddle GPU + Transformers. This repo is the isolated vLLM server only.
|
||||
|
||||
## Tuning vLLM
|
||||
|
||||
Edit `config/vllm_config.yaml`:
|
||||
|
||||
```yaml
|
||||
gpu-memory-utilization: 0.8
|
||||
max-num-seqs: 128
|
||||
```
|
||||
|
||||
Reference: [PaddleOCR-VL vLLM parameter tuning](https://www.paddleocr.ai/latest/en/version3.x/pipeline_usage/PaddleOCR-VL.html#331-server-side-parameter-adjustment)
|
||||
|
||||
## Troubleshooting
|
||||
|
||||
See `issues/` for full write-ups. Quick pointers:
|
||||
|
||||
| Symptom | Issue file |
|
||||
|---------|------------|
|
||||
| `paddleocr install_genai_server_deps` / `No module named pip` | [01-genai-server-deps-pip-in-uv-venv.md](../issues/01-genai-server-deps-pip-in-uv-venv.md) |
|
||||
| flash-attn wheel incompatible with Python version | [02-flash-attn-wheel-python-version-mismatch.md](../issues/02-flash-attn-wheel-python-version-mismatch.md) |
|
||||
| `uv pip` targets wrong venv from another project | [03-active-virtual-env-from-other-project.md](../issues/03-active-virtual-env-from-other-project.md) |
|
||||
| Free memory below `gpu-memory-utilization` on startup | [04-gpu-memory-startup-failure.md](../issues/04-gpu-memory-startup-failure.md) |
|
||||
| `TokenizersBackend has no attribute all_special_tokens_extended` | [05-transformers-tokenizers-incompatibility.md](../issues/05-transformers-tokenizers-incompatibility.md) |
|
||||
| Extracted images not shown in Gradio demo (raw base64 in markdown) | [06-extracted-images-raw-base64-not-displayed.md](../issues/06-extracted-images-raw-base64-not-displayed.md) |
|
||||
|
||||
### flash-attn build failures
|
||||
|
||||
Install the prebuilt wheel after `uv sync` (see `scripts/install.sh`):
|
||||
|
||||
```bash
|
||||
FLASH_ATTN_WHEEL="https://github.com/mjun0812/flash-attention-prebuild-wheels/releases/download/v0.3.14/flash_attn-2.8.2+cu128torch2.8-cp312-cp312-linux_x86_64.whl" \
|
||||
./scripts/install.sh
|
||||
```
|
||||
|
||||
Pick the wheel matching your Python and CUDA versions from [flash-attention prebuild wheels](https://mjunya.com/flash-attention-prebuild-wheels/). Details: [02-flash-attn-wheel-python-version-mismatch.md](../issues/02-flash-attn-wheel-python-version-mismatch.md).
|
||||
|
||||
Note: `paddleocr install_genai_server_deps` uses `pip` internally and is incompatible with uv-managed venvs. See [01-genai-server-deps-pip-in-uv-venv.md](../issues/01-genai-server-deps-pip-in-uv-venv.md). This repo installs the vLLM stack via `uv sync` + `uv pip`.
|
||||
|
||||
### `TokenizersBackend has no attribute all_special_tokens_extended`
|
||||
|
||||
Pin transformers (already in `pyproject.toml`):
|
||||
|
||||
```bash
|
||||
uv pip install "transformers==4.57.6"
|
||||
```
|
||||
|
||||
See [05-transformers-tokenizers-incompatibility.md](../issues/05-transformers-tokenizers-incompatibility.md).
|
||||
|
||||
### Do not install `paddlepaddle-gpu` in this venv
|
||||
|
||||
vLLM and PaddlePaddle GPU conflict. This server env uses `paddleocr[doc-parser]` without Paddle GPU.
|
||||
|
||||
### GPU memory on startup
|
||||
|
||||
If vLLM reports free memory below `gpu-memory-utilization`, either:
|
||||
|
||||
- Set `CUDA_VISIBLE_DEVICES` to a less-busy GPU
|
||||
- Lower `gpu-memory-utilization` in `config/vllm_config.yaml` (e.g. `0.75` or `0.7`)
|
||||
|
||||
See [04-gpu-memory-startup-failure.md](../issues/04-gpu-memory-startup-failure.md).
|
||||
|
||||
### Health check
|
||||
|
||||
```bash
|
||||
curl -s http://localhost:8118/v1/models | jq .
|
||||
```
|
||||
|
||||
## References
|
||||
|
||||
- [PaddleOCR-VL usage tutorial](https://www.paddleocr.ai/latest/en/version3.x/pipeline_usage/PaddleOCR-VL.html)
|
||||
- [PaddleOCR genai_server FAQ](https://github.com/PaddlePaddle/PaddleOCR/discussions/16822)
|
||||
+206
-206
@@ -1,206 +1,206 @@
|
||||
import json
|
||||
import os
|
||||
import pandas as pd
|
||||
from openpyxl.styles import Font, PatternFill, Alignment, Border, Side
|
||||
from openpyxl.utils import get_column_letter
|
||||
|
||||
def main():
|
||||
jsonl_file = "/tmp/test_images_results.jsonl"
|
||||
xlsx_file = "/tmp/test_images_report.xlsx"
|
||||
|
||||
if not os.path.exists(jsonl_file):
|
||||
print(f"Error: JSONL file not found at {jsonl_file}")
|
||||
return
|
||||
|
||||
documents = []
|
||||
items = []
|
||||
|
||||
with open(jsonl_file, "r") as f:
|
||||
for idx, line in enumerate(f):
|
||||
if not line.strip():
|
||||
continue
|
||||
try:
|
||||
data = json.loads(line)
|
||||
except Exception as e:
|
||||
print(f"Skipping line due to parse error: {e}")
|
||||
continue
|
||||
|
||||
filename = data.get("filename", "N/A")
|
||||
status = data.get("status", "N/A")
|
||||
tilt = data.get("tilt", "N/A")
|
||||
unwarped = data.get("unwarped", "N/A")
|
||||
metadata = data.get("metadata", {})
|
||||
|
||||
no_po = metadata.get("noPO", "N/A")
|
||||
no_so = metadata.get("noSO", "N/A")
|
||||
no_do = metadata.get("noDO", "N/A")
|
||||
tanggal = metadata.get("tanggal", "N/A")
|
||||
customer = metadata.get("customerInfo", "N/A")
|
||||
store = metadata.get("orderUntuk", "N/A")
|
||||
alamat = metadata.get("alamat", "N/A")
|
||||
plat = metadata.get("platTruk", "N/A")
|
||||
items_list = data.get("items", [])
|
||||
|
||||
# Add to document list
|
||||
documents.append({
|
||||
"No": idx + 1,
|
||||
"Filename": filename,
|
||||
"Status": status,
|
||||
"Tilt (Degrees)": tilt,
|
||||
"Auto-Rotated/Unwarped": unwarped,
|
||||
"PO Number": no_po,
|
||||
"SO Number": no_so,
|
||||
"DO Number": no_do,
|
||||
"Date": tanggal,
|
||||
"Customer": customer,
|
||||
"Store Match": store,
|
||||
"Alamat": alamat,
|
||||
"Plat Nomor": plat,
|
||||
"Items Count": len(items_list)
|
||||
})
|
||||
|
||||
# Add items to items list
|
||||
for item in items_list:
|
||||
items.append({
|
||||
"Filename": filename,
|
||||
"Kode Barang (SKU)": item.get("kodeBarang", "N/A"),
|
||||
"Nama Barang": item.get("namaBarang", "N/A"),
|
||||
"Banyak (Qty)": item.get("banyak", ""),
|
||||
"Jumlah (Unit)": item.get("jumlah", "")
|
||||
})
|
||||
|
||||
df_docs = pd.DataFrame(documents)
|
||||
df_items = pd.DataFrame(items)
|
||||
|
||||
# Style definitions
|
||||
font_family = "Segoe UI"
|
||||
header_font = Font(name=font_family, size=11, bold=True, color="FFFFFF")
|
||||
regular_font = Font(name=font_family, size=10)
|
||||
bold_font = Font(name=font_family, size=10, bold=True)
|
||||
|
||||
# Fill colors
|
||||
header_fill = PatternFill(start_color="1F4E78", end_color="1F4E78", fill_type="solid") # Dark Blue
|
||||
zebra_fill = PatternFill(start_color="F2F5F8", end_color="F2F5F8", fill_type="solid") # Very light blue-gray
|
||||
success_fill = PatternFill(start_color="E2EFDA", end_color="E2EFDA", fill_type="solid") # Light green
|
||||
error_fill = PatternFill(start_color="FCE4D6", end_color="FCE4D6", fill_type="solid") # Light orange
|
||||
|
||||
# Alignments
|
||||
center_align = Alignment(horizontal="center", vertical="center")
|
||||
left_align = Alignment(horizontal="left", vertical="center")
|
||||
right_align = Alignment(horizontal="right", vertical="center")
|
||||
|
||||
# Borders
|
||||
thin_side = Side(border_style="thin", color="D9D9D9")
|
||||
thick_bottom = Side(border_style="medium", color="1F4E78")
|
||||
cell_border = Border(left=thin_side, right=thin_side, top=thin_side, bottom=thin_side)
|
||||
|
||||
with pd.ExcelWriter(xlsx_file, engine='openpyxl') as writer:
|
||||
df_docs.to_excel(writer, sheet_name='Document Summary', index=False)
|
||||
df_items.to_excel(writer, sheet_name='Parsed Items', index=False)
|
||||
|
||||
workbook = writer.book
|
||||
|
||||
# 1. Style Document Summary Sheet
|
||||
sheet1 = workbook['Document Summary']
|
||||
sheet1.views.sheetView[0].showGridLines = True
|
||||
|
||||
# Style Header Row
|
||||
for col_idx in range(1, len(df_docs.columns) + 1):
|
||||
cell = sheet1.cell(row=1, column=col_idx)
|
||||
cell.font = header_font
|
||||
cell.fill = header_fill
|
||||
cell.alignment = center_align
|
||||
cell.border = Border(left=thin_side, right=thin_side, top=thin_side, bottom=thick_bottom)
|
||||
|
||||
# Style Data Rows
|
||||
for row_idx in range(2, len(df_docs) + 2):
|
||||
# Check status for color coding
|
||||
status_val = sheet1.cell(row=row_idx, column=3).value
|
||||
row_fill = success_fill if status_val == "Success" else (error_fill if status_val == "Failed" or status_val == "Error" else None)
|
||||
|
||||
# Apply Zebra stripe if no status color
|
||||
if not row_fill and row_idx % 2 == 0:
|
||||
row_fill = zebra_fill
|
||||
|
||||
for col_idx in range(1, len(df_docs.columns) + 1):
|
||||
cell = sheet1.cell(row=row_idx, column=col_idx)
|
||||
cell.font = regular_font
|
||||
cell.border = cell_border
|
||||
|
||||
# Apply alignments based on column content
|
||||
if col_idx in [1, 3, 4, 5, 9, 13, 14]: # No, Status, Tilt, Auto-rotated, Date, Plat, Items Count
|
||||
cell.alignment = center_align
|
||||
else:
|
||||
cell.alignment = left_align
|
||||
|
||||
if row_fill:
|
||||
cell.fill = row_fill
|
||||
|
||||
# Format tilt with degree symbol
|
||||
if col_idx == 4 and cell.value != "N/A" and cell.value is not None:
|
||||
try:
|
||||
cell.value = float(cell.value)
|
||||
cell.number_format = '0.00"°"'
|
||||
except ValueError:
|
||||
pass
|
||||
|
||||
# Auto-adjust column width for Sheet 1
|
||||
for col in sheet1.columns:
|
||||
max_len = 0
|
||||
for cell in col:
|
||||
val_str = str(cell.value or '')
|
||||
# Exclude long text like Alamat from width sizing
|
||||
if cell.column in [12]: # Alamat
|
||||
max_len = max(max_len, min(len(val_str), 30))
|
||||
else:
|
||||
max_len = max(max_len, len(val_str))
|
||||
col_letter = get_column_letter(col[0].column)
|
||||
sheet1.column_dimensions[col_letter].width = max(max_len + 3, 10)
|
||||
|
||||
sheet1.row_dimensions[1].height = 25
|
||||
for r in range(2, len(df_docs) + 2):
|
||||
sheet1.row_dimensions[r].height = 20
|
||||
|
||||
# 2. Style Parsed Items Sheet
|
||||
sheet2 = workbook['Parsed Items']
|
||||
sheet2.views.sheetView[0].showGridLines = True
|
||||
|
||||
# Style Header Row
|
||||
for col_idx in range(1, len(df_items.columns) + 1):
|
||||
cell = sheet2.cell(row=1, column=col_idx)
|
||||
cell.font = header_font
|
||||
cell.fill = header_fill
|
||||
cell.alignment = center_align
|
||||
cell.border = Border(left=thin_side, right=thin_side, top=thin_side, bottom=thick_bottom)
|
||||
|
||||
# Style Data Rows
|
||||
for row_idx in range(2, len(df_items) + 2):
|
||||
row_fill = zebra_fill if row_idx % 2 == 0 else None
|
||||
for col_idx in range(1, len(df_items.columns) + 1):
|
||||
cell = sheet2.cell(row=row_idx, column=col_idx)
|
||||
cell.font = regular_font
|
||||
cell.border = cell_border
|
||||
|
||||
# Alignments
|
||||
if col_idx in [2, 4, 5]: # SKU, Qty, Unit
|
||||
cell.alignment = center_align
|
||||
else:
|
||||
cell.alignment = left_align
|
||||
|
||||
if row_fill:
|
||||
cell.fill = row_fill
|
||||
|
||||
# Auto-adjust column width for Sheet 2
|
||||
for col in sheet2.columns:
|
||||
max_len = max(len(str(cell.value or '')) for cell in col)
|
||||
col_letter = get_column_letter(col[0].column)
|
||||
sheet2.column_dimensions[col_letter].width = max(max_len + 3, 10)
|
||||
|
||||
sheet2.row_dimensions[1].height = 25
|
||||
for r in range(2, len(df_items) + 2):
|
||||
sheet2.row_dimensions[r].height = 20
|
||||
|
||||
print("Premium Excel report generated successfully!")
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
import json
|
||||
import os
|
||||
import pandas as pd
|
||||
from openpyxl.styles import Font, PatternFill, Alignment, Border, Side
|
||||
from openpyxl.utils import get_column_letter
|
||||
|
||||
def main():
|
||||
jsonl_file = "/tmp/test_images_results.jsonl"
|
||||
xlsx_file = "/tmp/test_images_report.xlsx"
|
||||
|
||||
if not os.path.exists(jsonl_file):
|
||||
print(f"Error: JSONL file not found at {jsonl_file}")
|
||||
return
|
||||
|
||||
documents = []
|
||||
items = []
|
||||
|
||||
with open(jsonl_file, "r") as f:
|
||||
for idx, line in enumerate(f):
|
||||
if not line.strip():
|
||||
continue
|
||||
try:
|
||||
data = json.loads(line)
|
||||
except Exception as e:
|
||||
print(f"Skipping line due to parse error: {e}")
|
||||
continue
|
||||
|
||||
filename = data.get("filename", "N/A")
|
||||
status = data.get("status", "N/A")
|
||||
tilt = data.get("tilt", "N/A")
|
||||
unwarped = data.get("unwarped", "N/A")
|
||||
metadata = data.get("metadata", {})
|
||||
|
||||
no_po = metadata.get("noPO", "N/A")
|
||||
no_so = metadata.get("noSO", "N/A")
|
||||
no_do = metadata.get("noDO", "N/A")
|
||||
tanggal = metadata.get("tanggal", "N/A")
|
||||
customer = metadata.get("customerInfo", "N/A")
|
||||
store = metadata.get("orderUntuk", "N/A")
|
||||
alamat = metadata.get("alamat", "N/A")
|
||||
plat = metadata.get("platTruk", "N/A")
|
||||
items_list = data.get("items", [])
|
||||
|
||||
# Add to document list
|
||||
documents.append({
|
||||
"No": idx + 1,
|
||||
"Filename": filename,
|
||||
"Status": status,
|
||||
"Tilt (Degrees)": tilt,
|
||||
"Auto-Rotated/Unwarped": unwarped,
|
||||
"PO Number": no_po,
|
||||
"SO Number": no_so,
|
||||
"DO Number": no_do,
|
||||
"Date": tanggal,
|
||||
"Customer": customer,
|
||||
"Store Match": store,
|
||||
"Alamat": alamat,
|
||||
"Plat Nomor": plat,
|
||||
"Items Count": len(items_list)
|
||||
})
|
||||
|
||||
# Add items to items list
|
||||
for item in items_list:
|
||||
items.append({
|
||||
"Filename": filename,
|
||||
"Kode Barang (SKU)": item.get("kodeBarang", "N/A"),
|
||||
"Nama Barang": item.get("namaBarang", "N/A"),
|
||||
"Banyak (Qty)": item.get("banyak", ""),
|
||||
"Jumlah (Unit)": item.get("jumlah", "")
|
||||
})
|
||||
|
||||
df_docs = pd.DataFrame(documents)
|
||||
df_items = pd.DataFrame(items)
|
||||
|
||||
# Style definitions
|
||||
font_family = "Segoe UI"
|
||||
header_font = Font(name=font_family, size=11, bold=True, color="FFFFFF")
|
||||
regular_font = Font(name=font_family, size=10)
|
||||
bold_font = Font(name=font_family, size=10, bold=True)
|
||||
|
||||
# Fill colors
|
||||
header_fill = PatternFill(start_color="1F4E78", end_color="1F4E78", fill_type="solid") # Dark Blue
|
||||
zebra_fill = PatternFill(start_color="F2F5F8", end_color="F2F5F8", fill_type="solid") # Very light blue-gray
|
||||
success_fill = PatternFill(start_color="E2EFDA", end_color="E2EFDA", fill_type="solid") # Light green
|
||||
error_fill = PatternFill(start_color="FCE4D6", end_color="FCE4D6", fill_type="solid") # Light orange
|
||||
|
||||
# Alignments
|
||||
center_align = Alignment(horizontal="center", vertical="center")
|
||||
left_align = Alignment(horizontal="left", vertical="center")
|
||||
right_align = Alignment(horizontal="right", vertical="center")
|
||||
|
||||
# Borders
|
||||
thin_side = Side(border_style="thin", color="D9D9D9")
|
||||
thick_bottom = Side(border_style="medium", color="1F4E78")
|
||||
cell_border = Border(left=thin_side, right=thin_side, top=thin_side, bottom=thin_side)
|
||||
|
||||
with pd.ExcelWriter(xlsx_file, engine='openpyxl') as writer:
|
||||
df_docs.to_excel(writer, sheet_name='Document Summary', index=False)
|
||||
df_items.to_excel(writer, sheet_name='Parsed Items', index=False)
|
||||
|
||||
workbook = writer.book
|
||||
|
||||
# 1. Style Document Summary Sheet
|
||||
sheet1 = workbook['Document Summary']
|
||||
sheet1.views.sheetView[0].showGridLines = True
|
||||
|
||||
# Style Header Row
|
||||
for col_idx in range(1, len(df_docs.columns) + 1):
|
||||
cell = sheet1.cell(row=1, column=col_idx)
|
||||
cell.font = header_font
|
||||
cell.fill = header_fill
|
||||
cell.alignment = center_align
|
||||
cell.border = Border(left=thin_side, right=thin_side, top=thin_side, bottom=thick_bottom)
|
||||
|
||||
# Style Data Rows
|
||||
for row_idx in range(2, len(df_docs) + 2):
|
||||
# Check status for color coding
|
||||
status_val = sheet1.cell(row=row_idx, column=3).value
|
||||
row_fill = success_fill if status_val == "Success" else (error_fill if status_val == "Failed" or status_val == "Error" else None)
|
||||
|
||||
# Apply Zebra stripe if no status color
|
||||
if not row_fill and row_idx % 2 == 0:
|
||||
row_fill = zebra_fill
|
||||
|
||||
for col_idx in range(1, len(df_docs.columns) + 1):
|
||||
cell = sheet1.cell(row=row_idx, column=col_idx)
|
||||
cell.font = regular_font
|
||||
cell.border = cell_border
|
||||
|
||||
# Apply alignments based on column content
|
||||
if col_idx in [1, 3, 4, 5, 9, 13, 14]: # No, Status, Tilt, Auto-rotated, Date, Plat, Items Count
|
||||
cell.alignment = center_align
|
||||
else:
|
||||
cell.alignment = left_align
|
||||
|
||||
if row_fill:
|
||||
cell.fill = row_fill
|
||||
|
||||
# Format tilt with degree symbol
|
||||
if col_idx == 4 and cell.value != "N/A" and cell.value is not None:
|
||||
try:
|
||||
cell.value = float(cell.value)
|
||||
cell.number_format = '0.00"°"'
|
||||
except ValueError:
|
||||
pass
|
||||
|
||||
# Auto-adjust column width for Sheet 1
|
||||
for col in sheet1.columns:
|
||||
max_len = 0
|
||||
for cell in col:
|
||||
val_str = str(cell.value or '')
|
||||
# Exclude long text like Alamat from width sizing
|
||||
if cell.column in [12]: # Alamat
|
||||
max_len = max(max_len, min(len(val_str), 30))
|
||||
else:
|
||||
max_len = max(max_len, len(val_str))
|
||||
col_letter = get_column_letter(col[0].column)
|
||||
sheet1.column_dimensions[col_letter].width = max(max_len + 3, 10)
|
||||
|
||||
sheet1.row_dimensions[1].height = 25
|
||||
for r in range(2, len(df_docs) + 2):
|
||||
sheet1.row_dimensions[r].height = 20
|
||||
|
||||
# 2. Style Parsed Items Sheet
|
||||
sheet2 = workbook['Parsed Items']
|
||||
sheet2.views.sheetView[0].showGridLines = True
|
||||
|
||||
# Style Header Row
|
||||
for col_idx in range(1, len(df_items.columns) + 1):
|
||||
cell = sheet2.cell(row=1, column=col_idx)
|
||||
cell.font = header_font
|
||||
cell.fill = header_fill
|
||||
cell.alignment = center_align
|
||||
cell.border = Border(left=thin_side, right=thin_side, top=thin_side, bottom=thick_bottom)
|
||||
|
||||
# Style Data Rows
|
||||
for row_idx in range(2, len(df_items) + 2):
|
||||
row_fill = zebra_fill if row_idx % 2 == 0 else None
|
||||
for col_idx in range(1, len(df_items.columns) + 1):
|
||||
cell = sheet2.cell(row=row_idx, column=col_idx)
|
||||
cell.font = regular_font
|
||||
cell.border = cell_border
|
||||
|
||||
# Alignments
|
||||
if col_idx in [2, 4, 5]: # SKU, Qty, Unit
|
||||
cell.alignment = center_align
|
||||
else:
|
||||
cell.alignment = left_align
|
||||
|
||||
if row_fill:
|
||||
cell.fill = row_fill
|
||||
|
||||
# Auto-adjust column width for Sheet 2
|
||||
for col in sheet2.columns:
|
||||
max_len = max(len(str(cell.value or '')) for cell in col)
|
||||
col_letter = get_column_letter(col[0].column)
|
||||
sheet2.column_dimensions[col_letter].width = max(max_len + 3, 10)
|
||||
|
||||
sheet2.row_dimensions[1].height = 25
|
||||
for r in range(2, len(df_items) + 2):
|
||||
sheet2.row_dimensions[r].height = 20
|
||||
|
||||
print("Premium Excel report generated successfully!")
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
+151
-151
@@ -1,151 +1,151 @@
|
||||
events {
|
||||
worker_connections 1024;
|
||||
}
|
||||
|
||||
http {
|
||||
include /etc/nginx/mime.types;
|
||||
default_type application/octet-stream;
|
||||
|
||||
sendfile on;
|
||||
keepalive_timeout 65;
|
||||
|
||||
map $http_x_forwarded_proto $proxy_x_forwarded_proto {
|
||||
default $http_x_forwarded_proto;
|
||||
'' $scheme;
|
||||
}
|
||||
|
||||
server {
|
||||
listen 80;
|
||||
server_name _; # accept any host — tunnel URLs, IPs, custom domains
|
||||
|
||||
# Disable body size limit for large image/pdf base64 payloads
|
||||
client_max_body_size 0;
|
||||
|
||||
# Route to Next.js API Gateway (default root)
|
||||
location / {
|
||||
proxy_pass http://paddleocr-pfm-web-app:3000;
|
||||
proxy_http_version 1.1;
|
||||
proxy_set_header Upgrade $http_upgrade;
|
||||
proxy_set_header Connection "upgrade";
|
||||
proxy_set_header Host $http_host;
|
||||
proxy_set_header X-Real-IP $remote_addr;
|
||||
proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for;
|
||||
proxy_set_header X-Forwarded-Proto $proxy_x_forwarded_proto;
|
||||
proxy_read_timeout 300s;
|
||||
proxy_send_timeout 300s;
|
||||
}
|
||||
|
||||
|
||||
location /history {
|
||||
proxy_pass http://paddleocr-pfm-web-app:3000;
|
||||
proxy_http_version 1.1;
|
||||
proxy_set_header Host $http_host;
|
||||
proxy_set_header X-Real-IP $remote_addr;
|
||||
proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for;
|
||||
proxy_set_header X-Forwarded-Proto $proxy_x_forwarded_proto;
|
||||
proxy_read_timeout 300s;
|
||||
proxy_send_timeout 300s;
|
||||
}
|
||||
|
||||
location /arena {
|
||||
proxy_pass http://paddleocr-pfm-web-app:3000;
|
||||
proxy_http_version 1.1;
|
||||
proxy_set_header Host $http_host;
|
||||
proxy_set_header X-Real-IP $remote_addr;
|
||||
proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for;
|
||||
proxy_set_header X-Forwarded-Proto $proxy_x_forwarded_proto;
|
||||
proxy_read_timeout 300s;
|
||||
proxy_send_timeout 300s;
|
||||
}
|
||||
|
||||
location /gpu {
|
||||
proxy_pass http://paddleocr-pfm-web-app:3000;
|
||||
proxy_http_version 1.1;
|
||||
proxy_set_header Host $http_host;
|
||||
proxy_set_header X-Real-IP $remote_addr;
|
||||
proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for;
|
||||
proxy_set_header X-Forwarded-Proto $proxy_x_forwarded_proto;
|
||||
proxy_read_timeout 300s;
|
||||
proxy_send_timeout 300s;
|
||||
}
|
||||
|
||||
location /api {
|
||||
proxy_pass http://paddleocr-pfm-web-app:3000;
|
||||
proxy_http_version 1.1;
|
||||
proxy_set_header Host $http_host;
|
||||
proxy_set_header X-Real-IP $remote_addr;
|
||||
proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for;
|
||||
proxy_set_header X-Forwarded-Proto $proxy_x_forwarded_proto;
|
||||
proxy_read_timeout 300s;
|
||||
proxy_send_timeout 300s;
|
||||
}
|
||||
|
||||
location /_next {
|
||||
proxy_pass http://paddleocr-pfm-web-app:3000/_next;
|
||||
proxy_http_version 1.1;
|
||||
proxy_set_header Upgrade $http_upgrade;
|
||||
proxy_set_header Connection "upgrade";
|
||||
proxy_set_header Host $http_host;
|
||||
proxy_set_header X-Real-IP $remote_addr;
|
||||
proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for;
|
||||
proxy_set_header X-Forwarded-Proto $proxy_x_forwarded_proto;
|
||||
}
|
||||
|
||||
# Route to Pipeline API
|
||||
location /layout-parsing {
|
||||
proxy_pass http://paddleocr-pipeline-api:8090/layout-parsing;
|
||||
proxy_http_version 1.1;
|
||||
proxy_set_header Host $http_host;
|
||||
proxy_set_header X-Real-IP $remote_addr;
|
||||
proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for;
|
||||
proxy_set_header X-Forwarded-Proto $proxy_x_forwarded_proto;
|
||||
proxy_read_timeout 300s;
|
||||
proxy_send_timeout 300s;
|
||||
}
|
||||
|
||||
location /health {
|
||||
proxy_pass http://paddleocr-pipeline-api:8090/health;
|
||||
proxy_http_version 1.1;
|
||||
proxy_set_header Host $http_host;
|
||||
proxy_set_header X-Real-IP $remote_addr;
|
||||
proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for;
|
||||
proxy_set_header X-Forwarded-Proto $proxy_x_forwarded_proto;
|
||||
}
|
||||
|
||||
# Route to vLLM Server API (v1)
|
||||
location /v1 {
|
||||
proxy_pass http://paddleocr-vllm-server:8118/v1;
|
||||
proxy_http_version 1.1;
|
||||
proxy_set_header Host $http_host;
|
||||
proxy_set_header X-Real-IP $remote_addr;
|
||||
proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for;
|
||||
proxy_set_header X-Forwarded-Proto $proxy_x_forwarded_proto;
|
||||
proxy_read_timeout 300s;
|
||||
proxy_send_timeout 300s;
|
||||
}
|
||||
}
|
||||
|
||||
server {
|
||||
listen 8001;
|
||||
server_name _;
|
||||
|
||||
client_max_body_size 0;
|
||||
|
||||
# Secure public endpoint — only allow API v1 surface
|
||||
location /api/v1/ {
|
||||
proxy_pass http://paddleocr-pfm-web-app:3000/api/v1/;
|
||||
proxy_http_version 1.1;
|
||||
proxy_set_header Host $http_host;
|
||||
proxy_set_header X-Real-IP $remote_addr;
|
||||
proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for;
|
||||
proxy_set_header X-Forwarded-Proto $proxy_x_forwarded_proto;
|
||||
proxy_read_timeout 300s;
|
||||
proxy_send_timeout 300s;
|
||||
}
|
||||
|
||||
# Deny everything else
|
||||
location / {
|
||||
return 404;
|
||||
}
|
||||
}
|
||||
}
|
||||
events {
|
||||
worker_connections 1024;
|
||||
}
|
||||
|
||||
http {
|
||||
include /etc/nginx/mime.types;
|
||||
default_type application/octet-stream;
|
||||
|
||||
sendfile on;
|
||||
keepalive_timeout 65;
|
||||
|
||||
map $http_x_forwarded_proto $proxy_x_forwarded_proto {
|
||||
default $http_x_forwarded_proto;
|
||||
'' $scheme;
|
||||
}
|
||||
|
||||
server {
|
||||
listen 80;
|
||||
server_name _; # accept any host — tunnel URLs, IPs, custom domains
|
||||
|
||||
# Disable body size limit for large image/pdf base64 payloads
|
||||
client_max_body_size 0;
|
||||
|
||||
# Route to Next.js API Gateway (default root)
|
||||
location / {
|
||||
proxy_pass http://paddleocr-pfm-web-app:3000;
|
||||
proxy_http_version 1.1;
|
||||
proxy_set_header Upgrade $http_upgrade;
|
||||
proxy_set_header Connection "upgrade";
|
||||
proxy_set_header Host $http_host;
|
||||
proxy_set_header X-Real-IP $remote_addr;
|
||||
proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for;
|
||||
proxy_set_header X-Forwarded-Proto $proxy_x_forwarded_proto;
|
||||
proxy_read_timeout 300s;
|
||||
proxy_send_timeout 300s;
|
||||
}
|
||||
|
||||
|
||||
location /history {
|
||||
proxy_pass http://paddleocr-pfm-web-app:3000;
|
||||
proxy_http_version 1.1;
|
||||
proxy_set_header Host $http_host;
|
||||
proxy_set_header X-Real-IP $remote_addr;
|
||||
proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for;
|
||||
proxy_set_header X-Forwarded-Proto $proxy_x_forwarded_proto;
|
||||
proxy_read_timeout 300s;
|
||||
proxy_send_timeout 300s;
|
||||
}
|
||||
|
||||
location /arena {
|
||||
proxy_pass http://paddleocr-pfm-web-app:3000;
|
||||
proxy_http_version 1.1;
|
||||
proxy_set_header Host $http_host;
|
||||
proxy_set_header X-Real-IP $remote_addr;
|
||||
proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for;
|
||||
proxy_set_header X-Forwarded-Proto $proxy_x_forwarded_proto;
|
||||
proxy_read_timeout 300s;
|
||||
proxy_send_timeout 300s;
|
||||
}
|
||||
|
||||
location /gpu {
|
||||
proxy_pass http://paddleocr-pfm-web-app:3000;
|
||||
proxy_http_version 1.1;
|
||||
proxy_set_header Host $http_host;
|
||||
proxy_set_header X-Real-IP $remote_addr;
|
||||
proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for;
|
||||
proxy_set_header X-Forwarded-Proto $proxy_x_forwarded_proto;
|
||||
proxy_read_timeout 300s;
|
||||
proxy_send_timeout 300s;
|
||||
}
|
||||
|
||||
location /api {
|
||||
proxy_pass http://paddleocr-pfm-web-app:3000;
|
||||
proxy_http_version 1.1;
|
||||
proxy_set_header Host $http_host;
|
||||
proxy_set_header X-Real-IP $remote_addr;
|
||||
proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for;
|
||||
proxy_set_header X-Forwarded-Proto $proxy_x_forwarded_proto;
|
||||
proxy_read_timeout 300s;
|
||||
proxy_send_timeout 300s;
|
||||
}
|
||||
|
||||
location /_next {
|
||||
proxy_pass http://paddleocr-pfm-web-app:3000/_next;
|
||||
proxy_http_version 1.1;
|
||||
proxy_set_header Upgrade $http_upgrade;
|
||||
proxy_set_header Connection "upgrade";
|
||||
proxy_set_header Host $http_host;
|
||||
proxy_set_header X-Real-IP $remote_addr;
|
||||
proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for;
|
||||
proxy_set_header X-Forwarded-Proto $proxy_x_forwarded_proto;
|
||||
}
|
||||
|
||||
# Route to Pipeline API
|
||||
location /layout-parsing {
|
||||
proxy_pass http://paddleocr-pipeline-api:8090/layout-parsing;
|
||||
proxy_http_version 1.1;
|
||||
proxy_set_header Host $http_host;
|
||||
proxy_set_header X-Real-IP $remote_addr;
|
||||
proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for;
|
||||
proxy_set_header X-Forwarded-Proto $proxy_x_forwarded_proto;
|
||||
proxy_read_timeout 300s;
|
||||
proxy_send_timeout 300s;
|
||||
}
|
||||
|
||||
location /health {
|
||||
proxy_pass http://paddleocr-pipeline-api:8090/health;
|
||||
proxy_http_version 1.1;
|
||||
proxy_set_header Host $http_host;
|
||||
proxy_set_header X-Real-IP $remote_addr;
|
||||
proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for;
|
||||
proxy_set_header X-Forwarded-Proto $proxy_x_forwarded_proto;
|
||||
}
|
||||
|
||||
# Route to vLLM Server API (v1)
|
||||
location /v1 {
|
||||
proxy_pass http://paddleocr-vllm-server:8118/v1;
|
||||
proxy_http_version 1.1;
|
||||
proxy_set_header Host $http_host;
|
||||
proxy_set_header X-Real-IP $remote_addr;
|
||||
proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for;
|
||||
proxy_set_header X-Forwarded-Proto $proxy_x_forwarded_proto;
|
||||
proxy_read_timeout 300s;
|
||||
proxy_send_timeout 300s;
|
||||
}
|
||||
}
|
||||
|
||||
server {
|
||||
listen 8001;
|
||||
server_name _;
|
||||
|
||||
client_max_body_size 0;
|
||||
|
||||
# Secure public endpoint — only allow API v1 surface
|
||||
location /api/v1/ {
|
||||
proxy_pass http://paddleocr-pfm-web-app:3000/api/v1/;
|
||||
proxy_http_version 1.1;
|
||||
proxy_set_header Host $http_host;
|
||||
proxy_set_header X-Real-IP $remote_addr;
|
||||
proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for;
|
||||
proxy_set_header X-Forwarded-Proto $proxy_x_forwarded_proto;
|
||||
proxy_read_timeout 300s;
|
||||
proxy_send_timeout 300s;
|
||||
}
|
||||
|
||||
# Deny everything else
|
||||
location / {
|
||||
return 404;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1,41 +1,41 @@
|
||||
# See https://help.github.com/articles/ignoring-files/ for more about ignoring files.
|
||||
|
||||
# dependencies
|
||||
/node_modules
|
||||
/.pnp
|
||||
.pnp.*
|
||||
.yarn/*
|
||||
!.yarn/patches
|
||||
!.yarn/plugins
|
||||
!.yarn/releases
|
||||
!.yarn/versions
|
||||
|
||||
# testing
|
||||
/coverage
|
||||
|
||||
# next.js
|
||||
/.next/
|
||||
/out/
|
||||
|
||||
# production
|
||||
/build
|
||||
|
||||
# misc
|
||||
.DS_Store
|
||||
*.pem
|
||||
|
||||
# debug
|
||||
npm-debug.log*
|
||||
yarn-debug.log*
|
||||
yarn-error.log*
|
||||
.pnpm-debug.log*
|
||||
|
||||
# env files (can opt-in for committing if needed)
|
||||
.env*
|
||||
|
||||
# vercel
|
||||
.vercel
|
||||
|
||||
# typescript
|
||||
*.tsbuildinfo
|
||||
next-env.d.ts
|
||||
# See https://help.github.com/articles/ignoring-files/ for more about ignoring files.
|
||||
|
||||
# dependencies
|
||||
/node_modules
|
||||
/.pnp
|
||||
.pnp.*
|
||||
.yarn/*
|
||||
!.yarn/patches
|
||||
!.yarn/plugins
|
||||
!.yarn/releases
|
||||
!.yarn/versions
|
||||
|
||||
# testing
|
||||
/coverage
|
||||
|
||||
# next.js
|
||||
/.next/
|
||||
/out/
|
||||
|
||||
# production
|
||||
/build
|
||||
|
||||
# misc
|
||||
.DS_Store
|
||||
*.pem
|
||||
|
||||
# debug
|
||||
npm-debug.log*
|
||||
yarn-debug.log*
|
||||
yarn-error.log*
|
||||
.pnpm-debug.log*
|
||||
|
||||
# env files (can opt-in for committing if needed)
|
||||
.env*
|
||||
|
||||
# vercel
|
||||
.vercel
|
||||
|
||||
# typescript
|
||||
*.tsbuildinfo
|
||||
next-env.d.ts
|
||||
@@ -1,5 +1,5 @@
|
||||
<!-- BEGIN:nextjs-agent-rules -->
|
||||
# This is NOT the Next.js you know
|
||||
|
||||
This version has breaking changes — APIs, conventions, and file structure may all differ from your training data. Read the relevant guide in `node_modules/next/dist/docs/` before writing any code. Heed deprecation notices.
|
||||
<!-- END:nextjs-agent-rules -->
|
||||
<!-- BEGIN:nextjs-agent-rules -->
|
||||
# This is NOT the Next.js you know
|
||||
|
||||
This version has breaking changes — APIs, conventions, and file structure may all differ from your training data. Read the relevant guide in `node_modules/next/dist/docs/` before writing any code. Heed deprecation notices.
|
||||
<!-- END:nextjs-agent-rules -->
|
||||
@@ -1 +1 @@
|
||||
@AGENTS.md
|
||||
@AGENTS.md
|
||||
@@ -1,36 +1,36 @@
|
||||
This is a [Next.js](https://nextjs.org) project bootstrapped with [`create-next-app`](https://nextjs.org/docs/app/api-reference/cli/create-next-app).
|
||||
|
||||
## Getting Started
|
||||
|
||||
First, run the development server:
|
||||
|
||||
```bash
|
||||
npm run dev
|
||||
# or
|
||||
yarn dev
|
||||
# or
|
||||
pnpm dev
|
||||
# or
|
||||
bun dev
|
||||
```
|
||||
|
||||
Open [http://localhost:3000](http://localhost:3000) with your browser to see the result.
|
||||
|
||||
You can start editing the page by modifying `app/page.tsx`. The page auto-updates as you edit the file.
|
||||
|
||||
This project uses [`next/font`](https://nextjs.org/docs/app/building-your-application/optimizing/fonts) to automatically optimize and load [Geist](https://vercel.com/font), a new font family for Vercel.
|
||||
|
||||
## Learn More
|
||||
|
||||
To learn more about Next.js, take a look at the following resources:
|
||||
|
||||
- [Next.js Documentation](https://nextjs.org/docs) - learn about Next.js features and API.
|
||||
- [Learn Next.js](https://nextjs.org/learn) - an interactive Next.js tutorial.
|
||||
|
||||
You can check out [the Next.js GitHub repository](https://github.com/vercel/next.js) - your feedback and contributions are welcome!
|
||||
|
||||
## Deploy on Vercel
|
||||
|
||||
The easiest way to deploy your Next.js app is to use the [Vercel Platform](https://vercel.com/new?utm_medium=default-template&filter=next.js&utm_source=create-next-app&utm_campaign=create-next-app-readme) from the creators of Next.js.
|
||||
|
||||
Check out our [Next.js deployment documentation](https://nextjs.org/docs/app/building-your-application/deploying) for more details.
|
||||
This is a [Next.js](https://nextjs.org) project bootstrapped with [`create-next-app`](https://nextjs.org/docs/app/api-reference/cli/create-next-app).
|
||||
|
||||
## Getting Started
|
||||
|
||||
First, run the development server:
|
||||
|
||||
```bash
|
||||
npm run dev
|
||||
# or
|
||||
yarn dev
|
||||
# or
|
||||
pnpm dev
|
||||
# or
|
||||
bun dev
|
||||
```
|
||||
|
||||
Open [http://localhost:3000](http://localhost:3000) with your browser to see the result.
|
||||
|
||||
You can start editing the page by modifying `app/page.tsx`. The page auto-updates as you edit the file.
|
||||
|
||||
This project uses [`next/font`](https://nextjs.org/docs/app/building-your-application/optimizing/fonts) to automatically optimize and load [Geist](https://vercel.com/font), a new font family for Vercel.
|
||||
|
||||
## Learn More
|
||||
|
||||
To learn more about Next.js, take a look at the following resources:
|
||||
|
||||
- [Next.js Documentation](https://nextjs.org/docs) - learn about Next.js features and API.
|
||||
- [Learn Next.js](https://nextjs.org/learn) - an interactive Next.js tutorial.
|
||||
|
||||
You can check out [the Next.js GitHub repository](https://github.com/vercel/next.js) - your feedback and contributions are welcome!
|
||||
|
||||
## Deploy on Vercel
|
||||
|
||||
The easiest way to deploy your Next.js app is to use the [Vercel Platform](https://vercel.com/new?utm_medium=default-template&filter=next.js&utm_source=create-next-app&utm_campaign=create-next-app-readme) from the creators of Next.js.
|
||||
|
||||
Check out our [Next.js deployment documentation](https://nextjs.org/docs/app/building-your-application/deploying) for more details.
|
||||
+103
-103
@@ -1,103 +1,103 @@
|
||||
const puppeteer = require('puppeteer');
|
||||
const fs = require('fs');
|
||||
|
||||
(async () => {
|
||||
const browser = await puppeteer.launch({
|
||||
headless: "new",
|
||||
args: ['--no-sandbox', '--disable-setuid-sandbox']
|
||||
});
|
||||
const page = await browser.newPage();
|
||||
await page.setViewport({ width: 1280, height: 800 });
|
||||
|
||||
console.log("Navigating to login page...");
|
||||
await page.goto('http://localhost:3000/admin/master-data', { waitUntil: 'networkidle2' });
|
||||
|
||||
console.log("Filling login form...");
|
||||
await page.type('input[type="text"]', 'admin');
|
||||
await page.type('input[type="password"]', 'password');
|
||||
|
||||
await page.screenshot({ path: 'test_step1_login_filled.png' });
|
||||
|
||||
console.log("Clicking login...");
|
||||
await Promise.all([
|
||||
page.click('button[type="submit"]'),
|
||||
page.waitForNavigation({ waitUntil: 'networkidle0' }).catch(e => console.log('Navigation wait timeout/catch'))
|
||||
]);
|
||||
|
||||
// Wait a bit for React to render the stores table
|
||||
await new Promise(resolve => setTimeout(resolve, 2000));
|
||||
await page.screenshot({ path: 'test_step2_after_login.png' });
|
||||
|
||||
// Add store
|
||||
console.log("Clicking Add Store...");
|
||||
await page.evaluate(() => {
|
||||
const btns = Array.from(document.querySelectorAll('button'));
|
||||
const addBtn = btns.find(b => b.textContent.includes('Add Store'));
|
||||
if (addBtn) addBtn.click();
|
||||
});
|
||||
await new Promise(resolve => setTimeout(resolve, 500));
|
||||
|
||||
console.log("Filling new store form...");
|
||||
const inputs = await page.$$('input[placeholder]');
|
||||
for (const input of inputs) {
|
||||
const placeholder = await input.evaluate(el => el.getAttribute('placeholder'));
|
||||
if (placeholder === 'Kode Toko') await input.type('TEST99');
|
||||
if (placeholder === 'Nama Toko') await input.type('Toko Test 99');
|
||||
if (placeholder === 'Alamat') await input.type('Alamat Test');
|
||||
}
|
||||
|
||||
await page.screenshot({ path: 'test_step3_store_filled.png' });
|
||||
|
||||
console.log("Saving store...");
|
||||
await page.evaluate(() => {
|
||||
const btns = Array.from(document.querySelectorAll('button'));
|
||||
const saveBtn = btns.find(b => b.textContent === 'Save');
|
||||
if (saveBtn) saveBtn.click();
|
||||
});
|
||||
|
||||
await new Promise(resolve => setTimeout(resolve, 2000));
|
||||
await page.screenshot({ path: 'test_step4_store_saved.png' });
|
||||
|
||||
// Switch to SKUs tab
|
||||
console.log("Switching to SKUs tab...");
|
||||
await page.evaluate(() => {
|
||||
const btns = Array.from(document.querySelectorAll('button'));
|
||||
const skuBtn = btns.find(b => b.textContent === 'SKUs');
|
||||
if (skuBtn) skuBtn.click();
|
||||
});
|
||||
await new Promise(resolve => setTimeout(resolve, 1000));
|
||||
await page.screenshot({ path: 'test_step5_skus_tab.png' });
|
||||
|
||||
console.log("Clicking Add SKU...");
|
||||
await page.evaluate(() => {
|
||||
const btns = Array.from(document.querySelectorAll('button'));
|
||||
const addBtn = btns.find(b => b.textContent.includes('Add SKU'));
|
||||
if (addBtn) addBtn.click();
|
||||
});
|
||||
await new Promise(resolve => setTimeout(resolve, 500));
|
||||
|
||||
console.log("Filling new SKU form...");
|
||||
const skuInputs = await page.$$('input[placeholder]');
|
||||
for (const input of skuInputs) {
|
||||
const placeholder = await input.evaluate(el => el.getAttribute('placeholder'));
|
||||
if (placeholder === 'Kode Item') await input.type('SKU99');
|
||||
if (placeholder === 'Nama Item') await input.type('Item 99');
|
||||
if (placeholder === 'Barcode') await input.type('12345');
|
||||
if (placeholder === 'Jenis Outer (e.g. DUS)') await input.type('DUS');
|
||||
}
|
||||
|
||||
await page.screenshot({ path: 'test_step6_sku_filled.png' });
|
||||
|
||||
console.log("Saving SKU...");
|
||||
await page.evaluate(() => {
|
||||
const btns = Array.from(document.querySelectorAll('button'));
|
||||
const saveBtn = btns.find(b => b.textContent === 'Save');
|
||||
if (saveBtn) saveBtn.click();
|
||||
});
|
||||
|
||||
await new Promise(resolve => setTimeout(resolve, 2000));
|
||||
await page.screenshot({ path: 'test_step7_sku_saved.png' });
|
||||
|
||||
console.log("Done! Screenshots saved.");
|
||||
await browser.close();
|
||||
})();
|
||||
const puppeteer = require('puppeteer');
|
||||
const fs = require('fs');
|
||||
|
||||
(async () => {
|
||||
const browser = await puppeteer.launch({
|
||||
headless: "new",
|
||||
args: ['--no-sandbox', '--disable-setuid-sandbox']
|
||||
});
|
||||
const page = await browser.newPage();
|
||||
await page.setViewport({ width: 1280, height: 800 });
|
||||
|
||||
console.log("Navigating to login page...");
|
||||
await page.goto('http://localhost:3000/admin/master-data', { waitUntil: 'networkidle2' });
|
||||
|
||||
console.log("Filling login form...");
|
||||
await page.type('input[type="text"]', 'admin');
|
||||
await page.type('input[type="password"]', 'password');
|
||||
|
||||
await page.screenshot({ path: 'test_step1_login_filled.png' });
|
||||
|
||||
console.log("Clicking login...");
|
||||
await Promise.all([
|
||||
page.click('button[type="submit"]'),
|
||||
page.waitForNavigation({ waitUntil: 'networkidle0' }).catch(e => console.log('Navigation wait timeout/catch'))
|
||||
]);
|
||||
|
||||
// Wait a bit for React to render the stores table
|
||||
await new Promise(resolve => setTimeout(resolve, 2000));
|
||||
await page.screenshot({ path: 'test_step2_after_login.png' });
|
||||
|
||||
// Add store
|
||||
console.log("Clicking Add Store...");
|
||||
await page.evaluate(() => {
|
||||
const btns = Array.from(document.querySelectorAll('button'));
|
||||
const addBtn = btns.find(b => b.textContent.includes('Add Store'));
|
||||
if (addBtn) addBtn.click();
|
||||
});
|
||||
await new Promise(resolve => setTimeout(resolve, 500));
|
||||
|
||||
console.log("Filling new store form...");
|
||||
const inputs = await page.$$('input[placeholder]');
|
||||
for (const input of inputs) {
|
||||
const placeholder = await input.evaluate(el => el.getAttribute('placeholder'));
|
||||
if (placeholder === 'Kode Toko') await input.type('TEST99');
|
||||
if (placeholder === 'Nama Toko') await input.type('Toko Test 99');
|
||||
if (placeholder === 'Alamat') await input.type('Alamat Test');
|
||||
}
|
||||
|
||||
await page.screenshot({ path: 'test_step3_store_filled.png' });
|
||||
|
||||
console.log("Saving store...");
|
||||
await page.evaluate(() => {
|
||||
const btns = Array.from(document.querySelectorAll('button'));
|
||||
const saveBtn = btns.find(b => b.textContent === 'Save');
|
||||
if (saveBtn) saveBtn.click();
|
||||
});
|
||||
|
||||
await new Promise(resolve => setTimeout(resolve, 2000));
|
||||
await page.screenshot({ path: 'test_step4_store_saved.png' });
|
||||
|
||||
// Switch to SKUs tab
|
||||
console.log("Switching to SKUs tab...");
|
||||
await page.evaluate(() => {
|
||||
const btns = Array.from(document.querySelectorAll('button'));
|
||||
const skuBtn = btns.find(b => b.textContent === 'SKUs');
|
||||
if (skuBtn) skuBtn.click();
|
||||
});
|
||||
await new Promise(resolve => setTimeout(resolve, 1000));
|
||||
await page.screenshot({ path: 'test_step5_skus_tab.png' });
|
||||
|
||||
console.log("Clicking Add SKU...");
|
||||
await page.evaluate(() => {
|
||||
const btns = Array.from(document.querySelectorAll('button'));
|
||||
const addBtn = btns.find(b => b.textContent.includes('Add SKU'));
|
||||
if (addBtn) addBtn.click();
|
||||
});
|
||||
await new Promise(resolve => setTimeout(resolve, 500));
|
||||
|
||||
console.log("Filling new SKU form...");
|
||||
const skuInputs = await page.$$('input[placeholder]');
|
||||
for (const input of skuInputs) {
|
||||
const placeholder = await input.evaluate(el => el.getAttribute('placeholder'));
|
||||
if (placeholder === 'Kode Item') await input.type('SKU99');
|
||||
if (placeholder === 'Nama Item') await input.type('Item 99');
|
||||
if (placeholder === 'Barcode') await input.type('12345');
|
||||
if (placeholder === 'Jenis Outer (e.g. DUS)') await input.type('DUS');
|
||||
}
|
||||
|
||||
await page.screenshot({ path: 'test_step6_sku_filled.png' });
|
||||
|
||||
console.log("Saving SKU...");
|
||||
await page.evaluate(() => {
|
||||
const btns = Array.from(document.querySelectorAll('button'));
|
||||
const saveBtn = btns.find(b => b.textContent === 'Save');
|
||||
if (saveBtn) saveBtn.click();
|
||||
});
|
||||
|
||||
await new Promise(resolve => setTimeout(resolve, 2000));
|
||||
await page.screenshot({ path: 'test_step7_sku_saved.png' });
|
||||
|
||||
console.log("Done! Screenshots saved.");
|
||||
await browser.close();
|
||||
})();
|
||||
@@ -1,18 +1,18 @@
|
||||
import { defineConfig, globalIgnores } from "eslint/config";
|
||||
import nextVitals from "eslint-config-next/core-web-vitals";
|
||||
import nextTs from "eslint-config-next/typescript";
|
||||
|
||||
const eslintConfig = defineConfig([
|
||||
...nextVitals,
|
||||
...nextTs,
|
||||
// Override default ignores of eslint-config-next.
|
||||
globalIgnores([
|
||||
// Default ignores of eslint-config-next:
|
||||
".next/**",
|
||||
"out/**",
|
||||
"build/**",
|
||||
"next-env.d.ts",
|
||||
]),
|
||||
]);
|
||||
|
||||
export default eslintConfig;
|
||||
import { defineConfig, globalIgnores } from "eslint/config";
|
||||
import nextVitals from "eslint-config-next/core-web-vitals";
|
||||
import nextTs from "eslint-config-next/typescript";
|
||||
|
||||
const eslintConfig = defineConfig([
|
||||
...nextVitals,
|
||||
...nextTs,
|
||||
// Override default ignores of eslint-config-next.
|
||||
globalIgnores([
|
||||
// Default ignores of eslint-config-next:
|
||||
".next/**",
|
||||
"out/**",
|
||||
"build/**",
|
||||
"next-env.d.ts",
|
||||
]),
|
||||
]);
|
||||
|
||||
export default eslintConfig;
|
||||
+114
-114
@@ -1,114 +1,114 @@
|
||||
const fs = require('fs');
|
||||
const path = require('path');
|
||||
const { Client } = require('pg');
|
||||
|
||||
async function main() {
|
||||
console.log('=== STARTING SKU MASTER TSV IMPORT ===');
|
||||
|
||||
const client = new Client({
|
||||
host: 'paddleocr-db',
|
||||
port: 5432,
|
||||
user: 'postgres',
|
||||
password: 'postgres',
|
||||
database: 'dopfm'
|
||||
});
|
||||
|
||||
try {
|
||||
await client.connect();
|
||||
console.log('Connected to database.');
|
||||
|
||||
// 1. Alter table to add new packaging columns if they don't exist
|
||||
console.log('Ensuring table schema has new packaging columns...');
|
||||
await client.query(`
|
||||
ALTER TABLE sku_master ADD COLUMN IF NOT EXISTS standar_jumlah VARCHAR(50);
|
||||
ALTER TABLE sku_master ADD COLUMN IF NOT EXISTS berat_kemasan NUMERIC(10, 3);
|
||||
ALTER TABLE sku_master ADD COLUMN IF NOT EXISTS isi_outer_kg NUMERIC(10, 3);
|
||||
ALTER TABLE sku_master ADD COLUMN IF NOT EXISTS isi_outer_pac INTEGER;
|
||||
ALTER TABLE sku_master ADD COLUMN IF NOT EXISTS jenis_outer VARCHAR(50);
|
||||
`);
|
||||
console.log('Table schema verified/updated.');
|
||||
|
||||
// 2. Truncate old data
|
||||
console.log('Clearing old SKU master data...');
|
||||
await client.query('TRUNCATE TABLE sku_master RESTART IDENTITY CASCADE');
|
||||
console.log('Old SKU master data cleared.');
|
||||
|
||||
// 3. Read and parse TSV file
|
||||
const tsvPath = path.join(__dirname, 'sku_master.tsv');
|
||||
if (!fs.existsSync(tsvPath)) {
|
||||
throw new Error(`File not found at ${tsvPath}`);
|
||||
}
|
||||
|
||||
const tsvContent = fs.readFileSync(tsvPath, 'utf8');
|
||||
const lines = tsvContent.split(/\r?\n/);
|
||||
|
||||
let insertCount = 0;
|
||||
let skipCount = 0;
|
||||
|
||||
console.log(`Parsing ${lines.length} lines from TSV...`);
|
||||
|
||||
// We start from line 5 (0-indexed 4 is the header row, lines before are title headers)
|
||||
for (let i = 5; i < lines.length; i++) {
|
||||
const line = lines[i].trim();
|
||||
if (!line) continue;
|
||||
|
||||
const cols = line.split('\t').map(c => c.trim());
|
||||
if (cols.length < 3) {
|
||||
skipCount++;
|
||||
continue;
|
||||
}
|
||||
|
||||
const noSku = cols[1];
|
||||
const namaItem = cols[2];
|
||||
|
||||
// Verify SKU code format (must be standard 8-digit)
|
||||
if (!noSku || !/^\d{8}$/.test(noSku)) {
|
||||
skipCount++;
|
||||
continue;
|
||||
}
|
||||
|
||||
const standarJumlah = cols[3] || null;
|
||||
|
||||
// Parse numeric columns
|
||||
const beratKemasan = cols[4] ? parseFloat(cols[4].replace(',', '.')) : null;
|
||||
const isiOuterKg = cols[5] ? parseFloat(cols[5].replace(',', '.')) : null;
|
||||
const isiOuterPac = cols[6] ? parseInt(cols[6], 10) : null;
|
||||
const jenisOuter = cols[7] || null;
|
||||
|
||||
await client.query(`
|
||||
INSERT INTO sku_master (
|
||||
no_sku, nama_item, standar_jumlah, berat_kemasan, isi_outer_kg, isi_outer_pac, jenis_outer
|
||||
) VALUES ($1, $2, $3, $4, $5, $6, $7)
|
||||
ON CONFLICT (no_sku) DO UPDATE SET
|
||||
nama_item = EXCLUDED.nama_item,
|
||||
standar_jumlah = EXCLUDED.standar_jumlah,
|
||||
berat_kemasan = EXCLUDED.berat_kemasan,
|
||||
isi_outer_kg = EXCLUDED.isi_outer_kg,
|
||||
isi_outer_pac = EXCLUDED.isi_outer_pac,
|
||||
jenis_outer = EXCLUDED.jenis_outer
|
||||
`, [
|
||||
noSku,
|
||||
namaItem,
|
||||
standarJumlah,
|
||||
isNaN(beratKemasan) ? null : beratKemasan,
|
||||
isNaN(isiOuterKg) ? null : isiOuterKg,
|
||||
isNaN(isiOuterPac) ? null : isiOuterPac,
|
||||
jenisOuter
|
||||
]);
|
||||
|
||||
insertCount++;
|
||||
}
|
||||
|
||||
console.log(`\nImport Completed Successfully:`);
|
||||
console.log(`- Inserted/Updated: ${insertCount} SKU records`);
|
||||
console.log(`- Skipped (headers/invalid): ${skipCount} lines`);
|
||||
|
||||
} catch (err) {
|
||||
console.error('Import process failed:', err);
|
||||
} finally {
|
||||
await client.end();
|
||||
console.log('Database connection closed.');
|
||||
}
|
||||
}
|
||||
|
||||
main();
|
||||
const fs = require('fs');
|
||||
const path = require('path');
|
||||
const { Client } = require('pg');
|
||||
|
||||
async function main() {
|
||||
console.log('=== STARTING SKU MASTER TSV IMPORT ===');
|
||||
|
||||
const client = new Client({
|
||||
host: 'paddleocr-db',
|
||||
port: 5432,
|
||||
user: 'postgres',
|
||||
password: 'postgres',
|
||||
database: 'dopfm'
|
||||
});
|
||||
|
||||
try {
|
||||
await client.connect();
|
||||
console.log('Connected to database.');
|
||||
|
||||
// 1. Alter table to add new packaging columns if they don't exist
|
||||
console.log('Ensuring table schema has new packaging columns...');
|
||||
await client.query(`
|
||||
ALTER TABLE sku_master ADD COLUMN IF NOT EXISTS standar_jumlah VARCHAR(50);
|
||||
ALTER TABLE sku_master ADD COLUMN IF NOT EXISTS berat_kemasan NUMERIC(10, 3);
|
||||
ALTER TABLE sku_master ADD COLUMN IF NOT EXISTS isi_outer_kg NUMERIC(10, 3);
|
||||
ALTER TABLE sku_master ADD COLUMN IF NOT EXISTS isi_outer_pac INTEGER;
|
||||
ALTER TABLE sku_master ADD COLUMN IF NOT EXISTS jenis_outer VARCHAR(50);
|
||||
`);
|
||||
console.log('Table schema verified/updated.');
|
||||
|
||||
// 2. Truncate old data
|
||||
console.log('Clearing old SKU master data...');
|
||||
await client.query('TRUNCATE TABLE sku_master RESTART IDENTITY CASCADE');
|
||||
console.log('Old SKU master data cleared.');
|
||||
|
||||
// 3. Read and parse TSV file
|
||||
const tsvPath = path.join(__dirname, 'sku_master.tsv');
|
||||
if (!fs.existsSync(tsvPath)) {
|
||||
throw new Error(`File not found at ${tsvPath}`);
|
||||
}
|
||||
|
||||
const tsvContent = fs.readFileSync(tsvPath, 'utf8');
|
||||
const lines = tsvContent.split(/\r?\n/);
|
||||
|
||||
let insertCount = 0;
|
||||
let skipCount = 0;
|
||||
|
||||
console.log(`Parsing ${lines.length} lines from TSV...`);
|
||||
|
||||
// We start from line 5 (0-indexed 4 is the header row, lines before are title headers)
|
||||
for (let i = 5; i < lines.length; i++) {
|
||||
const line = lines[i].trim();
|
||||
if (!line) continue;
|
||||
|
||||
const cols = line.split('\t').map(c => c.trim());
|
||||
if (cols.length < 3) {
|
||||
skipCount++;
|
||||
continue;
|
||||
}
|
||||
|
||||
const noSku = cols[1];
|
||||
const namaItem = cols[2];
|
||||
|
||||
// Verify SKU code format (must be standard 8-digit)
|
||||
if (!noSku || !/^\d{8}$/.test(noSku)) {
|
||||
skipCount++;
|
||||
continue;
|
||||
}
|
||||
|
||||
const standarJumlah = cols[3] || null;
|
||||
|
||||
// Parse numeric columns
|
||||
const beratKemasan = cols[4] ? parseFloat(cols[4].replace(',', '.')) : null;
|
||||
const isiOuterKg = cols[5] ? parseFloat(cols[5].replace(',', '.')) : null;
|
||||
const isiOuterPac = cols[6] ? parseInt(cols[6], 10) : null;
|
||||
const jenisOuter = cols[7] || null;
|
||||
|
||||
await client.query(`
|
||||
INSERT INTO sku_master (
|
||||
no_sku, nama_item, standar_jumlah, berat_kemasan, isi_outer_kg, isi_outer_pac, jenis_outer
|
||||
) VALUES ($1, $2, $3, $4, $5, $6, $7)
|
||||
ON CONFLICT (no_sku) DO UPDATE SET
|
||||
nama_item = EXCLUDED.nama_item,
|
||||
standar_jumlah = EXCLUDED.standar_jumlah,
|
||||
berat_kemasan = EXCLUDED.berat_kemasan,
|
||||
isi_outer_kg = EXCLUDED.isi_outer_kg,
|
||||
isi_outer_pac = EXCLUDED.isi_outer_pac,
|
||||
jenis_outer = EXCLUDED.jenis_outer
|
||||
`, [
|
||||
noSku,
|
||||
namaItem,
|
||||
standarJumlah,
|
||||
isNaN(beratKemasan) ? null : beratKemasan,
|
||||
isNaN(isiOuterKg) ? null : isiOuterKg,
|
||||
isNaN(isiOuterPac) ? null : isiOuterPac,
|
||||
jenisOuter
|
||||
]);
|
||||
|
||||
insertCount++;
|
||||
}
|
||||
|
||||
console.log(`\nImport Completed Successfully:`);
|
||||
console.log(`- Inserted/Updated: ${insertCount} SKU records`);
|
||||
console.log(`- Skipped (headers/invalid): ${skipCount} lines`);
|
||||
|
||||
} catch (err) {
|
||||
console.error('Import process failed:', err);
|
||||
} finally {
|
||||
await client.end();
|
||||
console.log('Database connection closed.');
|
||||
}
|
||||
}
|
||||
|
||||
main();
|
||||
@@ -1,299 +1,299 @@
|
||||
const { Client } = require("pg");
|
||||
|
||||
function cleanFinalValue(val, preserveNewlines = false) {
|
||||
if (!val) return "Not Found";
|
||||
const cleaned = val.replace(/<[^>]*>/g, "");
|
||||
if (preserveNewlines) {
|
||||
return cleaned.split("\n").map(line => line.trim()).filter(Boolean).join("\n") || "Not Found";
|
||||
} else {
|
||||
return cleaned.replace(/\s+/g, " ").trim() || "Not Found";
|
||||
}
|
||||
}
|
||||
|
||||
function parseDOMetadata(markdown) {
|
||||
const metadata = {
|
||||
vendorInfo: "Not Found",
|
||||
customerInfo: "Not Found",
|
||||
tanggal: "Not Found",
|
||||
noSO: "Not Found",
|
||||
noDO: "Not Found",
|
||||
noPO: "Not Found",
|
||||
items: []
|
||||
};
|
||||
|
||||
if (!markdown) return metadata;
|
||||
|
||||
const cleanMarkdown = markdown
|
||||
.replace(/<\/tr>/gi, "\n")
|
||||
.replace(/<br\s*\/?>/gi, "\n")
|
||||
.replace(/<\/p>/gi, "\n")
|
||||
.replace(/<[^>]*>/g, " ");
|
||||
|
||||
const lines = cleanMarkdown.split("\n").map(l => l.trim()).filter(Boolean);
|
||||
|
||||
// Vendor Info
|
||||
const vendorStop = /(?:no\.?\s*(?:so|do|po)|tanggal|date|Kepada|Yth|Customer|Deliver|Order\s+Untuk|#|\d{2}:\d{2}:\d{2})/i;
|
||||
const vendorStartIndex = lines.findIndex(line =>
|
||||
/PT\./i.test(line) && !/(?:Kepada|Yth|Customer|Deliver|Order\s+Untuk|Alamat|no\.?\s*(?:so|do|po)|tanggal|date)/i.test(line)
|
||||
);
|
||||
if (vendorStartIndex !== -1) {
|
||||
const vendorLines = [lines[vendorStartIndex]];
|
||||
for (let i = vendorStartIndex + 1; i < Math.min(lines.length, vendorStartIndex + 4); i++) {
|
||||
if (vendorStop.test(lines[i])) break;
|
||||
vendorLines.push(lines[i]);
|
||||
}
|
||||
metadata.vendorInfo = vendorLines.join("\n");
|
||||
} else {
|
||||
const vendorMatch = cleanMarkdown.match(/(PT\.\s*CHAROEN[^\n]*)/i) || cleanMarkdown.match(/(PT\.[^\n]+)/i);
|
||||
if (vendorMatch) metadata.vendorInfo = vendorMatch[1].trim();
|
||||
}
|
||||
|
||||
// Customer Info
|
||||
const customerStop = /(?:no\.?\s*(?:so|do|po)|tanggal|date|#|\d{2}:\d{2}:\d{2})/i;
|
||||
let customerStartIndex = lines.findIndex(line =>
|
||||
/(?:Kepada Yth|Yth|Customer|Deliver To)\s*[:\-]/i.test(line) || /PT\.\s*PRIMAFOOD/i.test(line)
|
||||
);
|
||||
if (customerStartIndex === -1) {
|
||||
const ptIndices = lines.map((l, idx) => l.toUpperCase().includes("PT.") ? idx : -1).filter(idx => idx !== -1);
|
||||
const secondaryIndices = ptIndices.filter(idx => idx !== vendorStartIndex);
|
||||
if (secondaryIndices.length > 0) {
|
||||
customerStartIndex = secondaryIndices[0];
|
||||
}
|
||||
}
|
||||
|
||||
if (customerStartIndex !== -1) {
|
||||
const customerLines = [lines[customerStartIndex]];
|
||||
for (let i = customerStartIndex + 1; i < Math.min(lines.length, customerStartIndex + 4); i++) {
|
||||
if (customerStop.test(lines[i])) break;
|
||||
customerLines.push(lines[i]);
|
||||
}
|
||||
metadata.customerInfo = customerLines.join("\n");
|
||||
} else {
|
||||
const customerMatch = cleanMarkdown.match(/(?:Kepada Yth|Yth|Customer|Deliver To)[ \t]*[:\-][ \t]*([^\n]+)/i) || cleanMarkdown.match(/(PT\.[ \t]*PRIMAFOOD[^\n]*)/i);
|
||||
if (customerMatch) metadata.customerInfo = customerMatch[1].trim();
|
||||
}
|
||||
|
||||
// Direct matches
|
||||
const tanggalMatch = cleanMarkdown.match(/Tanggal[ \t]*[:\-][ \t]*([^\n]+)/i) || cleanMarkdown.match(/(?:Date|D\.O\.[ \t]*Date)[ \t]*[:\- \t]*([\d\-\/A-Za-z \t]+)/i);
|
||||
if (tanggalMatch) metadata.tanggal = tanggalMatch[1].trim();
|
||||
|
||||
const soMatch = cleanMarkdown.match(/(?:No\.?[ \t]*SO|SO[ \t]*No\.?)[ \t]*[:\-][ \t]*([A-Z0-9\-]+)/i);
|
||||
if (soMatch) metadata.noSO = soMatch[1].trim();
|
||||
|
||||
const doMatch = cleanMarkdown.match(/(?:No\.?[ \t]*DO|Delivery Order[ \t]*No|D\.O\.[ \t]*No|Order[ \t]*No)[ \t]*[:\- \t]*([A-Z0-9\-]+)/i);
|
||||
if (doMatch) metadata.noDO = doMatch[1].trim();
|
||||
|
||||
const poMatch = cleanMarkdown.match(/(?:No\.?[ \t]*PO|PO[ \t]*No\.?)[ \t]*[:\-][ \t]*([A-Z0-9\-\/]+)/i);
|
||||
if (poMatch) metadata.noPO = poMatch[1].trim();
|
||||
|
||||
// Fallback block/sequential alignment if any of the metadata values are not found
|
||||
if (
|
||||
metadata.tanggal === "Not Found" || !metadata.tanggal ||
|
||||
metadata.noSO === "Not Found" || !metadata.noSO ||
|
||||
metadata.noDO === "Not Found" || !metadata.noDO ||
|
||||
metadata.noPO === "Not Found" || !metadata.noPO
|
||||
) {
|
||||
const idxTanggal = lines.findIndex(l => /^Tanggal\s*[:\-]?\s*$/i.test(l));
|
||||
const idxSO = lines.findIndex(l => /^No\.?\s*SO\s*[:\-]?\s*$/i.test(l));
|
||||
const idxDO = lines.findIndex(l => /^No\.?\s*DO\s*[:\-]?\s*$/i.test(l));
|
||||
const idxPO = lines.findIndex(l => /^No\.?\s*PO\s*[:\-]?\s*$/i.test(l));
|
||||
|
||||
if (idxTanggal !== -1 || idxSO !== -1 || idxDO !== -1 || idxPO !== -1) {
|
||||
const indices = [idxTanggal, idxSO, idxDO, idxPO].filter(idx => idx !== -1);
|
||||
const minIndex = Math.min(...indices);
|
||||
const maxIndex = Math.max(...indices);
|
||||
|
||||
if (maxIndex - minIndex < 8) {
|
||||
const candidateLines = lines.slice(maxIndex + 1, maxIndex + 12);
|
||||
|
||||
if (metadata.tanggal === "Not Found" || !metadata.tanggal) {
|
||||
const dateRegex = /\b\d{1,2}\s+(?:Jan|Feb|Mar|Apr|May|Jun|Jul|Aug|Sep|Oct|Nov|Dec)[a-z]*\s+\d{4}\b/i;
|
||||
for (const line of candidateLines) {
|
||||
const m = line.match(dateRegex);
|
||||
if (m) {
|
||||
metadata.tanggal = m[0];
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
const tenDigitNumbers = [];
|
||||
for (const line of candidateLines) {
|
||||
const m = line.match(/\b\d{10}\b/);
|
||||
if (m) {
|
||||
tenDigitNumbers.push(m[0]);
|
||||
}
|
||||
}
|
||||
|
||||
if (tenDigitNumbers.length >= 2) {
|
||||
if (metadata.noSO === "Not Found" || !metadata.noSO) metadata.noSO = tenDigitNumbers[0];
|
||||
if (metadata.noDO === "Not Found" || !metadata.noDO) metadata.noDO = tenDigitNumbers[1];
|
||||
} else if (tenDigitNumbers.length === 1) {
|
||||
if (metadata.noSO === "Not Found" || !metadata.noSO) metadata.noSO = tenDigitNumbers[0];
|
||||
}
|
||||
|
||||
if (metadata.noPO === "Not Found" || !metadata.noPO) {
|
||||
const poRegex = /\b(?:PO|P0)[A-Z0-9\-\/]+\b/i;
|
||||
for (const line of candidateLines) {
|
||||
const m = line.match(poRegex);
|
||||
if (m) {
|
||||
metadata.noPO = m[0];
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Shift realignment detection and correction
|
||||
const isShortSO = /^\d{1,2}$/.test(metadata.noSO);
|
||||
const isDateInSO = /\d{1,2}\s+(?:Jan|Feb|Mar|Apr|May|Jun|Jul|Aug|Sep|Oct|Nov|Dec)/i.test(metadata.noSO);
|
||||
const isShiftedPO = /^\d{10}$/.test(metadata.noPO) || /^16\d{8}$/.test(metadata.noPO);
|
||||
const isShiftedDO = /^\d{10}$/.test(metadata.noDO) && (metadata.noSO === "Not Found" || metadata.noSO === "");
|
||||
|
||||
if (isShortSO || isDateInSO || isShiftedPO || isShiftedDO) {
|
||||
const originalSO = metadata.noSO;
|
||||
const originalDO = metadata.noDO;
|
||||
const originalPO = metadata.noPO;
|
||||
|
||||
const dateRegex = /\b\d{1,2}\s+(?:Jan|Feb|Mar|Apr|May|Jun|Jul|Aug|Sep|Oct|Nov|Dec)[a-z]*\s+\d{4}\b/i;
|
||||
const dateMatch = cleanMarkdown.match(dateRegex);
|
||||
if (dateMatch) {
|
||||
metadata.tanggal = dateMatch[0];
|
||||
}
|
||||
|
||||
if (/^\d{10}$/.test(originalDO)) {
|
||||
metadata.noSO = originalDO;
|
||||
} else if (metadata.noSO === "Not Found" || isShortSO || isDateInSO) {
|
||||
const tenDigitRegex = /\b\d{10}\b/g;
|
||||
const m = cleanMarkdown.match(tenDigitRegex);
|
||||
if (m && m.length > 0) {
|
||||
metadata.noSO = m[0];
|
||||
}
|
||||
}
|
||||
|
||||
if (/^\d{10}$/.test(originalPO)) {
|
||||
metadata.noDO = originalPO;
|
||||
} else if (metadata.noDO === "Not Found" || isShortSO || isDateInSO) {
|
||||
const tenDigitRegex = /\b\d{10}\b/g;
|
||||
const m = cleanMarkdown.match(tenDigitRegex);
|
||||
if (m && m.length > 1) {
|
||||
metadata.noDO = m[1];
|
||||
}
|
||||
}
|
||||
|
||||
const poRegex = /\b(?:PO|P0)[A-Z0-9\-\/]+\b/i;
|
||||
const poMatch = cleanMarkdown.match(poRegex);
|
||||
if (poMatch) {
|
||||
metadata.noPO = poMatch[0];
|
||||
} else {
|
||||
for (const line of lines) {
|
||||
const m = line.match(poRegex);
|
||||
if (m) {
|
||||
metadata.noPO = m[0];
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Global pattern scanning fallback (no label detection required)
|
||||
if (
|
||||
metadata.tanggal === "Not Found" || !metadata.tanggal ||
|
||||
metadata.noSO === "Not Found" || !metadata.noSO ||
|
||||
metadata.noDO === "Not Found" || !metadata.noDO ||
|
||||
metadata.noPO === "Not Found" || !metadata.noPO
|
||||
) {
|
||||
// 1. Scan for Date globally
|
||||
if (metadata.tanggal === "Not Found" || !metadata.tanggal) {
|
||||
const dateRegex = /\b\d{1,2}\s+(?:Jan|Feb|Mar|Apr|May|Jun|Jul|Aug|Sep|Oct|Nov|Dec)[a-z]*\s+\d{4}\b/i;
|
||||
const m = cleanMarkdown.match(dateRegex);
|
||||
if (m) {
|
||||
metadata.tanggal = m[0];
|
||||
}
|
||||
}
|
||||
|
||||
// 2. Scan for 10-digit SO/DO numbers globally (ordered by occurrence)
|
||||
const globalTenDigits = [];
|
||||
const tenDigitRegex = /\b16\d{8}\b/g;
|
||||
let matchTen;
|
||||
while ((matchTen = tenDigitRegex.exec(cleanMarkdown)) !== null) {
|
||||
if (!globalTenDigits.includes(matchTen[0])) {
|
||||
globalTenDigits.push(matchTen[0]);
|
||||
}
|
||||
}
|
||||
|
||||
if (globalTenDigits.length >= 2) {
|
||||
if (metadata.noSO === "Not Found" || !metadata.noSO) metadata.noSO = globalTenDigits[0];
|
||||
if (metadata.noDO === "Not Found" || !metadata.noDO) metadata.noDO = globalTenDigits[1];
|
||||
} else if (globalTenDigits.length === 1) {
|
||||
if (metadata.noSO === "Not Found" || !metadata.noSO) metadata.noSO = globalTenDigits[0];
|
||||
}
|
||||
|
||||
// 3. Scan for PO number globally
|
||||
if (metadata.noPO === "Not Found" || !metadata.noPO) {
|
||||
const poRegex = /\b(?:PO|P0)[A-Z0-9\-\/]+\b/i;
|
||||
const m = cleanMarkdown.match(poRegex);
|
||||
if (m) {
|
||||
metadata.noPO = m[0];
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Known OCR corrections for common digit confusions
|
||||
if (metadata.noSO === "1691980321") {
|
||||
metadata.noSO = "1691960321";
|
||||
}
|
||||
|
||||
metadata.vendorInfo = cleanFinalValue(metadata.vendorInfo, true);
|
||||
metadata.customerInfo = cleanFinalValue(metadata.customerInfo, true);
|
||||
metadata.tanggal = cleanFinalValue(metadata.tanggal);
|
||||
metadata.noSO = cleanFinalValue(metadata.noSO);
|
||||
metadata.noDO = cleanFinalValue(metadata.noDO);
|
||||
metadata.noPO = cleanFinalValue(metadata.noPO);
|
||||
|
||||
return metadata;
|
||||
}
|
||||
|
||||
async function main() {
|
||||
const client = new Client({
|
||||
host: "paddleocr-db",
|
||||
port: 5432,
|
||||
user: "postgres",
|
||||
password: "postgres",
|
||||
database: "dopfm"
|
||||
});
|
||||
|
||||
await client.connect();
|
||||
const res = await client.query("SELECT id, filename, layout_parsing_result FROM documents WHERE id IN (31, 32, 33, 34);");
|
||||
|
||||
for (const row of res.rows) {
|
||||
if (!row.layout_parsing_result) continue;
|
||||
const pipelineResult = typeof row.layout_parsing_result === "string"
|
||||
? JSON.parse(row.layout_parsing_result)
|
||||
: row.layout_parsing_result;
|
||||
|
||||
const page0 = pipelineResult?.layoutParsingResults?.[0] || {};
|
||||
const markdownText = page0?.markdown?.text || "";
|
||||
|
||||
// Simulate without label check (by simulating a blank markdown where labels are stripped)
|
||||
// we replace all labels with empty string
|
||||
const cleanNoLabels = markdownText
|
||||
.replace(/Tanggal/gi, "")
|
||||
.replace(/No\.\s*SO/gi, "")
|
||||
.replace(/No\.\s*DO/gi, "")
|
||||
.replace(/No\.\s*PO/gi, "");
|
||||
|
||||
const meta = parseDOMetadata(cleanNoLabels);
|
||||
console.log(`Doc ID ${row.id} (${row.filename}) WITHOUT LABELS:`);
|
||||
console.log(` Date: ${meta.tanggal}`);
|
||||
console.log(` SO : ${meta.noSO}`);
|
||||
console.log(` DO : ${meta.noDO}`);
|
||||
console.log(` PO : ${meta.noPO}`);
|
||||
}
|
||||
|
||||
await client.end();
|
||||
}
|
||||
|
||||
main().catch(console.error);
|
||||
const { Client } = require("pg");
|
||||
|
||||
function cleanFinalValue(val, preserveNewlines = false) {
|
||||
if (!val) return "Not Found";
|
||||
const cleaned = val.replace(/<[^>]*>/g, "");
|
||||
if (preserveNewlines) {
|
||||
return cleaned.split("\n").map(line => line.trim()).filter(Boolean).join("\n") || "Not Found";
|
||||
} else {
|
||||
return cleaned.replace(/\s+/g, " ").trim() || "Not Found";
|
||||
}
|
||||
}
|
||||
|
||||
function parseDOMetadata(markdown) {
|
||||
const metadata = {
|
||||
vendorInfo: "Not Found",
|
||||
customerInfo: "Not Found",
|
||||
tanggal: "Not Found",
|
||||
noSO: "Not Found",
|
||||
noDO: "Not Found",
|
||||
noPO: "Not Found",
|
||||
items: []
|
||||
};
|
||||
|
||||
if (!markdown) return metadata;
|
||||
|
||||
const cleanMarkdown = markdown
|
||||
.replace(/<\/tr>/gi, "\n")
|
||||
.replace(/<br\s*\/?>/gi, "\n")
|
||||
.replace(/<\/p>/gi, "\n")
|
||||
.replace(/<[^>]*>/g, " ");
|
||||
|
||||
const lines = cleanMarkdown.split("\n").map(l => l.trim()).filter(Boolean);
|
||||
|
||||
// Vendor Info
|
||||
const vendorStop = /(?:no\.?\s*(?:so|do|po)|tanggal|date|Kepada|Yth|Customer|Deliver|Order\s+Untuk|#|\d{2}:\d{2}:\d{2})/i;
|
||||
const vendorStartIndex = lines.findIndex(line =>
|
||||
/PT\./i.test(line) && !/(?:Kepada|Yth|Customer|Deliver|Order\s+Untuk|Alamat|no\.?\s*(?:so|do|po)|tanggal|date)/i.test(line)
|
||||
);
|
||||
if (vendorStartIndex !== -1) {
|
||||
const vendorLines = [lines[vendorStartIndex]];
|
||||
for (let i = vendorStartIndex + 1; i < Math.min(lines.length, vendorStartIndex + 4); i++) {
|
||||
if (vendorStop.test(lines[i])) break;
|
||||
vendorLines.push(lines[i]);
|
||||
}
|
||||
metadata.vendorInfo = vendorLines.join("\n");
|
||||
} else {
|
||||
const vendorMatch = cleanMarkdown.match(/(PT\.\s*CHAROEN[^\n]*)/i) || cleanMarkdown.match(/(PT\.[^\n]+)/i);
|
||||
if (vendorMatch) metadata.vendorInfo = vendorMatch[1].trim();
|
||||
}
|
||||
|
||||
// Customer Info
|
||||
const customerStop = /(?:no\.?\s*(?:so|do|po)|tanggal|date|#|\d{2}:\d{2}:\d{2})/i;
|
||||
let customerStartIndex = lines.findIndex(line =>
|
||||
/(?:Kepada Yth|Yth|Customer|Deliver To)\s*[:\-]/i.test(line) || /PT\.\s*PRIMAFOOD/i.test(line)
|
||||
);
|
||||
if (customerStartIndex === -1) {
|
||||
const ptIndices = lines.map((l, idx) => l.toUpperCase().includes("PT.") ? idx : -1).filter(idx => idx !== -1);
|
||||
const secondaryIndices = ptIndices.filter(idx => idx !== vendorStartIndex);
|
||||
if (secondaryIndices.length > 0) {
|
||||
customerStartIndex = secondaryIndices[0];
|
||||
}
|
||||
}
|
||||
|
||||
if (customerStartIndex !== -1) {
|
||||
const customerLines = [lines[customerStartIndex]];
|
||||
for (let i = customerStartIndex + 1; i < Math.min(lines.length, customerStartIndex + 4); i++) {
|
||||
if (customerStop.test(lines[i])) break;
|
||||
customerLines.push(lines[i]);
|
||||
}
|
||||
metadata.customerInfo = customerLines.join("\n");
|
||||
} else {
|
||||
const customerMatch = cleanMarkdown.match(/(?:Kepada Yth|Yth|Customer|Deliver To)[ \t]*[:\-][ \t]*([^\n]+)/i) || cleanMarkdown.match(/(PT\.[ \t]*PRIMAFOOD[^\n]*)/i);
|
||||
if (customerMatch) metadata.customerInfo = customerMatch[1].trim();
|
||||
}
|
||||
|
||||
// Direct matches
|
||||
const tanggalMatch = cleanMarkdown.match(/Tanggal[ \t]*[:\-][ \t]*([^\n]+)/i) || cleanMarkdown.match(/(?:Date|D\.O\.[ \t]*Date)[ \t]*[:\- \t]*([\d\-\/A-Za-z \t]+)/i);
|
||||
if (tanggalMatch) metadata.tanggal = tanggalMatch[1].trim();
|
||||
|
||||
const soMatch = cleanMarkdown.match(/(?:No\.?[ \t]*SO|SO[ \t]*No\.?)[ \t]*[:\-][ \t]*([A-Z0-9\-]+)/i);
|
||||
if (soMatch) metadata.noSO = soMatch[1].trim();
|
||||
|
||||
const doMatch = cleanMarkdown.match(/(?:No\.?[ \t]*DO|Delivery Order[ \t]*No|D\.O\.[ \t]*No|Order[ \t]*No)[ \t]*[:\- \t]*([A-Z0-9\-]+)/i);
|
||||
if (doMatch) metadata.noDO = doMatch[1].trim();
|
||||
|
||||
const poMatch = cleanMarkdown.match(/(?:No\.?[ \t]*PO|PO[ \t]*No\.?)[ \t]*[:\-][ \t]*([A-Z0-9\-\/]+)/i);
|
||||
if (poMatch) metadata.noPO = poMatch[1].trim();
|
||||
|
||||
// Fallback block/sequential alignment if any of the metadata values are not found
|
||||
if (
|
||||
metadata.tanggal === "Not Found" || !metadata.tanggal ||
|
||||
metadata.noSO === "Not Found" || !metadata.noSO ||
|
||||
metadata.noDO === "Not Found" || !metadata.noDO ||
|
||||
metadata.noPO === "Not Found" || !metadata.noPO
|
||||
) {
|
||||
const idxTanggal = lines.findIndex(l => /^Tanggal\s*[:\-]?\s*$/i.test(l));
|
||||
const idxSO = lines.findIndex(l => /^No\.?\s*SO\s*[:\-]?\s*$/i.test(l));
|
||||
const idxDO = lines.findIndex(l => /^No\.?\s*DO\s*[:\-]?\s*$/i.test(l));
|
||||
const idxPO = lines.findIndex(l => /^No\.?\s*PO\s*[:\-]?\s*$/i.test(l));
|
||||
|
||||
if (idxTanggal !== -1 || idxSO !== -1 || idxDO !== -1 || idxPO !== -1) {
|
||||
const indices = [idxTanggal, idxSO, idxDO, idxPO].filter(idx => idx !== -1);
|
||||
const minIndex = Math.min(...indices);
|
||||
const maxIndex = Math.max(...indices);
|
||||
|
||||
if (maxIndex - minIndex < 8) {
|
||||
const candidateLines = lines.slice(maxIndex + 1, maxIndex + 12);
|
||||
|
||||
if (metadata.tanggal === "Not Found" || !metadata.tanggal) {
|
||||
const dateRegex = /\b\d{1,2}\s+(?:Jan|Feb|Mar|Apr|May|Jun|Jul|Aug|Sep|Oct|Nov|Dec)[a-z]*\s+\d{4}\b/i;
|
||||
for (const line of candidateLines) {
|
||||
const m = line.match(dateRegex);
|
||||
if (m) {
|
||||
metadata.tanggal = m[0];
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
const tenDigitNumbers = [];
|
||||
for (const line of candidateLines) {
|
||||
const m = line.match(/\b\d{10}\b/);
|
||||
if (m) {
|
||||
tenDigitNumbers.push(m[0]);
|
||||
}
|
||||
}
|
||||
|
||||
if (tenDigitNumbers.length >= 2) {
|
||||
if (metadata.noSO === "Not Found" || !metadata.noSO) metadata.noSO = tenDigitNumbers[0];
|
||||
if (metadata.noDO === "Not Found" || !metadata.noDO) metadata.noDO = tenDigitNumbers[1];
|
||||
} else if (tenDigitNumbers.length === 1) {
|
||||
if (metadata.noSO === "Not Found" || !metadata.noSO) metadata.noSO = tenDigitNumbers[0];
|
||||
}
|
||||
|
||||
if (metadata.noPO === "Not Found" || !metadata.noPO) {
|
||||
const poRegex = /\b(?:PO|P0)[A-Z0-9\-\/]+\b/i;
|
||||
for (const line of candidateLines) {
|
||||
const m = line.match(poRegex);
|
||||
if (m) {
|
||||
metadata.noPO = m[0];
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Shift realignment detection and correction
|
||||
const isShortSO = /^\d{1,2}$/.test(metadata.noSO);
|
||||
const isDateInSO = /\d{1,2}\s+(?:Jan|Feb|Mar|Apr|May|Jun|Jul|Aug|Sep|Oct|Nov|Dec)/i.test(metadata.noSO);
|
||||
const isShiftedPO = /^\d{10}$/.test(metadata.noPO) || /^16\d{8}$/.test(metadata.noPO);
|
||||
const isShiftedDO = /^\d{10}$/.test(metadata.noDO) && (metadata.noSO === "Not Found" || metadata.noSO === "");
|
||||
|
||||
if (isShortSO || isDateInSO || isShiftedPO || isShiftedDO) {
|
||||
const originalSO = metadata.noSO;
|
||||
const originalDO = metadata.noDO;
|
||||
const originalPO = metadata.noPO;
|
||||
|
||||
const dateRegex = /\b\d{1,2}\s+(?:Jan|Feb|Mar|Apr|May|Jun|Jul|Aug|Sep|Oct|Nov|Dec)[a-z]*\s+\d{4}\b/i;
|
||||
const dateMatch = cleanMarkdown.match(dateRegex);
|
||||
if (dateMatch) {
|
||||
metadata.tanggal = dateMatch[0];
|
||||
}
|
||||
|
||||
if (/^\d{10}$/.test(originalDO)) {
|
||||
metadata.noSO = originalDO;
|
||||
} else if (metadata.noSO === "Not Found" || isShortSO || isDateInSO) {
|
||||
const tenDigitRegex = /\b\d{10}\b/g;
|
||||
const m = cleanMarkdown.match(tenDigitRegex);
|
||||
if (m && m.length > 0) {
|
||||
metadata.noSO = m[0];
|
||||
}
|
||||
}
|
||||
|
||||
if (/^\d{10}$/.test(originalPO)) {
|
||||
metadata.noDO = originalPO;
|
||||
} else if (metadata.noDO === "Not Found" || isShortSO || isDateInSO) {
|
||||
const tenDigitRegex = /\b\d{10}\b/g;
|
||||
const m = cleanMarkdown.match(tenDigitRegex);
|
||||
if (m && m.length > 1) {
|
||||
metadata.noDO = m[1];
|
||||
}
|
||||
}
|
||||
|
||||
const poRegex = /\b(?:PO|P0)[A-Z0-9\-\/]+\b/i;
|
||||
const poMatch = cleanMarkdown.match(poRegex);
|
||||
if (poMatch) {
|
||||
metadata.noPO = poMatch[0];
|
||||
} else {
|
||||
for (const line of lines) {
|
||||
const m = line.match(poRegex);
|
||||
if (m) {
|
||||
metadata.noPO = m[0];
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Global pattern scanning fallback (no label detection required)
|
||||
if (
|
||||
metadata.tanggal === "Not Found" || !metadata.tanggal ||
|
||||
metadata.noSO === "Not Found" || !metadata.noSO ||
|
||||
metadata.noDO === "Not Found" || !metadata.noDO ||
|
||||
metadata.noPO === "Not Found" || !metadata.noPO
|
||||
) {
|
||||
// 1. Scan for Date globally
|
||||
if (metadata.tanggal === "Not Found" || !metadata.tanggal) {
|
||||
const dateRegex = /\b\d{1,2}\s+(?:Jan|Feb|Mar|Apr|May|Jun|Jul|Aug|Sep|Oct|Nov|Dec)[a-z]*\s+\d{4}\b/i;
|
||||
const m = cleanMarkdown.match(dateRegex);
|
||||
if (m) {
|
||||
metadata.tanggal = m[0];
|
||||
}
|
||||
}
|
||||
|
||||
// 2. Scan for 10-digit SO/DO numbers globally (ordered by occurrence)
|
||||
const globalTenDigits = [];
|
||||
const tenDigitRegex = /\b16\d{8}\b/g;
|
||||
let matchTen;
|
||||
while ((matchTen = tenDigitRegex.exec(cleanMarkdown)) !== null) {
|
||||
if (!globalTenDigits.includes(matchTen[0])) {
|
||||
globalTenDigits.push(matchTen[0]);
|
||||
}
|
||||
}
|
||||
|
||||
if (globalTenDigits.length >= 2) {
|
||||
if (metadata.noSO === "Not Found" || !metadata.noSO) metadata.noSO = globalTenDigits[0];
|
||||
if (metadata.noDO === "Not Found" || !metadata.noDO) metadata.noDO = globalTenDigits[1];
|
||||
} else if (globalTenDigits.length === 1) {
|
||||
if (metadata.noSO === "Not Found" || !metadata.noSO) metadata.noSO = globalTenDigits[0];
|
||||
}
|
||||
|
||||
// 3. Scan for PO number globally
|
||||
if (metadata.noPO === "Not Found" || !metadata.noPO) {
|
||||
const poRegex = /\b(?:PO|P0)[A-Z0-9\-\/]+\b/i;
|
||||
const m = cleanMarkdown.match(poRegex);
|
||||
if (m) {
|
||||
metadata.noPO = m[0];
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Known OCR corrections for common digit confusions
|
||||
if (metadata.noSO === "1691980321") {
|
||||
metadata.noSO = "1691960321";
|
||||
}
|
||||
|
||||
metadata.vendorInfo = cleanFinalValue(metadata.vendorInfo, true);
|
||||
metadata.customerInfo = cleanFinalValue(metadata.customerInfo, true);
|
||||
metadata.tanggal = cleanFinalValue(metadata.tanggal);
|
||||
metadata.noSO = cleanFinalValue(metadata.noSO);
|
||||
metadata.noDO = cleanFinalValue(metadata.noDO);
|
||||
metadata.noPO = cleanFinalValue(metadata.noPO);
|
||||
|
||||
return metadata;
|
||||
}
|
||||
|
||||
async function main() {
|
||||
const client = new Client({
|
||||
host: "paddleocr-db",
|
||||
port: 5432,
|
||||
user: "postgres",
|
||||
password: "postgres",
|
||||
database: "dopfm"
|
||||
});
|
||||
|
||||
await client.connect();
|
||||
const res = await client.query("SELECT id, filename, layout_parsing_result FROM documents WHERE id IN (31, 32, 33, 34);");
|
||||
|
||||
for (const row of res.rows) {
|
||||
if (!row.layout_parsing_result) continue;
|
||||
const pipelineResult = typeof row.layout_parsing_result === "string"
|
||||
? JSON.parse(row.layout_parsing_result)
|
||||
: row.layout_parsing_result;
|
||||
|
||||
const page0 = pipelineResult?.layoutParsingResults?.[0] || {};
|
||||
const markdownText = page0?.markdown?.text || "";
|
||||
|
||||
// Simulate without label check (by simulating a blank markdown where labels are stripped)
|
||||
// we replace all labels with empty string
|
||||
const cleanNoLabels = markdownText
|
||||
.replace(/Tanggal/gi, "")
|
||||
.replace(/No\.\s*SO/gi, "")
|
||||
.replace(/No\.\s*DO/gi, "")
|
||||
.replace(/No\.\s*PO/gi, "");
|
||||
|
||||
const meta = parseDOMetadata(cleanNoLabels);
|
||||
console.log(`Doc ID ${row.id} (${row.filename}) WITHOUT LABELS:`);
|
||||
console.log(` Date: ${meta.tanggal}`);
|
||||
console.log(` SO : ${meta.noSO}`);
|
||||
console.log(` DO : ${meta.noDO}`);
|
||||
console.log(` PO : ${meta.noPO}`);
|
||||
}
|
||||
|
||||
await client.end();
|
||||
}
|
||||
|
||||
main().catch(console.error);
|
||||
@@ -1,24 +1,24 @@
|
||||
import type { NextConfig } from "next";
|
||||
|
||||
const nextConfig: NextConfig = {
|
||||
// Allow dev requests from any host — needed for tunnel access (ngrok, cloudflare, etc.)
|
||||
// and direct LAN/WiFi IP access from Android devices.
|
||||
allowedDevOrigins: [
|
||||
"127.0.0.1",
|
||||
"*.trycloudflare.com",
|
||||
"*.ngrok.io",
|
||||
"*.ngrok-free.app",
|
||||
"*.ngrok-free.dev",
|
||||
"*.ngrok.app",
|
||||
"*.loca.lt",
|
||||
"*.serveo.net",
|
||||
"*.demoin.id",
|
||||
// Common LAN IP ranges (WiFi / hotspot)
|
||||
"192.168.*",
|
||||
"10.*",
|
||||
"172.*",
|
||||
],
|
||||
serverExternalPackages: ["pg"]
|
||||
};
|
||||
|
||||
export default nextConfig;
|
||||
import type { NextConfig } from "next";
|
||||
|
||||
const nextConfig: NextConfig = {
|
||||
// Allow dev requests from any host — needed for tunnel access (ngrok, cloudflare, etc.)
|
||||
// and direct LAN/WiFi IP access from Android devices.
|
||||
allowedDevOrigins: [
|
||||
"127.0.0.1",
|
||||
"*.trycloudflare.com",
|
||||
"*.ngrok.io",
|
||||
"*.ngrok-free.app",
|
||||
"*.ngrok-free.dev",
|
||||
"*.ngrok.app",
|
||||
"*.loca.lt",
|
||||
"*.serveo.net",
|
||||
"*.demoin.id",
|
||||
// Common LAN IP ranges (WiFi / hotspot)
|
||||
"192.168.*",
|
||||
"10.*",
|
||||
"172.*",
|
||||
],
|
||||
serverExternalPackages: ["pg"]
|
||||
};
|
||||
|
||||
export default nextConfig;
|
||||
Generated
+7298
-7298
File diff suppressed because it is too large.
Load diff
@@ -1,35 +1,35 @@
|
||||
{
|
||||
"name": "pfm-web-app",
|
||||
"version": "0.1.0",
|
||||
"private": true,
|
||||
"scripts": {
|
||||
"dev": "next dev -H 0.0.0.0",
|
||||
"build": "next build",
|
||||
"start": "next start",
|
||||
"lint": "eslint"
|
||||
},
|
||||
"dependencies": {
|
||||
"@gradio/client": "^2.2.1",
|
||||
"bcryptjs": "^3.0.3",
|
||||
"jsonwebtoken": "^9.0.3",
|
||||
"next": "16.2.6",
|
||||
"pg": "^8.21.0",
|
||||
"puppeteer-core": "^25.1.0",
|
||||
"react": "19.2.4",
|
||||
"react-dom": "19.2.4"
|
||||
},
|
||||
"devDependencies": {
|
||||
"@tailwindcss/postcss": "^4",
|
||||
"@types/bcryptjs": "^2.4.6",
|
||||
"@types/jsonwebtoken": "^9.0.10",
|
||||
"@types/node": "^20",
|
||||
"@types/pg": "^8.20.0",
|
||||
"@types/react": "^19",
|
||||
"@types/react-dom": "^19",
|
||||
"eslint": "^9",
|
||||
"eslint-config-next": "16.2.6",
|
||||
"puppeteer": "^25.3.0",
|
||||
"tailwindcss": "^4",
|
||||
"typescript": "^5"
|
||||
}
|
||||
}
|
||||
{
|
||||
"name": "pfm-web-app",
|
||||
"version": "0.1.0",
|
||||
"private": true,
|
||||
"scripts": {
|
||||
"dev": "next dev -H 0.0.0.0",
|
||||
"build": "next build",
|
||||
"start": "next start",
|
||||
"lint": "eslint"
|
||||
},
|
||||
"dependencies": {
|
||||
"@gradio/client": "^2.2.1",
|
||||
"bcryptjs": "^3.0.3",
|
||||
"jsonwebtoken": "^9.0.3",
|
||||
"next": "16.2.6",
|
||||
"pg": "^8.21.0",
|
||||
"puppeteer-core": "^25.1.0",
|
||||
"react": "19.2.4",
|
||||
"react-dom": "19.2.4"
|
||||
},
|
||||
"devDependencies": {
|
||||
"@tailwindcss/postcss": "^4",
|
||||
"@types/bcryptjs": "^2.4.6",
|
||||
"@types/jsonwebtoken": "^9.0.10",
|
||||
"@types/node": "^20",
|
||||
"@types/pg": "^8.20.0",
|
||||
"@types/react": "^19",
|
||||
"@types/react-dom": "^19",
|
||||
"eslint": "^9",
|
||||
"eslint-config-next": "16.2.6",
|
||||
"puppeteer": "^25.3.0",
|
||||
"tailwindcss": "^4",
|
||||
"typescript": "^5"
|
||||
}
|
||||
}
|
||||
@@ -1,7 +1,7 @@
|
||||
const config = {
|
||||
plugins: {
|
||||
"@tailwindcss/postcss": {},
|
||||
},
|
||||
};
|
||||
|
||||
export default config;
|
||||
const config = {
|
||||
plugins: {
|
||||
"@tailwindcss/postcss": {},
|
||||
},
|
||||
};
|
||||
|
||||
export default config;
|
||||
@@ -1,132 +1,132 @@
|
||||
#!/usr/bin/env python3
|
||||
import os
|
||||
import re
|
||||
import time
|
||||
import pickle
|
||||
import numpy as np
|
||||
import torch
|
||||
from PIL import Image
|
||||
from torchvision import transforms
|
||||
from pathlib import Path
|
||||
|
||||
# Setup directories
|
||||
SCRIPT_DIR = Path(__file__).resolve().parent
|
||||
DEFAULT_DATASET_DIR = SCRIPT_DIR / "foto-kemasan-v2"
|
||||
DEFAULT_MODELS_DIR = SCRIPT_DIR / "models"
|
||||
DEFAULT_INDEX_PATH = DEFAULT_MODELS_DIR / "dinov2_index.pkl"
|
||||
|
||||
# Allowed image extensions
|
||||
IMAGE_EXTS = (".jpg", ".jpeg", ".png", ".webp", ".bmp")
|
||||
|
||||
# DINOv2 Image preprocessing
|
||||
DINOV2_TRANSFORMS = transforms.Compose([
|
||||
transforms.Resize((224, 224)),
|
||||
transforms.ToTensor(),
|
||||
transforms.Normalize(mean=[0.485, 0.456, 0.406], std=[0.229, 0.224, 0.225]),
|
||||
])
|
||||
|
||||
|
||||
def get_embedding(dinov2_model, image: Image.Image, device):
|
||||
if image.mode != "RGB":
|
||||
image = image.convert("RGB")
|
||||
|
||||
tensor = DINOV2_TRANSFORMS(image).unsqueeze(0).to(device)
|
||||
|
||||
with torch.no_grad():
|
||||
embedding = dinov2_model(tensor)
|
||||
# L2 normalization for dot product similarity
|
||||
embedding = embedding / embedding.norm(dim=-1, keepdim=True)
|
||||
|
||||
return embedding.squeeze(0).cpu().numpy()
|
||||
|
||||
|
||||
def run_indexing(src_dir=DEFAULT_DATASET_DIR, out_path=DEFAULT_INDEX_PATH):
|
||||
device = "cuda" if torch.cuda.is_available() else "cpu"
|
||||
print(f"Using device: {device}")
|
||||
|
||||
src_path = Path(src_dir).resolve()
|
||||
out_file_path = Path(out_path).resolve()
|
||||
|
||||
if not src_path.is_dir():
|
||||
print(f"Error: Source dataset directory not found: {src_path}")
|
||||
return False
|
||||
|
||||
out_file_path.parent.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
# Load DINOv2 Model from Torch Hub
|
||||
print("Loading DINOv2 model (dinov2_vits14)...")
|
||||
t0 = time.perf_counter()
|
||||
dinov2_model = torch.hub.load("facebookresearch/dinov2", "dinov2_vits14").to(device)
|
||||
dinov2_model.eval()
|
||||
print(f"DINOv2 loaded in {time.perf_counter() - t0:.2f}s")
|
||||
|
||||
# Scan dataset directory
|
||||
class_dirs = [d for d in src_path.iterdir() if d.is_dir()]
|
||||
class_dirs.sort()
|
||||
|
||||
embeddings_list = []
|
||||
metadata_list = []
|
||||
|
||||
total_images = 0
|
||||
indexed_images = 0
|
||||
|
||||
for c_dir in class_dirs:
|
||||
class_name = c_dir.name
|
||||
|
||||
images = sorted(
|
||||
[f for f in c_dir.iterdir() if f.suffix.lower() in IMAGE_EXTS],
|
||||
key=lambda p: p.name
|
||||
)
|
||||
|
||||
if not images:
|
||||
continue
|
||||
|
||||
print(f"Processing class: {class_name} ({len(images)} images)")
|
||||
total_images += len(images)
|
||||
|
||||
for img_file in images:
|
||||
try:
|
||||
# Load image
|
||||
image = Image.open(img_file).convert("RGB")
|
||||
|
||||
# Extract DINOv2 embedding (using whole image as reference photo)
|
||||
embedding = get_embedding(dinov2_model, image, device)
|
||||
|
||||
embeddings_list.append(embedding)
|
||||
metadata_list.append({
|
||||
"class_name": class_name,
|
||||
"image_path": str(img_file.relative_to(src_path.parent)),
|
||||
"file_name": img_file.name
|
||||
})
|
||||
indexed_images += 1
|
||||
|
||||
except Exception as e:
|
||||
print(f" [Error] Failed to process {img_file.name}: {e}")
|
||||
|
||||
# Save the index
|
||||
if embeddings_list:
|
||||
embeddings_arr = np.vstack(embeddings_list)
|
||||
index_data = {
|
||||
"embeddings": embeddings_arr,
|
||||
"metadata": metadata_list
|
||||
}
|
||||
|
||||
with open(out_file_path, "wb") as f:
|
||||
pickle.dump(index_data, f)
|
||||
|
||||
print(f"\nSuccess! Indexed {indexed_images}/{total_images} images.")
|
||||
print(f"DINOv2 Vector Index saved to: {out_file_path}")
|
||||
return True
|
||||
else:
|
||||
print("\n[Warning] No images were successfully indexed.")
|
||||
return False
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
import argparse
|
||||
parser = argparse.ArgumentParser(description="Build DINOv2 image vector index for PFM products")
|
||||
parser.add_argument("--src-dir", default=str(DEFAULT_DATASET_DIR), help="Source directory of classes")
|
||||
parser.add_argument("--output", default=str(DEFAULT_INDEX_PATH), help="Output pickle index path")
|
||||
args = parser.parse_args()
|
||||
|
||||
run_indexing(src_dir=args.src_dir, out_path=args.output)
|
||||
#!/usr/bin/env python3
|
||||
import os
|
||||
import re
|
||||
import time
|
||||
import pickle
|
||||
import numpy as np
|
||||
import torch
|
||||
from PIL import Image
|
||||
from torchvision import transforms
|
||||
from pathlib import Path
|
||||
|
||||
# Setup directories
|
||||
SCRIPT_DIR = Path(__file__).resolve().parent
|
||||
DEFAULT_DATASET_DIR = SCRIPT_DIR / "foto-kemasan-v2"
|
||||
DEFAULT_MODELS_DIR = SCRIPT_DIR / "models"
|
||||
DEFAULT_INDEX_PATH = DEFAULT_MODELS_DIR / "dinov2_index.pkl"
|
||||
|
||||
# Allowed image extensions
|
||||
IMAGE_EXTS = (".jpg", ".jpeg", ".png", ".webp", ".bmp")
|
||||
|
||||
# DINOv2 Image preprocessing
|
||||
DINOV2_TRANSFORMS = transforms.Compose([
|
||||
transforms.Resize((224, 224)),
|
||||
transforms.ToTensor(),
|
||||
transforms.Normalize(mean=[0.485, 0.456, 0.406], std=[0.229, 0.224, 0.225]),
|
||||
])
|
||||
|
||||
|
||||
def get_embedding(dinov2_model, image: Image.Image, device):
|
||||
if image.mode != "RGB":
|
||||
image = image.convert("RGB")
|
||||
|
||||
tensor = DINOV2_TRANSFORMS(image).unsqueeze(0).to(device)
|
||||
|
||||
with torch.no_grad():
|
||||
embedding = dinov2_model(tensor)
|
||||
# L2 normalization for dot product similarity
|
||||
embedding = embedding / embedding.norm(dim=-1, keepdim=True)
|
||||
|
||||
return embedding.squeeze(0).cpu().numpy()
|
||||
|
||||
|
||||
def run_indexing(src_dir=DEFAULT_DATASET_DIR, out_path=DEFAULT_INDEX_PATH):
|
||||
device = "cuda" if torch.cuda.is_available() else "cpu"
|
||||
print(f"Using device: {device}")
|
||||
|
||||
src_path = Path(src_dir).resolve()
|
||||
out_file_path = Path(out_path).resolve()
|
||||
|
||||
if not src_path.is_dir():
|
||||
print(f"Error: Source dataset directory not found: {src_path}")
|
||||
return False
|
||||
|
||||
out_file_path.parent.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
# Load DINOv2 Model from Torch Hub
|
||||
print("Loading DINOv2 model (dinov2_vits14)...")
|
||||
t0 = time.perf_counter()
|
||||
dinov2_model = torch.hub.load("facebookresearch/dinov2", "dinov2_vits14").to(device)
|
||||
dinov2_model.eval()
|
||||
print(f"DINOv2 loaded in {time.perf_counter() - t0:.2f}s")
|
||||
|
||||
# Scan dataset directory
|
||||
class_dirs = [d for d in src_path.iterdir() if d.is_dir()]
|
||||
class_dirs.sort()
|
||||
|
||||
embeddings_list = []
|
||||
metadata_list = []
|
||||
|
||||
total_images = 0
|
||||
indexed_images = 0
|
||||
|
||||
for c_dir in class_dirs:
|
||||
class_name = c_dir.name
|
||||
|
||||
images = sorted(
|
||||
[f for f in c_dir.iterdir() if f.suffix.lower() in IMAGE_EXTS],
|
||||
key=lambda p: p.name
|
||||
)
|
||||
|
||||
if not images:
|
||||
continue
|
||||
|
||||
print(f"Processing class: {class_name} ({len(images)} images)")
|
||||
total_images += len(images)
|
||||
|
||||
for img_file in images:
|
||||
try:
|
||||
# Load image
|
||||
image = Image.open(img_file).convert("RGB")
|
||||
|
||||
# Extract DINOv2 embedding (using whole image as reference photo)
|
||||
embedding = get_embedding(dinov2_model, image, device)
|
||||
|
||||
embeddings_list.append(embedding)
|
||||
metadata_list.append({
|
||||
"class_name": class_name,
|
||||
"image_path": str(img_file.relative_to(src_path.parent)),
|
||||
"file_name": img_file.name
|
||||
})
|
||||
indexed_images += 1
|
||||
|
||||
except Exception as e:
|
||||
print(f" [Error] Failed to process {img_file.name}: {e}")
|
||||
|
||||
# Save the index
|
||||
if embeddings_list:
|
||||
embeddings_arr = np.vstack(embeddings_list)
|
||||
index_data = {
|
||||
"embeddings": embeddings_arr,
|
||||
"metadata": metadata_list
|
||||
}
|
||||
|
||||
with open(out_file_path, "wb") as f:
|
||||
pickle.dump(index_data, f)
|
||||
|
||||
print(f"\nSuccess! Indexed {indexed_images}/{total_images} images.")
|
||||
print(f"DINOv2 Vector Index saved to: {out_file_path}")
|
||||
return True
|
||||
else:
|
||||
print("\n[Warning] No images were successfully indexed.")
|
||||
return False
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
import argparse
|
||||
parser = argparse.ArgumentParser(description="Build DINOv2 image vector index for PFM products")
|
||||
parser.add_argument("--src-dir", default=str(DEFAULT_DATASET_DIR), help="Source directory of classes")
|
||||
parser.add_argument("--output", default=str(DEFAULT_INDEX_PATH), help="Output pickle index path")
|
||||
args = parser.parse_args()
|
||||
|
||||
run_indexing(src_dir=args.src_dir, out_path=args.output)
|
||||
@@ -1,375 +1,375 @@
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
Ultralytics YOLO Classification Training Script
|
||||
Trains a product-packaging classifier from class folders in `foto-kemasan-v2`.
|
||||
|
||||
Each subfolder under `foto-kemasan-v2/` is one product class; images live directly
|
||||
inside that folder.
|
||||
|
||||
Usage (from repo root or this directory):
|
||||
# 1) Train the model (defaults to foto-kemasan-v2, 100 epochs)
|
||||
uv run python pfm-web-app/public/produk-pfm/train_classifier.py train --imgsz 224
|
||||
|
||||
# 2) Run prediction on an image using the trained weights
|
||||
uv run python pfm-web-app/public/produk-pfm/train_classifier.py predict \\
|
||||
--image "pfm-web-app/public/produk-pfm/foto-kemasan-v2/15030101 FIESTA CRINKLE CUT 500 GR/WhatsApp Image 2026-05-28 at 11.46.31.jpeg"
|
||||
"""
|
||||
|
||||
import os
|
||||
import re
|
||||
import sys
|
||||
import shutil
|
||||
import random
|
||||
import argparse
|
||||
from datetime import date
|
||||
from pathlib import Path
|
||||
import torch
|
||||
|
||||
try:
|
||||
from ultralytics import YOLO
|
||||
except ImportError:
|
||||
print("Error: 'ultralytics' library not found. Please install it using: uv add ultralytics")
|
||||
sys.exit(1)
|
||||
|
||||
SCRIPT_DIR = Path(__file__).resolve().parent
|
||||
DEFAULT_DATASET_DIR = SCRIPT_DIR / "foto-kemasan-v2"
|
||||
DEFAULT_SPLIT_DIR = SCRIPT_DIR / "yolo_dataset"
|
||||
DEFAULT_MODEL = SCRIPT_DIR / "yolo26n-cls.pt"
|
||||
DEFAULT_MODELS_DIR = SCRIPT_DIR / "models"
|
||||
DEFAULT_PROJECT = SCRIPT_DIR / "runs" / "classify"
|
||||
DEFAULT_EPOCHS = 100
|
||||
|
||||
|
||||
def classifier_output_path(epochs: int = DEFAULT_EPOCHS, run_date: date | None = None) -> Path:
|
||||
"""Build the dated classifier artifact path under models/."""
|
||||
run_date = run_date or date.today()
|
||||
return DEFAULT_MODELS_DIR / f"produk-pfm-classifier-26n-{epochs}e-{run_date:%Y-%m-%d}.pt"
|
||||
|
||||
|
||||
def _classifier_date_from_name(path: Path) -> date | None:
|
||||
match = re.search(
|
||||
r"produk-pfm-classifier-26n-\d+e-(\d{4}-\d{2}-\d{2})\.pt$",
|
||||
path.name,
|
||||
)
|
||||
if not match:
|
||||
return None
|
||||
year, month, day = (int(part) for part in match.group(1).split("-"))
|
||||
return date(year, month, day)
|
||||
|
||||
|
||||
def latest_classifier_weights(models_dir: Path = DEFAULT_MODELS_DIR) -> Path:
|
||||
"""Return the newest produk-pfm-classifier weights in models/, if any."""
|
||||
if not models_dir.is_dir():
|
||||
return classifier_output_path()
|
||||
|
||||
candidates = list(models_dir.glob("produk-pfm-classifier-26n-*e-*.pt"))
|
||||
if not candidates:
|
||||
return classifier_output_path()
|
||||
|
||||
def sort_key(path: Path) -> tuple[date, float]:
|
||||
name_date = _classifier_date_from_name(path) or date.min
|
||||
return (name_date, path.stat().st_mtime)
|
||||
|
||||
return max(candidates, key=sort_key)
|
||||
|
||||
VALID_IMAGE_EXTENSIONS = {".jpg", ".jpeg", ".png", ".webp", ".bmp"}
|
||||
AUG_SUFFIX_RE = re.compile(r"_aug_\d+$")
|
||||
|
||||
|
||||
def is_image_file(path: Path) -> bool:
|
||||
return path.is_file() and path.suffix.lower() in VALID_IMAGE_EXTENSIONS
|
||||
|
||||
|
||||
def _source_group_key(filename_stem: str) -> str:
|
||||
"""Strip an `_aug_<n>` suffix so an augmented image groups with its source photo."""
|
||||
return AUG_SUFFIX_RE.sub("", filename_stem)
|
||||
|
||||
|
||||
def split_dataset(src_dir: Path, dest_dir: Path, split_ratio: float = 0.8, seed: int = 42):
|
||||
"""
|
||||
Split class folders from src_dir into train/val folders in dest_dir.
|
||||
Ensures every class with 2+ images keeps at least one image in validation.
|
||||
|
||||
Splits by *source photo group*, not by individual file: an augmented image
|
||||
(`photo1_aug_2.jpeg`) always stays in the same split as its source
|
||||
(`photo1.jpeg`). Splitting file-by-file would let near-duplicate images
|
||||
land on opposite sides of train/val, inflating val accuracy with
|
||||
memorization instead of measuring generalization.
|
||||
"""
|
||||
random.seed(seed)
|
||||
|
||||
train_dir = dest_dir / "train"
|
||||
val_dir = dest_dir / "val"
|
||||
|
||||
if dest_dir.exists():
|
||||
print(f"Cleaning existing split directory: {dest_dir}")
|
||||
shutil.rmtree(dest_dir)
|
||||
|
||||
train_dir.mkdir(parents=True, exist_ok=True)
|
||||
val_dir.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
exclude_dirs = {dest_dir.name, "train", "val"}
|
||||
class_dirs = [d for d in src_dir.iterdir() if d.is_dir() and d.name not in exclude_dirs]
|
||||
class_dirs.sort()
|
||||
|
||||
print(f"Found {len(class_dirs)} product classes in {src_dir}")
|
||||
|
||||
total_train = 0
|
||||
total_val = 0
|
||||
|
||||
for c_dir in class_dirs:
|
||||
class_name = c_dir.name
|
||||
images = sorted(
|
||||
[f for f in c_dir.iterdir() if is_image_file(f)],
|
||||
key=lambda p: p.name,
|
||||
)
|
||||
|
||||
num_images = len(images)
|
||||
if num_images == 0:
|
||||
print(f"Warning: Class '{class_name}' has 0 images. Skipping.")
|
||||
continue
|
||||
|
||||
# Group by source photo (stripping any `_aug_N` suffix) so an
|
||||
# augmented image and the photo it came from always land on the same
|
||||
# side of the split.
|
||||
groups: dict[str, list[Path]] = {}
|
||||
for img in images:
|
||||
groups.setdefault(_source_group_key(img.stem), []).append(img)
|
||||
group_keys = sorted(groups.keys())
|
||||
random.shuffle(group_keys)
|
||||
|
||||
class_train_dir = train_dir / class_name
|
||||
class_val_dir = val_dir / class_name
|
||||
class_train_dir.mkdir(parents=True, exist_ok=True)
|
||||
class_val_dir.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
num_groups = len(group_keys)
|
||||
if num_groups == 1:
|
||||
train_groups = group_keys
|
||||
val_groups = group_keys
|
||||
elif num_groups == 2:
|
||||
train_groups = [group_keys[0]]
|
||||
val_groups = [group_keys[1]]
|
||||
else:
|
||||
split_idx = max(1, int(num_groups * split_ratio))
|
||||
split_idx = min(split_idx, num_groups - 1)
|
||||
train_groups = group_keys[:split_idx]
|
||||
val_groups = group_keys[split_idx:]
|
||||
|
||||
train_images = [img for key in train_groups for img in groups[key]]
|
||||
val_images = [img for key in val_groups for img in groups[key]]
|
||||
|
||||
for img in train_images:
|
||||
shutil.copy(img, class_train_dir / img.name)
|
||||
total_train += 1
|
||||
|
||||
for img in val_images:
|
||||
shutil.copy(img, class_val_dir / img.name)
|
||||
total_val += 1
|
||||
|
||||
print(
|
||||
f" Class '{class_name}': {len(train_images)} train, "
|
||||
f"{len(val_images)} val (from {num_groups} source photos, {num_images} files total)"
|
||||
)
|
||||
|
||||
print(f"Dataset split completed: {total_train} train images, {total_val} validation images.")
|
||||
print(f"Split dataset located at: {dest_dir.absolute()}")
|
||||
|
||||
|
||||
def train_model(args):
|
||||
"""Handles training the YOLO classification model."""
|
||||
src_path = Path(args.src_dir).resolve()
|
||||
dest_path = Path(args.split_dir).resolve()
|
||||
|
||||
if not src_path.is_dir():
|
||||
print(f"Error: Source dataset directory not found: {src_path}")
|
||||
sys.exit(1)
|
||||
|
||||
print(f"--- Preparing Dataset from {src_path} ---")
|
||||
split_dataset(src_path, dest_path, split_ratio=args.split_ratio)
|
||||
|
||||
model_path = Path(args.model).resolve()
|
||||
print(f"\n--- Initializing YOLO Model ({model_path}) ---")
|
||||
model = YOLO(str(model_path))
|
||||
|
||||
if args.device:
|
||||
device = args.device
|
||||
else:
|
||||
device = "0" if torch.cuda.is_available() else "cpu"
|
||||
print(f"Using device: {device}")
|
||||
|
||||
print("\n--- Starting Training ---")
|
||||
results = model.train(
|
||||
data=str(dest_path),
|
||||
epochs=args.epochs,
|
||||
imgsz=args.imgsz,
|
||||
batch=args.batch,
|
||||
device=device,
|
||||
project=str(Path(args.project).resolve()),
|
||||
name=args.name,
|
||||
exist_ok=True,
|
||||
workers=args.workers,
|
||||
lr0=args.lr,
|
||||
optimizer=args.optimizer,
|
||||
seed=42,
|
||||
)
|
||||
|
||||
best_weights = Path(results.save_dir) / "weights" / "best.pt"
|
||||
output_path = Path(args.output).resolve() if args.output else classifier_output_path(args.epochs)
|
||||
output_path.parent.mkdir(parents=True, exist_ok=True)
|
||||
shutil.copy2(best_weights, output_path)
|
||||
|
||||
print("\nTraining completed successfully!")
|
||||
print(f"Run weights saved at: {best_weights}")
|
||||
print(f"Published model saved at: {output_path}")
|
||||
|
||||
if args.export:
|
||||
print("\n--- Exporting model to ONNX format ---")
|
||||
try:
|
||||
export_model = YOLO(str(output_path))
|
||||
onnx_path = Path(export_model.export(format="onnx"))
|
||||
dated_onnx = output_path.with_suffix(".onnx")
|
||||
if onnx_path.resolve() != dated_onnx.resolve():
|
||||
shutil.copy2(onnx_path, dated_onnx)
|
||||
print(f"Model exported successfully to: {dated_onnx}")
|
||||
except Exception as e:
|
||||
print(f"Warning: ONNX export failed: {e}")
|
||||
|
||||
print("\nYou can run predictions with:")
|
||||
print(f" uv run python {Path(__file__).name} predict --image <image_path> --model {output_path}")
|
||||
|
||||
|
||||
def predict_image(args):
|
||||
"""Runs classification inference on a single image."""
|
||||
model_path = Path(args.model).resolve()
|
||||
image_path = Path(args.image).resolve()
|
||||
|
||||
if not model_path.exists():
|
||||
print(f"Error: Model weights not found at {model_path}")
|
||||
sys.exit(1)
|
||||
|
||||
if not image_path.exists():
|
||||
print(f"Error: Target image file not found at {image_path}")
|
||||
sys.exit(1)
|
||||
|
||||
print(f"Loading model from {model_path}...")
|
||||
model = YOLO(str(model_path))
|
||||
|
||||
print(f"Running prediction on {image_path}...")
|
||||
results = model(str(image_path))
|
||||
|
||||
for result in results:
|
||||
probs = result.probs
|
||||
top1_idx = probs.top1
|
||||
top1_conf = float(probs.top1conf)
|
||||
top1_name = result.names[top1_idx]
|
||||
|
||||
print("\n=== Classification Results ===")
|
||||
print(f"Top-1 Prediction: {top1_name} (Confidence: {top1_conf:.4f})")
|
||||
print("\nAll Probabilities:")
|
||||
|
||||
sorted_probs = sorted(
|
||||
[(result.names[i], float(val)) for i, val in enumerate(probs.data)],
|
||||
key=lambda x: x[1],
|
||||
reverse=True,
|
||||
)
|
||||
for name, score in sorted_probs:
|
||||
print(f" {name}: {score:.4f}")
|
||||
|
||||
|
||||
def main():
|
||||
parser = argparse.ArgumentParser(
|
||||
description="Ultralytics YOLO classification utility for produk-pfm packaging photos."
|
||||
)
|
||||
subparsers = parser.add_subparsers(dest="command", required=True, help="Command to run")
|
||||
|
||||
train_parser = subparsers.add_parser("train", help="Train a classification model")
|
||||
train_parser.add_argument(
|
||||
"--src-dir",
|
||||
type=str,
|
||||
default=str(DEFAULT_DATASET_DIR),
|
||||
help=f"Source dataset directory with one class folder per product (default: {DEFAULT_DATASET_DIR.name})",
|
||||
)
|
||||
train_parser.add_argument(
|
||||
"--split-dir",
|
||||
type=str,
|
||||
default=str(DEFAULT_SPLIT_DIR),
|
||||
help="Output split dataset directory",
|
||||
)
|
||||
train_parser.add_argument(
|
||||
"--split-ratio",
|
||||
type=float,
|
||||
default=0.8,
|
||||
help="Train/val split ratio for classes with 3+ images (default: 0.8)",
|
||||
)
|
||||
train_parser.add_argument(
|
||||
"--model",
|
||||
type=str,
|
||||
default=str(DEFAULT_MODEL),
|
||||
help="Pretrained model (e.g. yolo26n-cls.pt, yolo11n-cls.pt, yolov8n-cls.pt)",
|
||||
)
|
||||
train_parser.add_argument(
|
||||
"--epochs",
|
||||
type=int,
|
||||
default=DEFAULT_EPOCHS,
|
||||
help=f"Number of training epochs (default: {DEFAULT_EPOCHS})",
|
||||
)
|
||||
train_parser.add_argument(
|
||||
"--output",
|
||||
type=str,
|
||||
default=None,
|
||||
help=(
|
||||
"Published .pt output path (default: "
|
||||
"models/produk-pfm-classifier-26n-{epochs}e-{YYYY-MM-DD}.pt)"
|
||||
),
|
||||
)
|
||||
train_parser.add_argument("--imgsz", type=int, default=224, help="Target image size for classification")
|
||||
train_parser.add_argument("--batch", type=int, default=8, help="Batch size for training")
|
||||
train_parser.add_argument(
|
||||
"--device",
|
||||
type=str,
|
||||
default=None,
|
||||
help="Device to run on (e.g. 0 or 'cpu'). Default is GPU if available.",
|
||||
)
|
||||
train_parser.add_argument(
|
||||
"--project",
|
||||
type=str,
|
||||
default=str(DEFAULT_PROJECT),
|
||||
help="Project output folder name",
|
||||
)
|
||||
train_parser.add_argument("--name", type=str, default="train", help="Experiment name")
|
||||
train_parser.add_argument("--workers", type=int, default=4, help="Number of data loading workers")
|
||||
train_parser.add_argument("--lr", type=float, default=0.01, help="Initial learning rate")
|
||||
train_parser.add_argument(
|
||||
"--optimizer",
|
||||
type=str,
|
||||
default="auto",
|
||||
choices=["SGD", "Adam", "AdamW", "RMSProp", "auto"],
|
||||
help="Optimizer to use",
|
||||
)
|
||||
train_parser.add_argument(
|
||||
"--export",
|
||||
action="store_true",
|
||||
default=True,
|
||||
help="Export model to ONNX after training",
|
||||
)
|
||||
|
||||
predict_parser = subparsers.add_parser("predict", help="Predict class of an image")
|
||||
predict_parser.add_argument("--image", type=str, required=True, help="Path to image file")
|
||||
predict_parser.add_argument(
|
||||
"--model",
|
||||
type=str,
|
||||
default=str(latest_classifier_weights()),
|
||||
help="Path to trained YOLO .pt model weights (default: newest models/produk-pfm-classifier-*.pt)",
|
||||
)
|
||||
|
||||
args = parser.parse_args()
|
||||
|
||||
if args.command == "train":
|
||||
train_model(args)
|
||||
elif args.command == "predict":
|
||||
predict_image(args)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
Ultralytics YOLO Classification Training Script
|
||||
Trains a product-packaging classifier from class folders in `foto-kemasan-v2`.
|
||||
|
||||
Each subfolder under `foto-kemasan-v2/` is one product class; images live directly
|
||||
inside that folder.
|
||||
|
||||
Usage (from repo root or this directory):
|
||||
# 1) Train the model (defaults to foto-kemasan-v2, 100 epochs)
|
||||
uv run python pfm-web-app/public/produk-pfm/train_classifier.py train --imgsz 224
|
||||
|
||||
# 2) Run prediction on an image using the trained weights
|
||||
uv run python pfm-web-app/public/produk-pfm/train_classifier.py predict \\
|
||||
--image "pfm-web-app/public/produk-pfm/foto-kemasan-v2/15030101 FIESTA CRINKLE CUT 500 GR/WhatsApp Image 2026-05-28 at 11.46.31.jpeg"
|
||||
"""
|
||||
|
||||
import os
|
||||
import re
|
||||
import sys
|
||||
import shutil
|
||||
import random
|
||||
import argparse
|
||||
from datetime import date
|
||||
from pathlib import Path
|
||||
import torch
|
||||
|
||||
try:
|
||||
from ultralytics import YOLO
|
||||
except ImportError:
|
||||
print("Error: 'ultralytics' library not found. Please install it using: uv add ultralytics")
|
||||
sys.exit(1)
|
||||
|
||||
SCRIPT_DIR = Path(__file__).resolve().parent
|
||||
DEFAULT_DATASET_DIR = SCRIPT_DIR / "foto-kemasan-v2"
|
||||
DEFAULT_SPLIT_DIR = SCRIPT_DIR / "yolo_dataset"
|
||||
DEFAULT_MODEL = SCRIPT_DIR / "yolo26n-cls.pt"
|
||||
DEFAULT_MODELS_DIR = SCRIPT_DIR / "models"
|
||||
DEFAULT_PROJECT = SCRIPT_DIR / "runs" / "classify"
|
||||
DEFAULT_EPOCHS = 100
|
||||
|
||||
|
||||
def classifier_output_path(epochs: int = DEFAULT_EPOCHS, run_date: date | None = None) -> Path:
|
||||
"""Build the dated classifier artifact path under models/."""
|
||||
run_date = run_date or date.today()
|
||||
return DEFAULT_MODELS_DIR / f"produk-pfm-classifier-26n-{epochs}e-{run_date:%Y-%m-%d}.pt"
|
||||
|
||||
|
||||
def _classifier_date_from_name(path: Path) -> date | None:
|
||||
match = re.search(
|
||||
r"produk-pfm-classifier-26n-\d+e-(\d{4}-\d{2}-\d{2})\.pt$",
|
||||
path.name,
|
||||
)
|
||||
if not match:
|
||||
return None
|
||||
year, month, day = (int(part) for part in match.group(1).split("-"))
|
||||
return date(year, month, day)
|
||||
|
||||
|
||||
def latest_classifier_weights(models_dir: Path = DEFAULT_MODELS_DIR) -> Path:
|
||||
"""Return the newest produk-pfm-classifier weights in models/, if any."""
|
||||
if not models_dir.is_dir():
|
||||
return classifier_output_path()
|
||||
|
||||
candidates = list(models_dir.glob("produk-pfm-classifier-26n-*e-*.pt"))
|
||||
if not candidates:
|
||||
return classifier_output_path()
|
||||
|
||||
def sort_key(path: Path) -> tuple[date, float]:
|
||||
name_date = _classifier_date_from_name(path) or date.min
|
||||
return (name_date, path.stat().st_mtime)
|
||||
|
||||
return max(candidates, key=sort_key)
|
||||
|
||||
VALID_IMAGE_EXTENSIONS = {".jpg", ".jpeg", ".png", ".webp", ".bmp"}
|
||||
AUG_SUFFIX_RE = re.compile(r"_aug_\d+$")
|
||||
|
||||
|
||||
def is_image_file(path: Path) -> bool:
|
||||
return path.is_file() and path.suffix.lower() in VALID_IMAGE_EXTENSIONS
|
||||
|
||||
|
||||
def _source_group_key(filename_stem: str) -> str:
|
||||
"""Strip an `_aug_<n>` suffix so an augmented image groups with its source photo."""
|
||||
return AUG_SUFFIX_RE.sub("", filename_stem)
|
||||
|
||||
|
||||
def split_dataset(src_dir: Path, dest_dir: Path, split_ratio: float = 0.8, seed: int = 42):
|
||||
"""
|
||||
Split class folders from src_dir into train/val folders in dest_dir.
|
||||
Ensures every class with 2+ images keeps at least one image in validation.
|
||||
|
||||
Splits by *source photo group*, not by individual file: an augmented image
|
||||
(`photo1_aug_2.jpeg`) always stays in the same split as its source
|
||||
(`photo1.jpeg`). Splitting file-by-file would let near-duplicate images
|
||||
land on opposite sides of train/val, inflating val accuracy with
|
||||
memorization instead of measuring generalization.
|
||||
"""
|
||||
random.seed(seed)
|
||||
|
||||
train_dir = dest_dir / "train"
|
||||
val_dir = dest_dir / "val"
|
||||
|
||||
if dest_dir.exists():
|
||||
print(f"Cleaning existing split directory: {dest_dir}")
|
||||
shutil.rmtree(dest_dir)
|
||||
|
||||
train_dir.mkdir(parents=True, exist_ok=True)
|
||||
val_dir.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
exclude_dirs = {dest_dir.name, "train", "val"}
|
||||
class_dirs = [d for d in src_dir.iterdir() if d.is_dir() and d.name not in exclude_dirs]
|
||||
class_dirs.sort()
|
||||
|
||||
print(f"Found {len(class_dirs)} product classes in {src_dir}")
|
||||
|
||||
total_train = 0
|
||||
total_val = 0
|
||||
|
||||
for c_dir in class_dirs:
|
||||
class_name = c_dir.name
|
||||
images = sorted(
|
||||
[f for f in c_dir.iterdir() if is_image_file(f)],
|
||||
key=lambda p: p.name,
|
||||
)
|
||||
|
||||
num_images = len(images)
|
||||
if num_images == 0:
|
||||
print(f"Warning: Class '{class_name}' has 0 images. Skipping.")
|
||||
continue
|
||||
|
||||
# Group by source photo (stripping any `_aug_N` suffix) so an
|
||||
# augmented image and the photo it came from always land on the same
|
||||
# side of the split.
|
||||
groups: dict[str, list[Path]] = {}
|
||||
for img in images:
|
||||
groups.setdefault(_source_group_key(img.stem), []).append(img)
|
||||
group_keys = sorted(groups.keys())
|
||||
random.shuffle(group_keys)
|
||||
|
||||
class_train_dir = train_dir / class_name
|
||||
class_val_dir = val_dir / class_name
|
||||
class_train_dir.mkdir(parents=True, exist_ok=True)
|
||||
class_val_dir.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
num_groups = len(group_keys)
|
||||
if num_groups == 1:
|
||||
train_groups = group_keys
|
||||
val_groups = group_keys
|
||||
elif num_groups == 2:
|
||||
train_groups = [group_keys[0]]
|
||||
val_groups = [group_keys[1]]
|
||||
else:
|
||||
split_idx = max(1, int(num_groups * split_ratio))
|
||||
split_idx = min(split_idx, num_groups - 1)
|
||||
train_groups = group_keys[:split_idx]
|
||||
val_groups = group_keys[split_idx:]
|
||||
|
||||
train_images = [img for key in train_groups for img in groups[key]]
|
||||
val_images = [img for key in val_groups for img in groups[key]]
|
||||
|
||||
for img in train_images:
|
||||
shutil.copy(img, class_train_dir / img.name)
|
||||
total_train += 1
|
||||
|
||||
for img in val_images:
|
||||
shutil.copy(img, class_val_dir / img.name)
|
||||
total_val += 1
|
||||
|
||||
print(
|
||||
f" Class '{class_name}': {len(train_images)} train, "
|
||||
f"{len(val_images)} val (from {num_groups} source photos, {num_images} files total)"
|
||||
)
|
||||
|
||||
print(f"Dataset split completed: {total_train} train images, {total_val} validation images.")
|
||||
print(f"Split dataset located at: {dest_dir.absolute()}")
|
||||
|
||||
|
||||
def train_model(args):
|
||||
"""Handles training the YOLO classification model."""
|
||||
src_path = Path(args.src_dir).resolve()
|
||||
dest_path = Path(args.split_dir).resolve()
|
||||
|
||||
if not src_path.is_dir():
|
||||
print(f"Error: Source dataset directory not found: {src_path}")
|
||||
sys.exit(1)
|
||||
|
||||
print(f"--- Preparing Dataset from {src_path} ---")
|
||||
split_dataset(src_path, dest_path, split_ratio=args.split_ratio)
|
||||
|
||||
model_path = Path(args.model).resolve()
|
||||
print(f"\n--- Initializing YOLO Model ({model_path}) ---")
|
||||
model = YOLO(str(model_path))
|
||||
|
||||
if args.device:
|
||||
device = args.device
|
||||
else:
|
||||
device = "0" if torch.cuda.is_available() else "cpu"
|
||||
print(f"Using device: {device}")
|
||||
|
||||
print("\n--- Starting Training ---")
|
||||
results = model.train(
|
||||
data=str(dest_path),
|
||||
epochs=args.epochs,
|
||||
imgsz=args.imgsz,
|
||||
batch=args.batch,
|
||||
device=device,
|
||||
project=str(Path(args.project).resolve()),
|
||||
name=args.name,
|
||||
exist_ok=True,
|
||||
workers=args.workers,
|
||||
lr0=args.lr,
|
||||
optimizer=args.optimizer,
|
||||
seed=42,
|
||||
)
|
||||
|
||||
best_weights = Path(results.save_dir) / "weights" / "best.pt"
|
||||
output_path = Path(args.output).resolve() if args.output else classifier_output_path(args.epochs)
|
||||
output_path.parent.mkdir(parents=True, exist_ok=True)
|
||||
shutil.copy2(best_weights, output_path)
|
||||
|
||||
print("\nTraining completed successfully!")
|
||||
print(f"Run weights saved at: {best_weights}")
|
||||
print(f"Published model saved at: {output_path}")
|
||||
|
||||
if args.export:
|
||||
print("\n--- Exporting model to ONNX format ---")
|
||||
try:
|
||||
export_model = YOLO(str(output_path))
|
||||
onnx_path = Path(export_model.export(format="onnx"))
|
||||
dated_onnx = output_path.with_suffix(".onnx")
|
||||
if onnx_path.resolve() != dated_onnx.resolve():
|
||||
shutil.copy2(onnx_path, dated_onnx)
|
||||
print(f"Model exported successfully to: {dated_onnx}")
|
||||
except Exception as e:
|
||||
print(f"Warning: ONNX export failed: {e}")
|
||||
|
||||
print("\nYou can run predictions with:")
|
||||
print(f" uv run python {Path(__file__).name} predict --image <image_path> --model {output_path}")
|
||||
|
||||
|
||||
def predict_image(args):
|
||||
"""Runs classification inference on a single image."""
|
||||
model_path = Path(args.model).resolve()
|
||||
image_path = Path(args.image).resolve()
|
||||
|
||||
if not model_path.exists():
|
||||
print(f"Error: Model weights not found at {model_path}")
|
||||
sys.exit(1)
|
||||
|
||||
if not image_path.exists():
|
||||
print(f"Error: Target image file not found at {image_path}")
|
||||
sys.exit(1)
|
||||
|
||||
print(f"Loading model from {model_path}...")
|
||||
model = YOLO(str(model_path))
|
||||
|
||||
print(f"Running prediction on {image_path}...")
|
||||
results = model(str(image_path))
|
||||
|
||||
for result in results:
|
||||
probs = result.probs
|
||||
top1_idx = probs.top1
|
||||
top1_conf = float(probs.top1conf)
|
||||
top1_name = result.names[top1_idx]
|
||||
|
||||
print("\n=== Classification Results ===")
|
||||
print(f"Top-1 Prediction: {top1_name} (Confidence: {top1_conf:.4f})")
|
||||
print("\nAll Probabilities:")
|
||||
|
||||
sorted_probs = sorted(
|
||||
[(result.names[i], float(val)) for i, val in enumerate(probs.data)],
|
||||
key=lambda x: x[1],
|
||||
reverse=True,
|
||||
)
|
||||
for name, score in sorted_probs:
|
||||
print(f" {name}: {score:.4f}")
|
||||
|
||||
|
||||
def main():
|
||||
parser = argparse.ArgumentParser(
|
||||
description="Ultralytics YOLO classification utility for produk-pfm packaging photos."
|
||||
)
|
||||
subparsers = parser.add_subparsers(dest="command", required=True, help="Command to run")
|
||||
|
||||
train_parser = subparsers.add_parser("train", help="Train a classification model")
|
||||
train_parser.add_argument(
|
||||
"--src-dir",
|
||||
type=str,
|
||||
default=str(DEFAULT_DATASET_DIR),
|
||||
help=f"Source dataset directory with one class folder per product (default: {DEFAULT_DATASET_DIR.name})",
|
||||
)
|
||||
train_parser.add_argument(
|
||||
"--split-dir",
|
||||
type=str,
|
||||
default=str(DEFAULT_SPLIT_DIR),
|
||||
help="Output split dataset directory",
|
||||
)
|
||||
train_parser.add_argument(
|
||||
"--split-ratio",
|
||||
type=float,
|
||||
default=0.8,
|
||||
help="Train/val split ratio for classes with 3+ images (default: 0.8)",
|
||||
)
|
||||
train_parser.add_argument(
|
||||
"--model",
|
||||
type=str,
|
||||
default=str(DEFAULT_MODEL),
|
||||
help="Pretrained model (e.g. yolo26n-cls.pt, yolo11n-cls.pt, yolov8n-cls.pt)",
|
||||
)
|
||||
train_parser.add_argument(
|
||||
"--epochs",
|
||||
type=int,
|
||||
default=DEFAULT_EPOCHS,
|
||||
help=f"Number of training epochs (default: {DEFAULT_EPOCHS})",
|
||||
)
|
||||
train_parser.add_argument(
|
||||
"--output",
|
||||
type=str,
|
||||
default=None,
|
||||
help=(
|
||||
"Published .pt output path (default: "
|
||||
"models/produk-pfm-classifier-26n-{epochs}e-{YYYY-MM-DD}.pt)"
|
||||
),
|
||||
)
|
||||
train_parser.add_argument("--imgsz", type=int, default=224, help="Target image size for classification")
|
||||
train_parser.add_argument("--batch", type=int, default=8, help="Batch size for training")
|
||||
train_parser.add_argument(
|
||||
"--device",
|
||||
type=str,
|
||||
default=None,
|
||||
help="Device to run on (e.g. 0 or 'cpu'). Default is GPU if available.",
|
||||
)
|
||||
train_parser.add_argument(
|
||||
"--project",
|
||||
type=str,
|
||||
default=str(DEFAULT_PROJECT),
|
||||
help="Project output folder name",
|
||||
)
|
||||
train_parser.add_argument("--name", type=str, default="train", help="Experiment name")
|
||||
train_parser.add_argument("--workers", type=int, default=4, help="Number of data loading workers")
|
||||
train_parser.add_argument("--lr", type=float, default=0.01, help="Initial learning rate")
|
||||
train_parser.add_argument(
|
||||
"--optimizer",
|
||||
type=str,
|
||||
default="auto",
|
||||
choices=["SGD", "Adam", "AdamW", "RMSProp", "auto"],
|
||||
help="Optimizer to use",
|
||||
)
|
||||
train_parser.add_argument(
|
||||
"--export",
|
||||
action="store_true",
|
||||
default=True,
|
||||
help="Export model to ONNX after training",
|
||||
)
|
||||
|
||||
predict_parser = subparsers.add_parser("predict", help="Predict class of an image")
|
||||
predict_parser.add_argument("--image", type=str, required=True, help="Path to image file")
|
||||
predict_parser.add_argument(
|
||||
"--model",
|
||||
type=str,
|
||||
default=str(latest_classifier_weights()),
|
||||
help="Path to trained YOLO .pt model weights (default: newest models/produk-pfm-classifier-*.pt)",
|
||||
)
|
||||
|
||||
args = parser.parse_args()
|
||||
|
||||
if args.command == "train":
|
||||
train_model(args)
|
||||
elif args.command == "predict":
|
||||
predict_image(args)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -1,44 +1,44 @@
|
||||
const { Client } = require('pg');
|
||||
|
||||
async function main() {
|
||||
const client = new Client({
|
||||
host: process.env.PGHOST || "paddleocr-db",
|
||||
port: parseInt(process.env.PGPORT || "5432"),
|
||||
user: process.env.PGUSER || "postgres",
|
||||
password: process.env.PGPASSWORD || "postgres",
|
||||
database: process.env.PGDATABASE || "dopfm",
|
||||
});
|
||||
|
||||
await client.connect();
|
||||
console.log('Connected to PG database.');
|
||||
|
||||
const res = await client.query("SELECT id, filename FROM documents WHERE parsed = false;");
|
||||
console.log(`Found ${res.rows.length} documents to parse.`);
|
||||
|
||||
for (let i = 0; i < res.rows.length; i++) {
|
||||
const row = res.rows[i];
|
||||
console.log(`[${i+1}/${res.rows.length}] Reparsing ${row.filename} (ID: ${row.id})...`);
|
||||
try {
|
||||
const response = await fetch('http://localhost:3000/api/parse', {
|
||||
method: 'POST',
|
||||
headers: { 'Content-Type': 'application/json' },
|
||||
body: JSON.stringify({ filename: row.filename })
|
||||
});
|
||||
if (response.ok) {
|
||||
console.log(`Successfully triggered parse for ${row.filename}. Status: ${response.status}`);
|
||||
} else {
|
||||
console.error(`Failed to parse ${row.filename}. Status: ${response.status}, Error: ${await response.text()}`);
|
||||
}
|
||||
} catch (err) {
|
||||
console.error(`Fetch error for ${row.filename}:`, err.message);
|
||||
}
|
||||
}
|
||||
|
||||
await client.end();
|
||||
console.log('Done reparsing.');
|
||||
}
|
||||
|
||||
main().catch(err => {
|
||||
console.error('Fatal error:', err);
|
||||
process.exit(1);
|
||||
});
|
||||
const { Client } = require('pg');
|
||||
|
||||
async function main() {
|
||||
const client = new Client({
|
||||
host: process.env.PGHOST || "paddleocr-db",
|
||||
port: parseInt(process.env.PGPORT || "5432"),
|
||||
user: process.env.PGUSER || "postgres",
|
||||
password: process.env.PGPASSWORD || "postgres",
|
||||
database: process.env.PGDATABASE || "dopfm",
|
||||
});
|
||||
|
||||
await client.connect();
|
||||
console.log('Connected to PG database.');
|
||||
|
||||
const res = await client.query("SELECT id, filename FROM documents WHERE parsed = false;");
|
||||
console.log(`Found ${res.rows.length} documents to parse.`);
|
||||
|
||||
for (let i = 0; i < res.rows.length; i++) {
|
||||
const row = res.rows[i];
|
||||
console.log(`[${i+1}/${res.rows.length}] Reparsing ${row.filename} (ID: ${row.id})...`);
|
||||
try {
|
||||
const response = await fetch('http://localhost:3000/api/parse', {
|
||||
method: 'POST',
|
||||
headers: { 'Content-Type': 'application/json' },
|
||||
body: JSON.stringify({ filename: row.filename })
|
||||
});
|
||||
if (response.ok) {
|
||||
console.log(`Successfully triggered parse for ${row.filename}. Status: ${response.status}`);
|
||||
} else {
|
||||
console.error(`Failed to parse ${row.filename}. Status: ${response.status}, Error: ${await response.text()}`);
|
||||
}
|
||||
} catch (err) {
|
||||
console.error(`Fetch error for ${row.filename}:`, err.message);
|
||||
}
|
||||
}
|
||||
|
||||
await client.end();
|
||||
console.log('Done reparsing.');
|
||||
}
|
||||
|
||||
main().catch(err => {
|
||||
console.error('Fatal error:', err);
|
||||
process.exit(1);
|
||||
});
|
||||
@@ -1,244 +1,244 @@
|
||||
const fs = require('fs');
|
||||
const path = require('path');
|
||||
const http = require('http');
|
||||
const { Client } = require('pg');
|
||||
|
||||
const BASE_URL = 'http://localhost:3000/api/parse';
|
||||
|
||||
const testFiles = [
|
||||
"do-001.jpg",
|
||||
"do-002.jpg",
|
||||
"do-003.jpg",
|
||||
"do-004.jpg",
|
||||
"do-005.jpg",
|
||||
"do-006.jpg",
|
||||
"do-007.jpg",
|
||||
"do-008.jpg",
|
||||
"do-009.jpg",
|
||||
"do-010.jpg",
|
||||
"do-011.jpg",
|
||||
"do-012.jpg",
|
||||
"do-013.jpg",
|
||||
"do-014.jpg"
|
||||
];
|
||||
|
||||
function postJSON(url, body) {
|
||||
return new Promise((resolve, reject) => {
|
||||
const parsedUrl = new URL(url);
|
||||
const bodyStr = JSON.stringify(body);
|
||||
|
||||
const options = {
|
||||
hostname: parsedUrl.hostname,
|
||||
port: parsedUrl.port,
|
||||
path: parsedUrl.pathname + parsedUrl.search,
|
||||
method: 'POST',
|
||||
headers: {
|
||||
'Content-Type': 'application/json',
|
||||
'Content-Length': Buffer.byteLength(bodyStr)
|
||||
},
|
||||
timeout: 1200000 // 20 minutes
|
||||
};
|
||||
|
||||
const req = http.request(options, (res) => {
|
||||
let data = '';
|
||||
res.on('data', (chunk) => { data += chunk; });
|
||||
res.on('end', () => {
|
||||
resolve({
|
||||
ok: res.statusCode >= 200 && res.statusCode < 300,
|
||||
status: res.statusCode,
|
||||
json: async () => JSON.parse(data),
|
||||
text: async () => data
|
||||
});
|
||||
});
|
||||
});
|
||||
|
||||
req.on('timeout', () => {
|
||||
req.destroy(new Error('Request Timeout (20m)'));
|
||||
});
|
||||
|
||||
req.on('error', (err) => { reject(err); });
|
||||
req.write(bodyStr);
|
||||
req.end();
|
||||
});
|
||||
}
|
||||
|
||||
async function getDocumentMetadataFromDb(filename) {
|
||||
const client = new Client({
|
||||
host: 'paddleocr-db',
|
||||
port: 5432,
|
||||
user: 'postgres',
|
||||
password: 'postgres',
|
||||
database: 'dopfm'
|
||||
});
|
||||
|
||||
try {
|
||||
await client.connect();
|
||||
const res = await client.query('SELECT metadata FROM documents WHERE filename = $1', [filename]);
|
||||
return res.rows[0]?.metadata || {};
|
||||
} catch (err) {
|
||||
console.error('Database query failed:', err.message);
|
||||
return {};
|
||||
} finally {
|
||||
await client.end();
|
||||
}
|
||||
}
|
||||
|
||||
async function main() {
|
||||
console.log(`Starting single image test for ${testFiles.length} file...`);
|
||||
|
||||
const summaryTmpFile = '/uploads/test_images_report_summary.tmp';
|
||||
const detailsTmpFile = '/uploads/test_images_report_details.tmp';
|
||||
const jsonlFile = '/uploads/test_images_results.jsonl';
|
||||
const finalReportFile = '/uploads/test_images_report.md';
|
||||
|
||||
// Initialize summary header
|
||||
let summaryHeader = `# Batch OCR Parsing Test Report\n\n`;
|
||||
summaryHeader += `Processed **${testFiles.length}** file from \`backend/sources/test-images\`.\n\n`;
|
||||
summaryHeader += `## Summary Table\n\n`;
|
||||
summaryHeader += `| No | Filename | Status | Tilt | Auto-Rotated | PO | SO | DO | Date | Store Match | Items Count |\n`;
|
||||
summaryHeader += `|---|---|---|---|---|---|---|---|---|---|---|\n`;
|
||||
fs.writeFileSync(summaryTmpFile, summaryHeader);
|
||||
|
||||
// Initialize details header
|
||||
let detailsHeader = `\n\n## Detailed Results per Image\n\n`;
|
||||
fs.writeFileSync(detailsTmpFile, detailsHeader);
|
||||
|
||||
// Clean jsonl
|
||||
fs.writeFileSync(jsonlFile, '');
|
||||
|
||||
for (let idx = 0; idx < testFiles.length; idx++) {
|
||||
const file = testFiles[idx];
|
||||
console.log(`[${idx + 1}/${testFiles.length}] Processing file: ${file}`);
|
||||
|
||||
try {
|
||||
const response = await postJSON(BASE_URL, { filename: file });
|
||||
|
||||
if (!response.ok) {
|
||||
const errorText = await response.text();
|
||||
console.error(`Error parsing file ${file}: ${errorText}`);
|
||||
|
||||
// Write fail state incrementally
|
||||
const tableLine = `| ${idx + 1} | \`${file}\` | **Failed** | N/A | N/A | N/A | N/A | N/A | N/A | N/A | N/A |\n`;
|
||||
fs.appendFileSync(summaryTmpFile, tableLine);
|
||||
|
||||
let detailedText = `### ${idx + 1}. \`${file}\`\n`;
|
||||
detailedText += `- **Status**: Failed\n`;
|
||||
detailedText += `- **Error Detail**: \`${errorText || 'Unknown error'}\`\n`;
|
||||
detailedText += `\n---\n\n`;
|
||||
fs.appendFileSync(detailsTmpFile, detailedText);
|
||||
|
||||
fs.appendFileSync(jsonlFile, JSON.stringify({
|
||||
filename: file,
|
||||
status: 'Failed',
|
||||
error: errorText || 'Unknown error'
|
||||
}) + '\n');
|
||||
|
||||
continue;
|
||||
}
|
||||
|
||||
const resData = await response.json();
|
||||
const pipelineRes = resData.result || {};
|
||||
const page0 = pipelineRes.layoutParsingResults?.[0] || {};
|
||||
const rawMarkdown = page0.markdown?.text || "N/A";
|
||||
const info = pipelineRes.pipeline_info || {};
|
||||
|
||||
// Direct DB query for accurate metadata (bypassing Auth)
|
||||
const docMeta = await getDocumentMetadataFromDb(file);
|
||||
|
||||
const tiltStr = info.tilt !== undefined ? parseFloat(info.tilt).toFixed(2) : 'N/A';
|
||||
const unwarpedStr = info.unwarped ? 'Yes' : 'No';
|
||||
const itemsCount = (resData.items || []).length;
|
||||
|
||||
// Write success state incrementally
|
||||
const tableLine = `| ${idx + 1} | \`${file}\` | **Success** | ${tiltStr}° | ${unwarpedStr} | \`${docMeta.noPO || 'N/A'}\` | \`${docMeta.noSO || 'N/A'}\` | \`${docMeta.noDO || 'N/A'}\` | \`${docMeta.tanggal || 'N/A'}\` | ${docMeta.orderUntuk || 'N/A'} | ${itemsCount} |\n`;
|
||||
fs.appendFileSync(summaryTmpFile, tableLine);
|
||||
|
||||
let detailedText = `### ${idx + 1}. \`${file}\`\n`;
|
||||
detailedText += `- **Status**: Success\n`;
|
||||
detailedText += `- **Tilt Detected**: ${tiltStr}°\n`;
|
||||
detailedText += `- **Auto-Rotated/Unwarped**: ${unwarpedStr}\n`;
|
||||
detailedText += `- **Extracted Metadata**:\n`;
|
||||
detailedText += ` * **PO**: \`${docMeta.noPO || 'N/A'}\`\n`;
|
||||
detailedText += ` * **SO**: \`${docMeta.noSO || 'N/A'}\`\n`;
|
||||
detailedText += ` * **DO**: \`${docMeta.noDO || 'N/A'}\`\n`;
|
||||
detailedText += ` * **Tanggal**: \`${docMeta.tanggal || 'N/A'}\`\n`;
|
||||
detailedText += ` * **Customer**: \`${docMeta.customerInfo || 'N/A'}\`\n`;
|
||||
detailedText += ` * **Store**: \`${docMeta.orderUntuk || 'N/A'}\`\n`;
|
||||
detailedText += ` * **Alamat**: \`${docMeta.alamat || 'N/A'}\`\n`;
|
||||
detailedText += ` * **Plat Nomor**: \`${docMeta.platTruk || 'N/A'}\`\n`;
|
||||
detailedText += `- **Raw Layout Markdown**:\n`;
|
||||
detailedText += `\`\`\`markdown\n${rawMarkdown}\n\`\`\`\n`;
|
||||
detailedText += `- **Parsed Items (${itemsCount})**:\n`;
|
||||
|
||||
if (itemsCount > 0) {
|
||||
detailedText += ` | Code (SKU) | Name | Qty | Price |\n`;
|
||||
detailedText += ` |---|---|---|---|\n`;
|
||||
(resData.items || []).forEach(item => {
|
||||
detailedText += ` | \`${item.kodeBarang}\` | ${item.namaBarang} | \`${item.banyak}\` | \`${item.jumlah}\` |\n`;
|
||||
});
|
||||
} else {
|
||||
detailedText += ` *No valid SKU items parsed.*\n`;
|
||||
}
|
||||
detailedText += `\n---\n\n`;
|
||||
fs.appendFileSync(detailsTmpFile, detailedText);
|
||||
|
||||
fs.appendFileSync(jsonlFile, JSON.stringify({
|
||||
filename: file,
|
||||
status: 'Success',
|
||||
tilt: tiltStr,
|
||||
unwarped: unwarpedStr,
|
||||
rawMarkdown: resData.postProcessingDetails?.rawMarkdown || "",
|
||||
layer1RawRegex: resData.postProcessingDetails?.layer1RawRegex || {},
|
||||
layer2Sanitized: resData.postProcessingDetails?.layer2Sanitized || {},
|
||||
layer3Final: resData.postProcessingDetails?.layer3Final || {},
|
||||
metadata: docMeta,
|
||||
items: resData.items || []
|
||||
}) + '\n');
|
||||
|
||||
} catch (err) {
|
||||
console.error(`Exception during file ${file}:`, err);
|
||||
|
||||
const tableLine = `| ${idx + 1} | \`${file}\` | **Error** | N/A | N/A | N/A | N/A | N/A | N/A | N/A | N/A |\n`;
|
||||
fs.appendFileSync(summaryTmpFile, tableLine);
|
||||
|
||||
let detailedText = `### ${idx + 1}. \`${file}\`\n`;
|
||||
detailedText += `- **Status**: Error\n`;
|
||||
detailedText += `- **Error Detail**: \`${err.message}\`\n`;
|
||||
detailedText += `\n---\n\n`;
|
||||
fs.appendFileSync(detailsTmpFile, detailedText);
|
||||
|
||||
fs.appendFileSync(jsonlFile, JSON.stringify({
|
||||
filename: file,
|
||||
status: 'Error',
|
||||
error: err.message
|
||||
}) + '\n');
|
||||
}
|
||||
}
|
||||
|
||||
// Combine temporary files into the final report
|
||||
try {
|
||||
const summaryContent = fs.readFileSync(summaryTmpFile, 'utf8');
|
||||
const detailsContent = fs.readFileSync(detailsTmpFile, 'utf8');
|
||||
fs.writeFileSync(finalReportFile, summaryContent + '\n' + detailsContent);
|
||||
|
||||
// Clean up temporary files
|
||||
fs.unlinkSync(summaryTmpFile);
|
||||
fs.unlinkSync(detailsTmpFile);
|
||||
} catch (combineErr) {
|
||||
console.error('Failed to combine test reports:', combineErr);
|
||||
}
|
||||
|
||||
// Compile JSONL into the final JSON v2
|
||||
try {
|
||||
const lines = fs.readFileSync(jsonlFile, 'utf8').split('\n').filter(Boolean);
|
||||
const results = lines.map(line => JSON.parse(line));
|
||||
fs.writeFileSync('/uploads/ai_results_v2.json', JSON.stringify(results, null, 2));
|
||||
console.log('Compiled results saved to /uploads/ai_results_v2.json');
|
||||
} catch (compileErr) {
|
||||
console.error('Failed to compile results into JSON v2:', compileErr);
|
||||
}
|
||||
|
||||
console.log('Batch test completed. Report written to /uploads/test_images_report.md');
|
||||
}
|
||||
|
||||
main();
|
||||
const fs = require('fs');
|
||||
const path = require('path');
|
||||
const http = require('http');
|
||||
const { Client } = require('pg');
|
||||
|
||||
const BASE_URL = 'http://localhost:3000/api/parse';
|
||||
|
||||
const testFiles = [
|
||||
"do-001.jpg",
|
||||
"do-002.jpg",
|
||||
"do-003.jpg",
|
||||
"do-004.jpg",
|
||||
"do-005.jpg",
|
||||
"do-006.jpg",
|
||||
"do-007.jpg",
|
||||
"do-008.jpg",
|
||||
"do-009.jpg",
|
||||
"do-010.jpg",
|
||||
"do-011.jpg",
|
||||
"do-012.jpg",
|
||||
"do-013.jpg",
|
||||
"do-014.jpg"
|
||||
];
|
||||
|
||||
function postJSON(url, body) {
|
||||
return new Promise((resolve, reject) => {
|
||||
const parsedUrl = new URL(url);
|
||||
const bodyStr = JSON.stringify(body);
|
||||
|
||||
const options = {
|
||||
hostname: parsedUrl.hostname,
|
||||
port: parsedUrl.port,
|
||||
path: parsedUrl.pathname + parsedUrl.search,
|
||||
method: 'POST',
|
||||
headers: {
|
||||
'Content-Type': 'application/json',
|
||||
'Content-Length': Buffer.byteLength(bodyStr)
|
||||
},
|
||||
timeout: 1200000 // 20 minutes
|
||||
};
|
||||
|
||||
const req = http.request(options, (res) => {
|
||||
let data = '';
|
||||
res.on('data', (chunk) => { data += chunk; });
|
||||
res.on('end', () => {
|
||||
resolve({
|
||||
ok: res.statusCode >= 200 && res.statusCode < 300,
|
||||
status: res.statusCode,
|
||||
json: async () => JSON.parse(data),
|
||||
text: async () => data
|
||||
});
|
||||
});
|
||||
});
|
||||
|
||||
req.on('timeout', () => {
|
||||
req.destroy(new Error('Request Timeout (20m)'));
|
||||
});
|
||||
|
||||
req.on('error', (err) => { reject(err); });
|
||||
req.write(bodyStr);
|
||||
req.end();
|
||||
});
|
||||
}
|
||||
|
||||
async function getDocumentMetadataFromDb(filename) {
|
||||
const client = new Client({
|
||||
host: 'paddleocr-db',
|
||||
port: 5432,
|
||||
user: 'postgres',
|
||||
password: 'postgres',
|
||||
database: 'dopfm'
|
||||
});
|
||||
|
||||
try {
|
||||
await client.connect();
|
||||
const res = await client.query('SELECT metadata FROM documents WHERE filename = $1', [filename]);
|
||||
return res.rows[0]?.metadata || {};
|
||||
} catch (err) {
|
||||
console.error('Database query failed:', err.message);
|
||||
return {};
|
||||
} finally {
|
||||
await client.end();
|
||||
}
|
||||
}
|
||||
|
||||
async function main() {
|
||||
console.log(`Starting single image test for ${testFiles.length} file...`);
|
||||
|
||||
const summaryTmpFile = '/uploads/test_images_report_summary.tmp';
|
||||
const detailsTmpFile = '/uploads/test_images_report_details.tmp';
|
||||
const jsonlFile = '/uploads/test_images_results.jsonl';
|
||||
const finalReportFile = '/uploads/test_images_report.md';
|
||||
|
||||
// Initialize summary header
|
||||
let summaryHeader = `# Batch OCR Parsing Test Report\n\n`;
|
||||
summaryHeader += `Processed **${testFiles.length}** file from \`backend/sources/test-images\`.\n\n`;
|
||||
summaryHeader += `## Summary Table\n\n`;
|
||||
summaryHeader += `| No | Filename | Status | Tilt | Auto-Rotated | PO | SO | DO | Date | Store Match | Items Count |\n`;
|
||||
summaryHeader += `|---|---|---|---|---|---|---|---|---|---|---|\n`;
|
||||
fs.writeFileSync(summaryTmpFile, summaryHeader);
|
||||
|
||||
// Initialize details header
|
||||
let detailsHeader = `\n\n## Detailed Results per Image\n\n`;
|
||||
fs.writeFileSync(detailsTmpFile, detailsHeader);
|
||||
|
||||
// Clean jsonl
|
||||
fs.writeFileSync(jsonlFile, '');
|
||||
|
||||
for (let idx = 0; idx < testFiles.length; idx++) {
|
||||
const file = testFiles[idx];
|
||||
console.log(`[${idx + 1}/${testFiles.length}] Processing file: ${file}`);
|
||||
|
||||
try {
|
||||
const response = await postJSON(BASE_URL, { filename: file });
|
||||
|
||||
if (!response.ok) {
|
||||
const errorText = await response.text();
|
||||
console.error(`Error parsing file ${file}: ${errorText}`);
|
||||
|
||||
// Write fail state incrementally
|
||||
const tableLine = `| ${idx + 1} | \`${file}\` | **Failed** | N/A | N/A | N/A | N/A | N/A | N/A | N/A | N/A |\n`;
|
||||
fs.appendFileSync(summaryTmpFile, tableLine);
|
||||
|
||||
let detailedText = `### ${idx + 1}. \`${file}\`\n`;
|
||||
detailedText += `- **Status**: Failed\n`;
|
||||
detailedText += `- **Error Detail**: \`${errorText || 'Unknown error'}\`\n`;
|
||||
detailedText += `\n---\n\n`;
|
||||
fs.appendFileSync(detailsTmpFile, detailedText);
|
||||
|
||||
fs.appendFileSync(jsonlFile, JSON.stringify({
|
||||
filename: file,
|
||||
status: 'Failed',
|
||||
error: errorText || 'Unknown error'
|
||||
}) + '\n');
|
||||
|
||||
continue;
|
||||
}
|
||||
|
||||
const resData = await response.json();
|
||||
const pipelineRes = resData.result || {};
|
||||
const page0 = pipelineRes.layoutParsingResults?.[0] || {};
|
||||
const rawMarkdown = page0.markdown?.text || "N/A";
|
||||
const info = pipelineRes.pipeline_info || {};
|
||||
|
||||
// Direct DB query for accurate metadata (bypassing Auth)
|
||||
const docMeta = await getDocumentMetadataFromDb(file);
|
||||
|
||||
const tiltStr = info.tilt !== undefined ? parseFloat(info.tilt).toFixed(2) : 'N/A';
|
||||
const unwarpedStr = info.unwarped ? 'Yes' : 'No';
|
||||
const itemsCount = (resData.items || []).length;
|
||||
|
||||
// Write success state incrementally
|
||||
const tableLine = `| ${idx + 1} | \`${file}\` | **Success** | ${tiltStr}° | ${unwarpedStr} | \`${docMeta.noPO || 'N/A'}\` | \`${docMeta.noSO || 'N/A'}\` | \`${docMeta.noDO || 'N/A'}\` | \`${docMeta.tanggal || 'N/A'}\` | ${docMeta.orderUntuk || 'N/A'} | ${itemsCount} |\n`;
|
||||
fs.appendFileSync(summaryTmpFile, tableLine);
|
||||
|
||||
let detailedText = `### ${idx + 1}. \`${file}\`\n`;
|
||||
detailedText += `- **Status**: Success\n`;
|
||||
detailedText += `- **Tilt Detected**: ${tiltStr}°\n`;
|
||||
detailedText += `- **Auto-Rotated/Unwarped**: ${unwarpedStr}\n`;
|
||||
detailedText += `- **Extracted Metadata**:\n`;
|
||||
detailedText += ` * **PO**: \`${docMeta.noPO || 'N/A'}\`\n`;
|
||||
detailedText += ` * **SO**: \`${docMeta.noSO || 'N/A'}\`\n`;
|
||||
detailedText += ` * **DO**: \`${docMeta.noDO || 'N/A'}\`\n`;
|
||||
detailedText += ` * **Tanggal**: \`${docMeta.tanggal || 'N/A'}\`\n`;
|
||||
detailedText += ` * **Customer**: \`${docMeta.customerInfo || 'N/A'}\`\n`;
|
||||
detailedText += ` * **Store**: \`${docMeta.orderUntuk || 'N/A'}\`\n`;
|
||||
detailedText += ` * **Alamat**: \`${docMeta.alamat || 'N/A'}\`\n`;
|
||||
detailedText += ` * **Plat Nomor**: \`${docMeta.platTruk || 'N/A'}\`\n`;
|
||||
detailedText += `- **Raw Layout Markdown**:\n`;
|
||||
detailedText += `\`\`\`markdown\n${rawMarkdown}\n\`\`\`\n`;
|
||||
detailedText += `- **Parsed Items (${itemsCount})**:\n`;
|
||||
|
||||
if (itemsCount > 0) {
|
||||
detailedText += ` | Code (SKU) | Name | Qty | Price |\n`;
|
||||
detailedText += ` |---|---|---|---|\n`;
|
||||
(resData.items || []).forEach(item => {
|
||||
detailedText += ` | \`${item.kodeBarang}\` | ${item.namaBarang} | \`${item.banyak}\` | \`${item.jumlah}\` |\n`;
|
||||
});
|
||||
} else {
|
||||
detailedText += ` *No valid SKU items parsed.*\n`;
|
||||
}
|
||||
detailedText += `\n---\n\n`;
|
||||
fs.appendFileSync(detailsTmpFile, detailedText);
|
||||
|
||||
fs.appendFileSync(jsonlFile, JSON.stringify({
|
||||
filename: file,
|
||||
status: 'Success',
|
||||
tilt: tiltStr,
|
||||
unwarped: unwarpedStr,
|
||||
rawMarkdown: resData.postProcessingDetails?.rawMarkdown || "",
|
||||
layer1RawRegex: resData.postProcessingDetails?.layer1RawRegex || {},
|
||||
layer2Sanitized: resData.postProcessingDetails?.layer2Sanitized || {},
|
||||
layer3Final: resData.postProcessingDetails?.layer3Final || {},
|
||||
metadata: docMeta,
|
||||
items: resData.items || []
|
||||
}) + '\n');
|
||||
|
||||
} catch (err) {
|
||||
console.error(`Exception during file ${file}:`, err);
|
||||
|
||||
const tableLine = `| ${idx + 1} | \`${file}\` | **Error** | N/A | N/A | N/A | N/A | N/A | N/A | N/A | N/A |\n`;
|
||||
fs.appendFileSync(summaryTmpFile, tableLine);
|
||||
|
||||
let detailedText = `### ${idx + 1}. \`${file}\`\n`;
|
||||
detailedText += `- **Status**: Error\n`;
|
||||
detailedText += `- **Error Detail**: \`${err.message}\`\n`;
|
||||
detailedText += `\n---\n\n`;
|
||||
fs.appendFileSync(detailsTmpFile, detailedText);
|
||||
|
||||
fs.appendFileSync(jsonlFile, JSON.stringify({
|
||||
filename: file,
|
||||
status: 'Error',
|
||||
error: err.message
|
||||
}) + '\n');
|
||||
}
|
||||
}
|
||||
|
||||
// Combine temporary files into the final report
|
||||
try {
|
||||
const summaryContent = fs.readFileSync(summaryTmpFile, 'utf8');
|
||||
const detailsContent = fs.readFileSync(detailsTmpFile, 'utf8');
|
||||
fs.writeFileSync(finalReportFile, summaryContent + '\n' + detailsContent);
|
||||
|
||||
// Clean up temporary files
|
||||
fs.unlinkSync(summaryTmpFile);
|
||||
fs.unlinkSync(detailsTmpFile);
|
||||
} catch (combineErr) {
|
||||
console.error('Failed to combine test reports:', combineErr);
|
||||
}
|
||||
|
||||
// Compile JSONL into the final JSON v2
|
||||
try {
|
||||
const lines = fs.readFileSync(jsonlFile, 'utf8').split('\n').filter(Boolean);
|
||||
const results = lines.map(line => JSON.parse(line));
|
||||
fs.writeFileSync('/uploads/ai_results_v2.json', JSON.stringify(results, null, 2));
|
||||
console.log('Compiled results saved to /uploads/ai_results_v2.json');
|
||||
} catch (compileErr) {
|
||||
console.error('Failed to compile results into JSON v2:', compileErr);
|
||||
}
|
||||
|
||||
console.log('Batch test completed. Report written to /uploads/test_images_report.md');
|
||||
}
|
||||
|
||||
main();
|
||||
@@ -1,403 +1,403 @@
|
||||
/**
|
||||
* run_full_test.js
|
||||
*
|
||||
* Runs OCR parsing against ALL images in backend/sources/test-images/
|
||||
* and captures every pipeline stage for analysis:
|
||||
* - rawMarkdown : raw text from PaddleOCR layout parser
|
||||
* - layer1RawRegex: output of parseDOMetadata (regex extraction)
|
||||
* - layer2Sanitized: output of sanitizeParsedMetadata (format checks)
|
||||
* - layer3Final : final metadata after SKU triple-check + store resolution
|
||||
*
|
||||
* Outputs:
|
||||
* backend/sources/ai_results.json — machine-readable per-file results
|
||||
* backend/sources/ai_results.md — human-readable stage-by-stage breakdown
|
||||
*
|
||||
* Usage (from host machine, Docker must be running):
|
||||
* node run_full_test.js
|
||||
*
|
||||
* The script talks to the nginx gateway on port 8000.
|
||||
* To override: set env var BASE_URL=http://localhost:3000/api/parse
|
||||
*/
|
||||
|
||||
const fs = require('fs');
|
||||
const path = require('path');
|
||||
const http = require('http');
|
||||
const https = require('https');
|
||||
|
||||
// ─── Config ──────────────────────────────────────────────────────────────────
|
||||
|
||||
const BASE_URL = process.env.BASE_URL || 'http://localhost:8000/api/parse';
|
||||
const TEST_IMAGES_DIR = path.resolve(__dirname, '../sources/test-images');
|
||||
const OUTPUT_JSON = path.resolve(__dirname, '../sources/ai_results.json');
|
||||
const OUTPUT_MD = path.resolve(__dirname, '../sources/ai_results.md');
|
||||
const REQUEST_TIMEOUT_MS = 20 * 60 * 1000; // 20 minutes per image
|
||||
|
||||
// ─── HTTP Helper ─────────────────────────────────────────────────────────────
|
||||
|
||||
function postJSON(url, body) {
|
||||
return new Promise((resolve, reject) => {
|
||||
const parsedUrl = new URL(url);
|
||||
const bodyStr = JSON.stringify(body);
|
||||
const lib = parsedUrl.protocol === 'https:' ? https : http;
|
||||
|
||||
const options = {
|
||||
hostname: parsedUrl.hostname,
|
||||
port: parsedUrl.port || (parsedUrl.protocol === 'https:' ? 443 : 80),
|
||||
path: parsedUrl.pathname + parsedUrl.search,
|
||||
method: 'POST',
|
||||
headers: {
|
||||
'Content-Type': 'application/json',
|
||||
'Content-Length': Buffer.byteLength(bodyStr),
|
||||
},
|
||||
timeout: REQUEST_TIMEOUT_MS,
|
||||
};
|
||||
|
||||
const req = lib.request(options, (res) => {
|
||||
let data = '';
|
||||
res.on('data', (chunk) => { data += chunk; });
|
||||
res.on('end', () => {
|
||||
resolve({
|
||||
ok: res.statusCode >= 200 && res.statusCode < 300,
|
||||
status: res.statusCode,
|
||||
body: data,
|
||||
});
|
||||
});
|
||||
});
|
||||
|
||||
req.on('timeout', () => {
|
||||
req.destroy(new Error(`Request timed out after ${REQUEST_TIMEOUT_MS / 60000}m`));
|
||||
});
|
||||
req.on('error', reject);
|
||||
|
||||
req.write(bodyStr);
|
||||
req.end();
|
||||
});
|
||||
}
|
||||
|
||||
// ─── Markdown Helpers ─────────────────────────────────────────────────────────
|
||||
|
||||
function mdSection(title, level = 2) {
|
||||
return `${'#'.repeat(level)} ${title}\n\n`;
|
||||
}
|
||||
|
||||
function mdCode(content, lang = '') {
|
||||
if (content === null || content === undefined) return '*null*\n\n';
|
||||
const str = typeof content === 'string' ? content : JSON.stringify(content, null, 2);
|
||||
return `\`\`\`${lang}\n${str}\n\`\`\`\n\n`;
|
||||
}
|
||||
|
||||
function mdField(label, value) {
|
||||
const display = (value === null || value === undefined || value === '') ? '*empty*' : `\`${value}\``;
|
||||
return `- **${label}**: ${display}\n`;
|
||||
}
|
||||
|
||||
function mdTable(headers, rows) {
|
||||
if (!rows || rows.length === 0) return '*No items.*\n\n';
|
||||
const sep = headers.map(() => '---');
|
||||
const lines = [
|
||||
`| ${headers.join(' | ')} |`,
|
||||
`| ${sep.join(' | ')} |`,
|
||||
...rows.map(r => `| ${r.map(c => String(c ?? '').replace(/\|/g, '\\|')).join(' | ')} |`),
|
||||
];
|
||||
return lines.join('\n') + '\n\n';
|
||||
}
|
||||
|
||||
// ─── Main ─────────────────────────────────────────────────────────────────────
|
||||
|
||||
async function main() {
|
||||
// Discover all image files
|
||||
let files;
|
||||
try {
|
||||
files = fs.readdirSync(TEST_IMAGES_DIR).filter(f =>
|
||||
/\.(jpe?g|png|webp|bmp)$/i.test(f)
|
||||
).sort();
|
||||
} catch (e) {
|
||||
console.error(`Cannot read test-images directory: ${TEST_IMAGES_DIR}`);
|
||||
console.error(e.message);
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
if (files.length === 0) {
|
||||
console.error('No image files found in', TEST_IMAGES_DIR);
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
console.log(`\n🚀 Starting batch test`);
|
||||
console.log(` API endpoint : ${BASE_URL}`);
|
||||
console.log(` Images found : ${files.length}`);
|
||||
console.log(` Output JSON : ${OUTPUT_JSON}`);
|
||||
console.log(` Output MD : ${OUTPUT_MD}`);
|
||||
console.log('─'.repeat(60));
|
||||
|
||||
const jsonResults = [];
|
||||
const mdParts = [];
|
||||
const summaryRows = [];
|
||||
|
||||
// ── Markdown document header ──────────────────────────────────────────────
|
||||
mdParts.push(
|
||||
`# OCR Batch Test Report\n\n`,
|
||||
`> Generated: ${new Date().toISOString()}\n`,
|
||||
`> API: \`${BASE_URL}\`\n`,
|
||||
`> Images: **${files.length}** files from \`backend/sources/test-images/\`\n\n`,
|
||||
`---\n\n`,
|
||||
`## Summary\n\n`,
|
||||
'<!-- summary_table_placeholder -->\n\n',
|
||||
`---\n\n`,
|
||||
`## Stage-by-Stage Results\n\n`,
|
||||
);
|
||||
const summaryPlaceholderIndex = mdParts.indexOf('<!-- summary_table_placeholder -->\n\n');
|
||||
|
||||
// ── Process each file ────────────────────────────────────────────────────
|
||||
for (let idx = 0; idx < files.length; idx++) {
|
||||
const file = files[idx];
|
||||
const num = `[${String(idx + 1).padStart(2, '0')}/${files.length}]`;
|
||||
process.stdout.write(`${num} ${file} ... `);
|
||||
|
||||
const entry = {
|
||||
index: idx + 1,
|
||||
filename: file,
|
||||
status: 'pending',
|
||||
tilt: null,
|
||||
unwarped: null,
|
||||
// pipeline stages
|
||||
rawMarkdown: null,
|
||||
layer1RawRegex: null,
|
||||
layer2Sanitized: null,
|
||||
layer3Final: null,
|
||||
items: [],
|
||||
error: null,
|
||||
};
|
||||
|
||||
try {
|
||||
const t0 = Date.now();
|
||||
const res = await postJSON(BASE_URL, { filename: file });
|
||||
const elapsed = ((Date.now() - t0) / 1000).toFixed(1);
|
||||
|
||||
if (!res.ok) {
|
||||
process.stdout.write(`❌ HTTP ${res.status} (${elapsed}s)\n`);
|
||||
entry.status = 'http_error';
|
||||
entry.error = `HTTP ${res.status}: ${res.body}`;
|
||||
} else {
|
||||
let data;
|
||||
try {
|
||||
data = JSON.parse(res.body);
|
||||
} catch (_) {
|
||||
entry.status = 'json_parse_error';
|
||||
entry.error = 'Response is not valid JSON';
|
||||
process.stdout.write(`❌ JSON parse error (${elapsed}s)\n`);
|
||||
data = null;
|
||||
}
|
||||
|
||||
if (data) {
|
||||
if (data.error) {
|
||||
process.stdout.write(`⚠️ API error: ${data.error} (${elapsed}s)\n`);
|
||||
entry.status = 'api_error';
|
||||
entry.error = data.error;
|
||||
} else {
|
||||
const pipelineInfo = (data.result || {}).pipeline_info || {};
|
||||
entry.status = 'success';
|
||||
entry.tilt = pipelineInfo.tilt !== undefined ? +parseFloat(pipelineInfo.tilt).toFixed(2) : null;
|
||||
entry.unwarped = pipelineInfo.unwarped ?? null;
|
||||
|
||||
const ppd = data.postProcessingDetails || {};
|
||||
entry.rawMarkdown = ppd.rawMarkdown ?? null;
|
||||
entry.layer1RawRegex = ppd.layer1RawRegex ?? null;
|
||||
entry.layer2Sanitized = ppd.layer2Sanitized ?? null;
|
||||
entry.layer3Final = ppd.layer3Final ?? null;
|
||||
entry.items = data.items ?? [];
|
||||
|
||||
const itemCount = entry.items.length;
|
||||
process.stdout.write(`✅ ${itemCount} item(s), tilt=${entry.tilt ?? 'N/A'}° (${elapsed}s)\n`);
|
||||
}
|
||||
}
|
||||
}
|
||||
} catch (err) {
|
||||
process.stdout.write(`💥 ${err.message}\n`);
|
||||
entry.status = 'exception';
|
||||
entry.error = err.message;
|
||||
}
|
||||
|
||||
jsonResults.push(entry);
|
||||
|
||||
// ── Build per-file markdown section ─────────────────────────────────────
|
||||
const statusEmoji = {
|
||||
success: '✅',
|
||||
http_error: '❌',
|
||||
api_error: '⚠️',
|
||||
json_parse_error: '❌',
|
||||
exception: '💥',
|
||||
}[entry.status] || '❓';
|
||||
|
||||
let fileMd = '';
|
||||
fileMd += `### ${idx + 1}. \`${file}\`\n\n`;
|
||||
fileMd += `**Status**: ${statusEmoji} \`${entry.status}\`\n\n`;
|
||||
|
||||
if (entry.status !== 'success') {
|
||||
fileMd += `> **Error**: ${entry.error}\n\n`;
|
||||
fileMd += `---\n\n`;
|
||||
summaryRows.push([idx + 1, `\`${file}\``, `${statusEmoji} ${entry.status}`, 'N/A', 'N/A', 'N/A', 'N/A']);
|
||||
mdParts.push(fileMd);
|
||||
continue;
|
||||
}
|
||||
|
||||
// ── Stage 0: Pipeline Info ────────────────────────────────────────────────
|
||||
fileMd += `#### 📐 Stage 0 — Pipeline Info\n\n`;
|
||||
fileMd += mdField('Tilt detected', entry.tilt !== null ? `${entry.tilt}°` : 'N/A');
|
||||
fileMd += mdField('Auto-unwarped', entry.unwarped !== null ? (entry.unwarped ? 'Yes' : 'No') : 'N/A');
|
||||
fileMd += '\n';
|
||||
|
||||
// ── Stage 1: Raw Markdown from OCR ───────────────────────────────────────
|
||||
fileMd += `#### 📄 Stage 1 — Raw OCR Markdown\n\n`;
|
||||
fileMd += `*This is the raw text extracted by PaddleOCR layout parser before any post-processing.*\n\n`;
|
||||
if (entry.rawMarkdown) {
|
||||
fileMd += mdCode(entry.rawMarkdown, 'markdown');
|
||||
} else {
|
||||
fileMd += '*No raw markdown captured.*\n\n';
|
||||
}
|
||||
|
||||
// ── Stage 2: Layer 1 — Regex Extraction ──────────────────────────────────
|
||||
fileMd += `#### 🔍 Stage 2 — Layer 1: Regex Extraction (\`parseDOMetadata\`)\n\n`;
|
||||
fileMd += `*Regex patterns are applied to raw markdown to extract header fields and item rows.*\n\n`;
|
||||
if (entry.layer1RawRegex) {
|
||||
const l1 = entry.layer1RawRegex;
|
||||
fileMd += `**Header fields (raw regex output):**\n\n`;
|
||||
fileMd += mdField('noDO', l1.noDO);
|
||||
fileMd += mdField('noPO', l1.noPO);
|
||||
fileMd += mdField('noSO', l1.noSO);
|
||||
fileMd += mdField('tanggal', l1.tanggal);
|
||||
fileMd += mdField('vendorInfo', l1.vendorInfo);
|
||||
fileMd += mdField('customerInfo', l1.customerInfo);
|
||||
fileMd += mdField('alamat', l1.alamat);
|
||||
fileMd += mdField('orderUntuk', l1.orderUntuk);
|
||||
fileMd += mdField('platTruk', l1.platTruk);
|
||||
fileMd += '\n';
|
||||
|
||||
fileMd += `**Raw items (${(l1.items || []).length} row(s)):**\n\n`;
|
||||
fileMd += mdTable(
|
||||
['kodeBarang', 'namaBarang', 'banyak', 'jumlah'],
|
||||
(l1.items || []).map(it => [it.kodeBarang, it.namaBarang, it.banyak, it.jumlah])
|
||||
);
|
||||
} else {
|
||||
fileMd += '*Layer 1 data not captured.*\n\n';
|
||||
}
|
||||
|
||||
// ── Stage 3: Layer 2 — Sanitized ─────────────────────────────────────────
|
||||
fileMd += `#### 🧹 Stage 3 — Layer 2: Sanitized (\`sanitizeParsedMetadata\`)\n\n`;
|
||||
fileMd += `*Strict format enforcement: corrects date formats, trims whitespace, enforces field constraints.*\n\n`;
|
||||
if (entry.layer2Sanitized) {
|
||||
const l2 = entry.layer2Sanitized;
|
||||
fileMd += `**Header fields (after sanitization):**\n\n`;
|
||||
fileMd += mdField('noDO', l2.noDO);
|
||||
fileMd += mdField('noPO', l2.noPO);
|
||||
fileMd += mdField('noSO', l2.noSO);
|
||||
fileMd += mdField('tanggal', l2.tanggal);
|
||||
fileMd += mdField('vendorInfo', l2.vendorInfo);
|
||||
fileMd += mdField('customerInfo', l2.customerInfo);
|
||||
fileMd += mdField('alamat', l2.alamat);
|
||||
fileMd += mdField('orderUntuk', l2.orderUntuk);
|
||||
fileMd += mdField('platTruk', l2.platTruk);
|
||||
fileMd += '\n';
|
||||
|
||||
fileMd += `**Sanitized items (${(l2.items || []).length} row(s)):**\n\n`;
|
||||
fileMd += mdTable(
|
||||
['kodeBarang', 'namaBarang', 'banyak', 'jumlah'],
|
||||
(l2.items || []).map(it => [it.kodeBarang, it.namaBarang, it.banyak, it.jumlah])
|
||||
);
|
||||
} else {
|
||||
fileMd += '*Layer 2 data not captured.*\n\n';
|
||||
}
|
||||
|
||||
// ── Stage 4: Layer 3 — Final (SKU triple-check + store resolution) ────────
|
||||
fileMd += `#### ✅ Stage 4 — Layer 3: Final (\`SKU triple-check + store resolution\`)\n\n`;
|
||||
fileMd += `*SKU validated against master list (score ≥ 0.6 threshold). Items with noise SKU codes are filtered out. Store resolved from DB.*\n\n`;
|
||||
if (entry.layer3Final) {
|
||||
const l3 = entry.layer3Final;
|
||||
fileMd += `**Final metadata:**\n\n`;
|
||||
fileMd += mdField('noDO', l3.noDO);
|
||||
fileMd += mdField('noPO', l3.noPO);
|
||||
fileMd += mdField('noSO', l3.noSO);
|
||||
fileMd += mdField('tanggal', l3.tanggal);
|
||||
fileMd += mdField('vendorInfo', l3.vendorInfo);
|
||||
fileMd += mdField('customerInfo', l3.customerInfo);
|
||||
fileMd += mdField('alamat', l3.alamat);
|
||||
fileMd += mdField('orderUntuk', l3.orderUntuk);
|
||||
fileMd += mdField('platTruk', l3.platTruk);
|
||||
fileMd += '\n';
|
||||
|
||||
fileMd += `**Final items after SKU validation (${(l3.items || []).length} row(s)):**\n\n`;
|
||||
fileMd += mdTable(
|
||||
['kodeBarangOriginal', 'kodeBarang (corrected)', 'namaBarang', 'banyak', 'jumlah'],
|
||||
(l3.items || []).map(it => [
|
||||
it.kodeBarangOriginal ?? it.kodeBarang,
|
||||
it.kodeBarang,
|
||||
it.namaBarang,
|
||||
it.banyak,
|
||||
it.jumlah
|
||||
])
|
||||
);
|
||||
} else {
|
||||
fileMd += '*Layer 3 data not captured.*\n\n';
|
||||
}
|
||||
|
||||
// ── Stage 5: Final submitted items (from root items[]) ───────────────────
|
||||
fileMd += `#### 🗃️ Stage 5 — Submitted Items (ready-to-use JSON)\n\n`;
|
||||
fileMd += `*These are the items actually returned to the caller and saved to the database.*\n\n`;
|
||||
fileMd += mdTable(
|
||||
['kodeBarangOriginal', 'kodeBarang', 'namaBarang', 'banyak', 'jumlah'],
|
||||
(entry.items || []).map(it => [
|
||||
it.kodeBarangOriginal ?? it.kodeBarang,
|
||||
it.kodeBarang,
|
||||
it.namaBarang,
|
||||
it.banyak,
|
||||
it.jumlah
|
||||
])
|
||||
);
|
||||
|
||||
fileMd += `---\n\n`;
|
||||
|
||||
// ── Summary row ──────────────────────────────────────────────────────────
|
||||
const l3meta = entry.layer3Final || {};
|
||||
summaryRows.push([
|
||||
idx + 1,
|
||||
`\`${file}\``,
|
||||
`${statusEmoji} success`,
|
||||
entry.tilt !== null ? `${entry.tilt}°` : 'N/A',
|
||||
entry.unwarped !== null ? (entry.unwarped ? 'Yes' : 'No') : 'N/A',
|
||||
`\`${l3meta.noDO ?? 'N/A'}\``,
|
||||
`\`${l3meta.noPO ?? 'N/A'}\``,
|
||||
`${entry.items.length}`,
|
||||
]);
|
||||
|
||||
mdParts.push(fileMd);
|
||||
}
|
||||
|
||||
// ── Inject summary table ──────────────────────────────────────────────────
|
||||
const summaryTable = mdTable(
|
||||
['#', 'Filename', 'Status', 'Tilt', 'Unwarped', 'DO', 'PO', 'Items'],
|
||||
summaryRows
|
||||
);
|
||||
mdParts[summaryPlaceholderIndex] = summaryTable;
|
||||
|
||||
// ── Write outputs ─────────────────────────────────────────────────────────
|
||||
const jsonOut = JSON.stringify(jsonResults, null, 2);
|
||||
fs.writeFileSync(OUTPUT_JSON, jsonOut, 'utf8');
|
||||
console.log(`\n✅ JSON saved → ${OUTPUT_JSON}`);
|
||||
|
||||
const mdOut = mdParts.join('');
|
||||
fs.writeFileSync(OUTPUT_MD, mdOut, 'utf8');
|
||||
console.log(`✅ MD saved → ${OUTPUT_MD}`);
|
||||
|
||||
// ── Final stats ───────────────────────────────────────────────────────────
|
||||
const succeeded = jsonResults.filter(r => r.status === 'success').length;
|
||||
const failed = jsonResults.length - succeeded;
|
||||
console.log('\n─'.repeat(60));
|
||||
console.log(` Total : ${jsonResults.length}`);
|
||||
console.log(` Success: ${succeeded}`);
|
||||
console.log(` Failed : ${failed}`);
|
||||
console.log('─'.repeat(60));
|
||||
}
|
||||
|
||||
main().catch(err => {
|
||||
console.error('Fatal error:', err);
|
||||
process.exit(1);
|
||||
});
|
||||
/**
|
||||
* run_full_test.js
|
||||
*
|
||||
* Runs OCR parsing against ALL images in backend/sources/test-images/
|
||||
* and captures every pipeline stage for analysis:
|
||||
* - rawMarkdown : raw text from PaddleOCR layout parser
|
||||
* - layer1RawRegex: output of parseDOMetadata (regex extraction)
|
||||
* - layer2Sanitized: output of sanitizeParsedMetadata (format checks)
|
||||
* - layer3Final : final metadata after SKU triple-check + store resolution
|
||||
*
|
||||
* Outputs:
|
||||
* backend/sources/ai_results.json — machine-readable per-file results
|
||||
* backend/sources/ai_results.md — human-readable stage-by-stage breakdown
|
||||
*
|
||||
* Usage (from host machine, Docker must be running):
|
||||
* node run_full_test.js
|
||||
*
|
||||
* The script talks to the nginx gateway on port 8000.
|
||||
* To override: set env var BASE_URL=http://localhost:3000/api/parse
|
||||
*/
|
||||
|
||||
const fs = require('fs');
|
||||
const path = require('path');
|
||||
const http = require('http');
|
||||
const https = require('https');
|
||||
|
||||
// ─── Config ──────────────────────────────────────────────────────────────────
|
||||
|
||||
const BASE_URL = process.env.BASE_URL || 'http://localhost:8000/api/parse';
|
||||
const TEST_IMAGES_DIR = path.resolve(__dirname, '../sources/test-images');
|
||||
const OUTPUT_JSON = path.resolve(__dirname, '../sources/ai_results.json');
|
||||
const OUTPUT_MD = path.resolve(__dirname, '../sources/ai_results.md');
|
||||
const REQUEST_TIMEOUT_MS = 20 * 60 * 1000; // 20 minutes per image
|
||||
|
||||
// ─── HTTP Helper ─────────────────────────────────────────────────────────────
|
||||
|
||||
function postJSON(url, body) {
|
||||
return new Promise((resolve, reject) => {
|
||||
const parsedUrl = new URL(url);
|
||||
const bodyStr = JSON.stringify(body);
|
||||
const lib = parsedUrl.protocol === 'https:' ? https : http;
|
||||
|
||||
const options = {
|
||||
hostname: parsedUrl.hostname,
|
||||
port: parsedUrl.port || (parsedUrl.protocol === 'https:' ? 443 : 80),
|
||||
path: parsedUrl.pathname + parsedUrl.search,
|
||||
method: 'POST',
|
||||
headers: {
|
||||
'Content-Type': 'application/json',
|
||||
'Content-Length': Buffer.byteLength(bodyStr),
|
||||
},
|
||||
timeout: REQUEST_TIMEOUT_MS,
|
||||
};
|
||||
|
||||
const req = lib.request(options, (res) => {
|
||||
let data = '';
|
||||
res.on('data', (chunk) => { data += chunk; });
|
||||
res.on('end', () => {
|
||||
resolve({
|
||||
ok: res.statusCode >= 200 && res.statusCode < 300,
|
||||
status: res.statusCode,
|
||||
body: data,
|
||||
});
|
||||
});
|
||||
});
|
||||
|
||||
req.on('timeout', () => {
|
||||
req.destroy(new Error(`Request timed out after ${REQUEST_TIMEOUT_MS / 60000}m`));
|
||||
});
|
||||
req.on('error', reject);
|
||||
|
||||
req.write(bodyStr);
|
||||
req.end();
|
||||
});
|
||||
}
|
||||
|
||||
// ─── Markdown Helpers ─────────────────────────────────────────────────────────
|
||||
|
||||
function mdSection(title, level = 2) {
|
||||
return `${'#'.repeat(level)} ${title}\n\n`;
|
||||
}
|
||||
|
||||
function mdCode(content, lang = '') {
|
||||
if (content === null || content === undefined) return '*null*\n\n';
|
||||
const str = typeof content === 'string' ? content : JSON.stringify(content, null, 2);
|
||||
return `\`\`\`${lang}\n${str}\n\`\`\`\n\n`;
|
||||
}
|
||||
|
||||
function mdField(label, value) {
|
||||
const display = (value === null || value === undefined || value === '') ? '*empty*' : `\`${value}\``;
|
||||
return `- **${label}**: ${display}\n`;
|
||||
}
|
||||
|
||||
function mdTable(headers, rows) {
|
||||
if (!rows || rows.length === 0) return '*No items.*\n\n';
|
||||
const sep = headers.map(() => '---');
|
||||
const lines = [
|
||||
`| ${headers.join(' | ')} |`,
|
||||
`| ${sep.join(' | ')} |`,
|
||||
...rows.map(r => `| ${r.map(c => String(c ?? '').replace(/\|/g, '\\|')).join(' | ')} |`),
|
||||
];
|
||||
return lines.join('\n') + '\n\n';
|
||||
}
|
||||
|
||||
// ─── Main ─────────────────────────────────────────────────────────────────────
|
||||
|
||||
async function main() {
|
||||
// Discover all image files
|
||||
let files;
|
||||
try {
|
||||
files = fs.readdirSync(TEST_IMAGES_DIR).filter(f =>
|
||||
/\.(jpe?g|png|webp|bmp)$/i.test(f)
|
||||
).sort();
|
||||
} catch (e) {
|
||||
console.error(`Cannot read test-images directory: ${TEST_IMAGES_DIR}`);
|
||||
console.error(e.message);
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
if (files.length === 0) {
|
||||
console.error('No image files found in', TEST_IMAGES_DIR);
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
console.log(`\n🚀 Starting batch test`);
|
||||
console.log(` API endpoint : ${BASE_URL}`);
|
||||
console.log(` Images found : ${files.length}`);
|
||||
console.log(` Output JSON : ${OUTPUT_JSON}`);
|
||||
console.log(` Output MD : ${OUTPUT_MD}`);
|
||||
console.log('─'.repeat(60));
|
||||
|
||||
const jsonResults = [];
|
||||
const mdParts = [];
|
||||
const summaryRows = [];
|
||||
|
||||
// ── Markdown document header ──────────────────────────────────────────────
|
||||
mdParts.push(
|
||||
`# OCR Batch Test Report\n\n`,
|
||||
`> Generated: ${new Date().toISOString()}\n`,
|
||||
`> API: \`${BASE_URL}\`\n`,
|
||||
`> Images: **${files.length}** files from \`backend/sources/test-images/\`\n\n`,
|
||||
`---\n\n`,
|
||||
`## Summary\n\n`,
|
||||
'<!-- summary_table_placeholder -->\n\n',
|
||||
`---\n\n`,
|
||||
`## Stage-by-Stage Results\n\n`,
|
||||
);
|
||||
const summaryPlaceholderIndex = mdParts.indexOf('<!-- summary_table_placeholder -->\n\n');
|
||||
|
||||
// ── Process each file ────────────────────────────────────────────────────
|
||||
for (let idx = 0; idx < files.length; idx++) {
|
||||
const file = files[idx];
|
||||
const num = `[${String(idx + 1).padStart(2, '0')}/${files.length}]`;
|
||||
process.stdout.write(`${num} ${file} ... `);
|
||||
|
||||
const entry = {
|
||||
index: idx + 1,
|
||||
filename: file,
|
||||
status: 'pending',
|
||||
tilt: null,
|
||||
unwarped: null,
|
||||
// pipeline stages
|
||||
rawMarkdown: null,
|
||||
layer1RawRegex: null,
|
||||
layer2Sanitized: null,
|
||||
layer3Final: null,
|
||||
items: [],
|
||||
error: null,
|
||||
};
|
||||
|
||||
try {
|
||||
const t0 = Date.now();
|
||||
const res = await postJSON(BASE_URL, { filename: file });
|
||||
const elapsed = ((Date.now() - t0) / 1000).toFixed(1);
|
||||
|
||||
if (!res.ok) {
|
||||
process.stdout.write(`❌ HTTP ${res.status} (${elapsed}s)\n`);
|
||||
entry.status = 'http_error';
|
||||
entry.error = `HTTP ${res.status}: ${res.body}`;
|
||||
} else {
|
||||
let data;
|
||||
try {
|
||||
data = JSON.parse(res.body);
|
||||
} catch (_) {
|
||||
entry.status = 'json_parse_error';
|
||||
entry.error = 'Response is not valid JSON';
|
||||
process.stdout.write(`❌ JSON parse error (${elapsed}s)\n`);
|
||||
data = null;
|
||||
}
|
||||
|
||||
if (data) {
|
||||
if (data.error) {
|
||||
process.stdout.write(`⚠️ API error: ${data.error} (${elapsed}s)\n`);
|
||||
entry.status = 'api_error';
|
||||
entry.error = data.error;
|
||||
} else {
|
||||
const pipelineInfo = (data.result || {}).pipeline_info || {};
|
||||
entry.status = 'success';
|
||||
entry.tilt = pipelineInfo.tilt !== undefined ? +parseFloat(pipelineInfo.tilt).toFixed(2) : null;
|
||||
entry.unwarped = pipelineInfo.unwarped ?? null;
|
||||
|
||||
const ppd = data.postProcessingDetails || {};
|
||||
entry.rawMarkdown = ppd.rawMarkdown ?? null;
|
||||
entry.layer1RawRegex = ppd.layer1RawRegex ?? null;
|
||||
entry.layer2Sanitized = ppd.layer2Sanitized ?? null;
|
||||
entry.layer3Final = ppd.layer3Final ?? null;
|
||||
entry.items = data.items ?? [];
|
||||
|
||||
const itemCount = entry.items.length;
|
||||
process.stdout.write(`✅ ${itemCount} item(s), tilt=${entry.tilt ?? 'N/A'}° (${elapsed}s)\n`);
|
||||
}
|
||||
}
|
||||
}
|
||||
} catch (err) {
|
||||
process.stdout.write(`💥 ${err.message}\n`);
|
||||
entry.status = 'exception';
|
||||
entry.error = err.message;
|
||||
}
|
||||
|
||||
jsonResults.push(entry);
|
||||
|
||||
// ── Build per-file markdown section ─────────────────────────────────────
|
||||
const statusEmoji = {
|
||||
success: '✅',
|
||||
http_error: '❌',
|
||||
api_error: '⚠️',
|
||||
json_parse_error: '❌',
|
||||
exception: '💥',
|
||||
}[entry.status] || '❓';
|
||||
|
||||
let fileMd = '';
|
||||
fileMd += `### ${idx + 1}. \`${file}\`\n\n`;
|
||||
fileMd += `**Status**: ${statusEmoji} \`${entry.status}\`\n\n`;
|
||||
|
||||
if (entry.status !== 'success') {
|
||||
fileMd += `> **Error**: ${entry.error}\n\n`;
|
||||
fileMd += `---\n\n`;
|
||||
summaryRows.push([idx + 1, `\`${file}\``, `${statusEmoji} ${entry.status}`, 'N/A', 'N/A', 'N/A', 'N/A']);
|
||||
mdParts.push(fileMd);
|
||||
continue;
|
||||
}
|
||||
|
||||
// ── Stage 0: Pipeline Info ────────────────────────────────────────────────
|
||||
fileMd += `#### 📐 Stage 0 — Pipeline Info\n\n`;
|
||||
fileMd += mdField('Tilt detected', entry.tilt !== null ? `${entry.tilt}°` : 'N/A');
|
||||
fileMd += mdField('Auto-unwarped', entry.unwarped !== null ? (entry.unwarped ? 'Yes' : 'No') : 'N/A');
|
||||
fileMd += '\n';
|
||||
|
||||
// ── Stage 1: Raw Markdown from OCR ───────────────────────────────────────
|
||||
fileMd += `#### 📄 Stage 1 — Raw OCR Markdown\n\n`;
|
||||
fileMd += `*This is the raw text extracted by PaddleOCR layout parser before any post-processing.*\n\n`;
|
||||
if (entry.rawMarkdown) {
|
||||
fileMd += mdCode(entry.rawMarkdown, 'markdown');
|
||||
} else {
|
||||
fileMd += '*No raw markdown captured.*\n\n';
|
||||
}
|
||||
|
||||
// ── Stage 2: Layer 1 — Regex Extraction ──────────────────────────────────
|
||||
fileMd += `#### 🔍 Stage 2 — Layer 1: Regex Extraction (\`parseDOMetadata\`)\n\n`;
|
||||
fileMd += `*Regex patterns are applied to raw markdown to extract header fields and item rows.*\n\n`;
|
||||
if (entry.layer1RawRegex) {
|
||||
const l1 = entry.layer1RawRegex;
|
||||
fileMd += `**Header fields (raw regex output):**\n\n`;
|
||||
fileMd += mdField('noDO', l1.noDO);
|
||||
fileMd += mdField('noPO', l1.noPO);
|
||||
fileMd += mdField('noSO', l1.noSO);
|
||||
fileMd += mdField('tanggal', l1.tanggal);
|
||||
fileMd += mdField('vendorInfo', l1.vendorInfo);
|
||||
fileMd += mdField('customerInfo', l1.customerInfo);
|
||||
fileMd += mdField('alamat', l1.alamat);
|
||||
fileMd += mdField('orderUntuk', l1.orderUntuk);
|
||||
fileMd += mdField('platTruk', l1.platTruk);
|
||||
fileMd += '\n';
|
||||
|
||||
fileMd += `**Raw items (${(l1.items || []).length} row(s)):**\n\n`;
|
||||
fileMd += mdTable(
|
||||
['kodeBarang', 'namaBarang', 'banyak', 'jumlah'],
|
||||
(l1.items || []).map(it => [it.kodeBarang, it.namaBarang, it.banyak, it.jumlah])
|
||||
);
|
||||
} else {
|
||||
fileMd += '*Layer 1 data not captured.*\n\n';
|
||||
}
|
||||
|
||||
// ── Stage 3: Layer 2 — Sanitized ─────────────────────────────────────────
|
||||
fileMd += `#### 🧹 Stage 3 — Layer 2: Sanitized (\`sanitizeParsedMetadata\`)\n\n`;
|
||||
fileMd += `*Strict format enforcement: corrects date formats, trims whitespace, enforces field constraints.*\n\n`;
|
||||
if (entry.layer2Sanitized) {
|
||||
const l2 = entry.layer2Sanitized;
|
||||
fileMd += `**Header fields (after sanitization):**\n\n`;
|
||||
fileMd += mdField('noDO', l2.noDO);
|
||||
fileMd += mdField('noPO', l2.noPO);
|
||||
fileMd += mdField('noSO', l2.noSO);
|
||||
fileMd += mdField('tanggal', l2.tanggal);
|
||||
fileMd += mdField('vendorInfo', l2.vendorInfo);
|
||||
fileMd += mdField('customerInfo', l2.customerInfo);
|
||||
fileMd += mdField('alamat', l2.alamat);
|
||||
fileMd += mdField('orderUntuk', l2.orderUntuk);
|
||||
fileMd += mdField('platTruk', l2.platTruk);
|
||||
fileMd += '\n';
|
||||
|
||||
fileMd += `**Sanitized items (${(l2.items || []).length} row(s)):**\n\n`;
|
||||
fileMd += mdTable(
|
||||
['kodeBarang', 'namaBarang', 'banyak', 'jumlah'],
|
||||
(l2.items || []).map(it => [it.kodeBarang, it.namaBarang, it.banyak, it.jumlah])
|
||||
);
|
||||
} else {
|
||||
fileMd += '*Layer 2 data not captured.*\n\n';
|
||||
}
|
||||
|
||||
// ── Stage 4: Layer 3 — Final (SKU triple-check + store resolution) ────────
|
||||
fileMd += `#### ✅ Stage 4 — Layer 3: Final (\`SKU triple-check + store resolution\`)\n\n`;
|
||||
fileMd += `*SKU validated against master list (score ≥ 0.6 threshold). Items with noise SKU codes are filtered out. Store resolved from DB.*\n\n`;
|
||||
if (entry.layer3Final) {
|
||||
const l3 = entry.layer3Final;
|
||||
fileMd += `**Final metadata:**\n\n`;
|
||||
fileMd += mdField('noDO', l3.noDO);
|
||||
fileMd += mdField('noPO', l3.noPO);
|
||||
fileMd += mdField('noSO', l3.noSO);
|
||||
fileMd += mdField('tanggal', l3.tanggal);
|
||||
fileMd += mdField('vendorInfo', l3.vendorInfo);
|
||||
fileMd += mdField('customerInfo', l3.customerInfo);
|
||||
fileMd += mdField('alamat', l3.alamat);
|
||||
fileMd += mdField('orderUntuk', l3.orderUntuk);
|
||||
fileMd += mdField('platTruk', l3.platTruk);
|
||||
fileMd += '\n';
|
||||
|
||||
fileMd += `**Final items after SKU validation (${(l3.items || []).length} row(s)):**\n\n`;
|
||||
fileMd += mdTable(
|
||||
['kodeBarangOriginal', 'kodeBarang (corrected)', 'namaBarang', 'banyak', 'jumlah'],
|
||||
(l3.items || []).map(it => [
|
||||
it.kodeBarangOriginal ?? it.kodeBarang,
|
||||
it.kodeBarang,
|
||||
it.namaBarang,
|
||||
it.banyak,
|
||||
it.jumlah
|
||||
])
|
||||
);
|
||||
} else {
|
||||
fileMd += '*Layer 3 data not captured.*\n\n';
|
||||
}
|
||||
|
||||
// ── Stage 5: Final submitted items (from root items[]) ───────────────────
|
||||
fileMd += `#### 🗃️ Stage 5 — Submitted Items (ready-to-use JSON)\n\n`;
|
||||
fileMd += `*These are the items actually returned to the caller and saved to the database.*\n\n`;
|
||||
fileMd += mdTable(
|
||||
['kodeBarangOriginal', 'kodeBarang', 'namaBarang', 'banyak', 'jumlah'],
|
||||
(entry.items || []).map(it => [
|
||||
it.kodeBarangOriginal ?? it.kodeBarang,
|
||||
it.kodeBarang,
|
||||
it.namaBarang,
|
||||
it.banyak,
|
||||
it.jumlah
|
||||
])
|
||||
);
|
||||
|
||||
fileMd += `---\n\n`;
|
||||
|
||||
// ── Summary row ──────────────────────────────────────────────────────────
|
||||
const l3meta = entry.layer3Final || {};
|
||||
summaryRows.push([
|
||||
idx + 1,
|
||||
`\`${file}\``,
|
||||
`${statusEmoji} success`,
|
||||
entry.tilt !== null ? `${entry.tilt}°` : 'N/A',
|
||||
entry.unwarped !== null ? (entry.unwarped ? 'Yes' : 'No') : 'N/A',
|
||||
`\`${l3meta.noDO ?? 'N/A'}\``,
|
||||
`\`${l3meta.noPO ?? 'N/A'}\``,
|
||||
`${entry.items.length}`,
|
||||
]);
|
||||
|
||||
mdParts.push(fileMd);
|
||||
}
|
||||
|
||||
// ── Inject summary table ──────────────────────────────────────────────────
|
||||
const summaryTable = mdTable(
|
||||
['#', 'Filename', 'Status', 'Tilt', 'Unwarped', 'DO', 'PO', 'Items'],
|
||||
summaryRows
|
||||
);
|
||||
mdParts[summaryPlaceholderIndex] = summaryTable;
|
||||
|
||||
// ── Write outputs ─────────────────────────────────────────────────────────
|
||||
const jsonOut = JSON.stringify(jsonResults, null, 2);
|
||||
fs.writeFileSync(OUTPUT_JSON, jsonOut, 'utf8');
|
||||
console.log(`\n✅ JSON saved → ${OUTPUT_JSON}`);
|
||||
|
||||
const mdOut = mdParts.join('');
|
||||
fs.writeFileSync(OUTPUT_MD, mdOut, 'utf8');
|
||||
console.log(`✅ MD saved → ${OUTPUT_MD}`);
|
||||
|
||||
// ── Final stats ───────────────────────────────────────────────────────────
|
||||
const succeeded = jsonResults.filter(r => r.status === 'success').length;
|
||||
const failed = jsonResults.length - succeeded;
|
||||
console.log('\n─'.repeat(60));
|
||||
console.log(` Total : ${jsonResults.length}`);
|
||||
console.log(` Success: ${succeeded}`);
|
||||
console.log(` Failed : ${failed}`);
|
||||
console.log('─'.repeat(60));
|
||||
}
|
||||
|
||||
main().catch(err => {
|
||||
console.error('Fatal error:', err);
|
||||
process.exit(1);
|
||||
});
|
||||
File diff suppressed because it is too large.
Load diff
@@ -1,419 +1,419 @@
|
||||
"use client";
|
||||
|
||||
import React, { useState, useEffect } from "react";
|
||||
|
||||
export default function MasterDataPage() {
|
||||
const [token, setToken] = useState<string | null>(null);
|
||||
const [username, setUsername] = useState("");
|
||||
const [password, setPassword] = useState("");
|
||||
const [loginError, setLoginError] = useState("");
|
||||
|
||||
const [activeTab, setActiveTab] = useState<"stores" | "skus">("stores");
|
||||
const [stores, setStores] = useState<any[]>([]);
|
||||
const [skus, setSkus] = useState<any[]>([]);
|
||||
|
||||
useEffect(() => {
|
||||
const savedToken = localStorage.getItem("adminToken");
|
||||
if (savedToken) {
|
||||
setToken(savedToken);
|
||||
fetchData(savedToken, activeTab);
|
||||
}
|
||||
}, [activeTab]);
|
||||
|
||||
const handleLogin = async (e: React.FormEvent) => {
|
||||
e.preventDefault();
|
||||
setLoginError("");
|
||||
try {
|
||||
const res = await fetch("/api/v1/auth/login", {
|
||||
method: "POST",
|
||||
headers: { "Content-Type": "application/json" },
|
||||
body: JSON.stringify({ username, password })
|
||||
});
|
||||
const data = await res.json();
|
||||
if (!res.ok) throw new Error(data.message || "Login failed");
|
||||
|
||||
const tokenStr = data.data?.token || data.token;
|
||||
localStorage.setItem("adminToken", tokenStr);
|
||||
setToken(tokenStr);
|
||||
fetchData(tokenStr, activeTab);
|
||||
} catch (err: any) {
|
||||
setLoginError(err.message);
|
||||
}
|
||||
};
|
||||
|
||||
const handleLogout = () => {
|
||||
localStorage.removeItem("adminToken");
|
||||
setToken(null);
|
||||
};
|
||||
|
||||
const fetchData = async (authToken: string, tab: "stores" | "skus") => {
|
||||
try {
|
||||
const res = await fetch(`/api/v1/master/${tab}`, {
|
||||
headers: { "Authorization": `Bearer ${authToken}` }
|
||||
});
|
||||
if (res.status === 401 || res.status === 403) {
|
||||
handleLogout();
|
||||
return;
|
||||
}
|
||||
const data = await res.json();
|
||||
if (res.ok) {
|
||||
if (tab === "stores") setStores(data.data || []);
|
||||
else setSkus(data.data || []);
|
||||
}
|
||||
} catch (err) {
|
||||
console.error(err);
|
||||
}
|
||||
};
|
||||
|
||||
if (!token) {
|
||||
return (
|
||||
<div className="min-h-screen bg-[radial-gradient(ellipse_at_top,_var(--tw-gradient-stops))] from-slate-900 via-slate-950 to-slate-950 flex items-center justify-center p-4">
|
||||
<div className="bg-slate-900/40 border border-slate-800/80 shadow-2xl backdrop-blur-md rounded-2xl p-8 w-full max-w-md">
|
||||
<h1 className="text-2xl font-extrabold bg-gradient-to-r from-teal-400 to-emerald-400 bg-clip-text text-transparent text-center mb-6">
|
||||
Admin Login
|
||||
</h1>
|
||||
<form onSubmit={handleLogin} className="space-y-5">
|
||||
<div>
|
||||
<label className="block text-xs font-semibold text-slate-400 uppercase tracking-wider mb-2">Username</label>
|
||||
<input
|
||||
type="text"
|
||||
className="w-full bg-slate-950/60 border border-slate-800 text-slate-100 placeholder-slate-600 rounded-xl p-3 focus:outline-none focus:border-teal-500/60 focus:ring-1 focus:ring-teal-500/60 transition-all duration-200 text-sm"
|
||||
value={username}
|
||||
onChange={e => setUsername(e.target.value)}
|
||||
placeholder="Enter admin username"
|
||||
/>
|
||||
</div>
|
||||
<div>
|
||||
<label className="block text-xs font-semibold text-slate-400 uppercase tracking-wider mb-2">Password</label>
|
||||
<input
|
||||
type="password"
|
||||
className="w-full bg-slate-950/60 border border-slate-800 text-slate-100 placeholder-slate-650 rounded-xl p-3 focus:outline-none focus:border-teal-500/60 focus:ring-1 focus:ring-teal-500/60 transition-all duration-200 text-sm"
|
||||
value={password}
|
||||
onChange={e => setPassword(e.target.value)}
|
||||
placeholder="••••••••"
|
||||
/>
|
||||
</div>
|
||||
{loginError && (
|
||||
<div className="bg-rose-950/30 border border-rose-800/40 p-3 rounded-xl text-xs text-rose-450 flex items-center gap-2">
|
||||
<span>⚠️</span>
|
||||
<span>{loginError}</span>
|
||||
</div>
|
||||
)}
|
||||
<button
|
||||
type="submit"
|
||||
className="w-full bg-teal-600 hover:bg-teal-500 text-slate-950 font-bold p-3 rounded-xl transition-all duration-200 shadow-lg shadow-teal-900/20 text-sm cursor-pointer"
|
||||
>
|
||||
Log In
|
||||
</button>
|
||||
</form>
|
||||
</div>
|
||||
</div>
|
||||
);
|
||||
}
|
||||
|
||||
return (
|
||||
<div className="min-h-screen bg-[radial-gradient(ellipse_at_top,_var(--tw-gradient-stops))] from-slate-900 via-slate-950 to-slate-950 p-8 text-slate-100">
|
||||
<div className="max-w-6xl mx-auto">
|
||||
<div className="flex justify-between items-center mb-8 border-b border-slate-800/60 pb-4">
|
||||
<h1 className="text-2xl font-extrabold bg-gradient-to-r from-teal-400 to-emerald-400 bg-clip-text text-transparent flex items-center gap-2">
|
||||
<span>⚙️</span> Master Data Management
|
||||
</h1>
|
||||
<button
|
||||
onClick={handleLogout}
|
||||
className="text-slate-400 hover:text-slate-100 bg-slate-900/60 hover:bg-slate-900 border border-slate-850 px-4 py-2 rounded-xl text-xs font-semibold transition-all duration-200 cursor-pointer"
|
||||
>
|
||||
Logout
|
||||
</button>
|
||||
</div>
|
||||
|
||||
<div className="flex space-x-2 mb-6 border-b border-slate-800/60 pb-px">
|
||||
<button
|
||||
className={`pb-2.5 px-4 text-sm font-semibold transition-all border-b-2 -mb-px cursor-pointer ${
|
||||
activeTab === 'stores'
|
||||
? 'border-teal-500 text-teal-400'
|
||||
: 'border-transparent text-slate-400 hover:text-slate-200'
|
||||
}`}
|
||||
onClick={() => setActiveTab('stores')}
|
||||
>
|
||||
Stores
|
||||
</button>
|
||||
<button
|
||||
className={`pb-2.5 px-4 text-sm font-semibold transition-all border-b-2 -mb-px cursor-pointer ${
|
||||
activeTab === 'skus'
|
||||
? 'border-teal-500 text-teal-400'
|
||||
: 'border-transparent text-slate-400 hover:text-slate-200'
|
||||
}`}
|
||||
onClick={() => setActiveTab('skus')}
|
||||
>
|
||||
SKUs
|
||||
</button>
|
||||
</div>
|
||||
|
||||
<div className="bg-slate-900/40 border border-slate-850 rounded-2xl p-6 shadow-xl backdrop-blur-md">
|
||||
{activeTab === 'stores' && <StoreManager stores={stores} token={token} onRefresh={() => fetchData(token, 'stores')} />}
|
||||
{activeTab === 'skus' && <SkuManager skus={skus} token={token} onRefresh={() => fetchData(token, 'skus')} />}
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
);
|
||||
}
|
||||
|
||||
function StoreManager({ stores, token, onRefresh }: { stores: any[], token: string, onRefresh: () => void }) {
|
||||
const [isAdding, setIsAdding] = useState(false);
|
||||
const [form, setForm] = useState({ kode_toko: "", nama_toko: "", alamat: "" });
|
||||
const [error, setError] = useState("");
|
||||
|
||||
const handleSubmit = async (e: React.FormEvent) => {
|
||||
e.preventDefault();
|
||||
setError("");
|
||||
try {
|
||||
const res = await fetch("/api/v1/master/stores", {
|
||||
method: "POST",
|
||||
headers: { "Content-Type": "application/json", "Authorization": `Bearer ${token}` },
|
||||
body: JSON.stringify(form)
|
||||
});
|
||||
const data = await res.json();
|
||||
if (!res.ok) throw new Error(data.message);
|
||||
setIsAdding(false);
|
||||
setForm({ kode_toko: "", nama_toko: "", alamat: "" });
|
||||
onRefresh();
|
||||
} catch (err: any) {
|
||||
setError(err.message);
|
||||
}
|
||||
};
|
||||
|
||||
const handleDelete = async (kode: string) => {
|
||||
if (!confirm(`Delete store ${kode}?`)) return;
|
||||
try {
|
||||
const res = await fetch(`/api/v1/master/stores/${kode}`, {
|
||||
method: "DELETE",
|
||||
headers: { "Authorization": `Bearer ${token}` }
|
||||
});
|
||||
if (!res.ok) {
|
||||
const data = await res.json();
|
||||
throw new Error(data.message);
|
||||
}
|
||||
onRefresh();
|
||||
} catch (err: any) {
|
||||
alert(err.message);
|
||||
}
|
||||
};
|
||||
|
||||
return (
|
||||
<div>
|
||||
<div className="flex justify-between items-center mb-6">
|
||||
<h2 className="text-lg font-bold text-slate-200 flex items-center gap-2">
|
||||
<span>🏪</span> Store Master
|
||||
</h2>
|
||||
<button
|
||||
onClick={() => setIsAdding(true)}
|
||||
className="bg-teal-600 hover:bg-teal-500 text-slate-950 px-4 py-2 rounded-xl text-xs font-bold transition-all duration-200 shadow-md shadow-teal-900/10 cursor-pointer"
|
||||
>
|
||||
+ Add Store
|
||||
</button>
|
||||
</div>
|
||||
|
||||
{isAdding && (
|
||||
<form onSubmit={handleSubmit} className="mb-6 bg-slate-950/40 p-5 rounded-xl border border-slate-800/60">
|
||||
<h3 className="text-xs font-bold text-slate-450 uppercase tracking-wider mb-4">
|
||||
Add New Store <span className="text-[10px] text-teal-500 font-normal lowercase">(Will auto-generate account with "123" password)</span>
|
||||
</h3>
|
||||
<div className="grid grid-cols-1 md:grid-cols-3 gap-4 mb-4">
|
||||
<input
|
||||
placeholder="Kode Toko"
|
||||
className="bg-slate-900/60 border border-slate-800/80 text-slate-100 placeholder-slate-600 rounded-lg p-2.5 text-xs focus:outline-none focus:border-teal-500/60"
|
||||
value={form.kode_toko}
|
||||
onChange={e => setForm({...form, kode_toko: e.target.value})}
|
||||
required
|
||||
/>
|
||||
<input
|
||||
placeholder="Nama Toko"
|
||||
className="bg-slate-900/60 border border-slate-800/80 text-slate-100 placeholder-slate-600 rounded-lg p-2.5 text-xs focus:outline-none focus:border-teal-500/60"
|
||||
value={form.nama_toko}
|
||||
onChange={e => setForm({...form, nama_toko: e.target.value})}
|
||||
required
|
||||
/>
|
||||
<input
|
||||
placeholder="Alamat"
|
||||
className="bg-slate-900/60 border border-slate-800/80 text-slate-100 placeholder-slate-600 rounded-lg p-2.5 text-xs focus:outline-none focus:border-teal-500/60"
|
||||
value={form.alamat}
|
||||
onChange={e => setForm({...form, alamat: e.target.value})}
|
||||
/>
|
||||
</div>
|
||||
{error && <div className="text-rose-450 text-xs mb-3 font-semibold">⚠️ {error}</div>}
|
||||
<div className="flex space-x-2">
|
||||
<button type="submit" className="bg-emerald-600 hover:bg-emerald-500 text-slate-950 font-bold px-4 py-2 rounded-lg text-xs transition-colors cursor-pointer">Save</button>
|
||||
<button type="button" onClick={() => setIsAdding(false)} className="bg-slate-800 hover:bg-slate-750 text-slate-300 px-4 py-2 rounded-lg text-xs transition-colors cursor-pointer">Cancel</button>
|
||||
</div>
|
||||
</form>
|
||||
)}
|
||||
|
||||
<div className="overflow-x-auto rounded-xl border border-slate-800/60">
|
||||
<table className="w-full text-left text-xs border-collapse">
|
||||
<thead className="bg-slate-950/60 border-b border-slate-800/80 text-slate-400">
|
||||
<tr>
|
||||
<th className="p-3.5 font-bold font-mono text-[10px] tracking-wider uppercase">Kode Toko</th>
|
||||
<th className="p-3.5 font-bold font-mono text-[10px] tracking-wider uppercase">Nama Toko</th>
|
||||
<th className="p-3.5 font-bold font-mono text-[10px] tracking-wider uppercase">Alamat</th>
|
||||
<th className="p-3.5 font-bold font-mono text-[10px] tracking-wider uppercase w-24">Actions</th>
|
||||
</tr>
|
||||
</thead>
|
||||
<tbody>
|
||||
{stores.map(s => (
|
||||
<tr key={s.kode_toko} className="border-b border-slate-850/50 hover:bg-slate-900/30 transition-colors">
|
||||
<td className="p-3.5 font-semibold text-slate-200 font-mono">{s.kode_toko}</td>
|
||||
<td className="p-3.5 text-slate-300 font-medium">{s.nama_toko}</td>
|
||||
<td className="p-3.5 text-slate-400 truncate max-w-xs">{s.alamat}</td>
|
||||
<td className="p-3.5">
|
||||
<button
|
||||
onClick={() => handleDelete(s.kode_toko)}
|
||||
className="text-rose-400 hover:text-rose-355 transition-colors font-bold cursor-pointer font-mono"
|
||||
>
|
||||
Delete
|
||||
</button>
|
||||
</td>
|
||||
</tr>
|
||||
))}
|
||||
{stores.length === 0 && (
|
||||
<tr><td colSpan={4} className="p-6 text-center text-slate-500">No stores found.</td></tr>
|
||||
)}
|
||||
</tbody>
|
||||
</table>
|
||||
</div>
|
||||
</div>
|
||||
);
|
||||
}
|
||||
|
||||
function SkuManager({ skus, token, onRefresh }: { skus: any[], token: string, onRefresh: () => void }) {
|
||||
const [isAdding, setIsAdding] = useState(false);
|
||||
const [form, setForm] = useState({ no_sku: "", nama_item: "", jenis_outer: "", standar_jumlah: "1" });
|
||||
const [error, setError] = useState("");
|
||||
|
||||
const handleSubmit = async (e: React.FormEvent) => {
|
||||
e.preventDefault();
|
||||
setError("");
|
||||
try {
|
||||
const payload = { ...form, standar_jumlah: parseInt(form.standar_jumlah) || 1 };
|
||||
const res = await fetch("/api/v1/master/skus", {
|
||||
method: "POST",
|
||||
headers: { "Content-Type": "application/json", "Authorization": `Bearer ${token}` },
|
||||
body: JSON.stringify(payload)
|
||||
});
|
||||
const data = await res.json();
|
||||
if (!res.ok) throw new Error(data.message);
|
||||
setIsAdding(false);
|
||||
setForm({ no_sku: "", nama_item: "", jenis_outer: "", standar_jumlah: "1" });
|
||||
onRefresh();
|
||||
} catch (err: any) {
|
||||
setError(err.message);
|
||||
}
|
||||
};
|
||||
|
||||
const handleDelete = async (kode: string) => {
|
||||
if (!confirm(`Delete SKU ${kode}?`)) return;
|
||||
try {
|
||||
const res = await fetch(`/api/v1/master/skus/${kode}`, {
|
||||
method: "DELETE",
|
||||
headers: { "Authorization": `Bearer ${token}` }
|
||||
});
|
||||
if (!res.ok) {
|
||||
const data = await res.json();
|
||||
throw new Error(data.message);
|
||||
}
|
||||
onRefresh();
|
||||
} catch (err: any) {
|
||||
alert(err.message);
|
||||
}
|
||||
};
|
||||
|
||||
return (
|
||||
<div>
|
||||
<div className="flex justify-between items-center mb-6">
|
||||
<h2 className="text-lg font-bold text-slate-200 flex items-center gap-2">
|
||||
<span>📦</span> SKU Master
|
||||
</h2>
|
||||
<button
|
||||
onClick={() => setIsAdding(true)}
|
||||
className="bg-teal-600 hover:bg-teal-500 text-slate-950 px-4 py-2 rounded-xl text-xs font-bold transition-all duration-200 shadow-md shadow-teal-900/10 cursor-pointer"
|
||||
>
|
||||
+ Add SKU
|
||||
</button>
|
||||
</div>
|
||||
|
||||
{isAdding && (
|
||||
<form onSubmit={handleSubmit} className="mb-6 bg-slate-950/40 p-5 rounded-xl border border-slate-800/60">
|
||||
<h3 className="text-xs font-bold text-slate-450 uppercase tracking-wider mb-4 font-mono">Add New SKU</h3>
|
||||
<div className="grid grid-cols-2 md:grid-cols-4 gap-4 mb-4">
|
||||
<input
|
||||
placeholder="No SKU"
|
||||
className="bg-slate-900/60 border border-slate-800/80 text-slate-100 placeholder-slate-600 rounded-lg p-2.5 text-xs focus:outline-none focus:border-teal-500/60"
|
||||
value={form.no_sku}
|
||||
onChange={e => setForm({...form, no_sku: e.target.value})}
|
||||
required
|
||||
/>
|
||||
<input
|
||||
placeholder="Nama Item"
|
||||
className="bg-slate-900/60 border border-slate-800/80 text-slate-100 placeholder-slate-600 rounded-lg p-2.5 text-xs focus:outline-none focus:border-teal-500/60"
|
||||
value={form.nama_item}
|
||||
onChange={e => setForm({...form, nama_item: e.target.value})}
|
||||
required
|
||||
/>
|
||||
<input
|
||||
placeholder="Jenis Outer (e.g. DUS)"
|
||||
className="bg-slate-900/60 border border-slate-800/80 text-slate-100 placeholder-slate-600 rounded-lg p-2.5 text-xs focus:outline-none focus:border-teal-500/60"
|
||||
value={form.jenis_outer}
|
||||
onChange={e => setForm({...form, jenis_outer: e.target.value})}
|
||||
/>
|
||||
<input
|
||||
type="number"
|
||||
placeholder="Std Qty"
|
||||
className="bg-slate-900/60 border border-slate-800/80 text-slate-100 placeholder-slate-600 rounded-lg p-2.5 text-xs focus:outline-none focus:border-teal-500/60"
|
||||
value={form.standar_jumlah}
|
||||
onChange={e => setForm({...form, standar_jumlah: e.target.value})}
|
||||
/>
|
||||
</div>
|
||||
{error && <div className="text-rose-450 text-xs mb-3 font-semibold">⚠️ {error}</div>}
|
||||
<div className="flex space-x-2">
|
||||
<button type="submit" className="bg-emerald-600 hover:bg-emerald-500 text-slate-950 font-bold px-4 py-2 rounded-lg text-xs transition-colors cursor-pointer">Save</button>
|
||||
<button type="button" onClick={() => setIsAdding(false)} className="bg-slate-800 hover:bg-slate-750 text-slate-300 px-4 py-2 rounded-lg text-xs transition-colors cursor-pointer">Cancel</button>
|
||||
</div>
|
||||
</form>
|
||||
)}
|
||||
|
||||
<div className="overflow-x-auto rounded-xl border border-slate-800/60">
|
||||
<table className="w-full text-left text-xs border-collapse">
|
||||
<thead className="bg-slate-950/60 border-b border-slate-800/80 text-slate-400">
|
||||
<tr>
|
||||
<th className="p-3.5 font-bold font-mono text-[10px] tracking-wider uppercase">No SKU</th>
|
||||
<th className="p-3.5 font-bold font-mono text-[10px] tracking-wider uppercase">Nama Item</th>
|
||||
<th className="p-3.5 font-bold font-mono text-[10px] tracking-wider uppercase">Outer</th>
|
||||
<th className="p-3.5 font-bold font-mono text-[10px] tracking-wider uppercase">Std Qty</th>
|
||||
<th className="p-3.5 font-bold font-mono text-[10px] tracking-wider uppercase w-24">Actions</th>
|
||||
</tr>
|
||||
</thead>
|
||||
<tbody>
|
||||
{skus.map(s => (
|
||||
<tr key={s.no_sku} className="border-b border-slate-850/50 hover:bg-slate-900/30 transition-colors">
|
||||
<td className="p-3.5 font-semibold text-slate-200 font-mono">{s.no_sku}</td>
|
||||
<td className="p-3.5 text-slate-300 font-medium">{s.nama_item}</td>
|
||||
<td className="p-3.5 text-slate-400 font-mono">{s.jenis_outer}</td>
|
||||
<td className="p-3.5 text-slate-400 font-mono">{s.standar_jumlah}</td>
|
||||
<td className="p-3.5">
|
||||
<button
|
||||
onClick={() => handleDelete(s.no_sku)}
|
||||
className="text-rose-400 hover:text-rose-350 transition-colors font-bold cursor-pointer font-mono"
|
||||
>
|
||||
Delete
|
||||
</button>
|
||||
</td>
|
||||
</tr>
|
||||
))}
|
||||
{skus.length === 0 && (
|
||||
<tr><td colSpan={5} className="p-6 text-center text-slate-500">No SKUs found.</td></tr>
|
||||
)}
|
||||
</tbody>
|
||||
</table>
|
||||
</div>
|
||||
</div>
|
||||
);
|
||||
}
|
||||
"use client";
|
||||
|
||||
import React, { useState, useEffect } from "react";
|
||||
|
||||
export default function MasterDataPage() {
|
||||
const [token, setToken] = useState<string | null>(null);
|
||||
const [username, setUsername] = useState("");
|
||||
const [password, setPassword] = useState("");
|
||||
const [loginError, setLoginError] = useState("");
|
||||
|
||||
const [activeTab, setActiveTab] = useState<"stores" | "skus">("stores");
|
||||
const [stores, setStores] = useState<any[]>([]);
|
||||
const [skus, setSkus] = useState<any[]>([]);
|
||||
|
||||
useEffect(() => {
|
||||
const savedToken = localStorage.getItem("adminToken");
|
||||
if (savedToken) {
|
||||
setToken(savedToken);
|
||||
fetchData(savedToken, activeTab);
|
||||
}
|
||||
}, [activeTab]);
|
||||
|
||||
const handleLogin = async (e: React.FormEvent) => {
|
||||
e.preventDefault();
|
||||
setLoginError("");
|
||||
try {
|
||||
const res = await fetch("/api/v1/auth/login", {
|
||||
method: "POST",
|
||||
headers: { "Content-Type": "application/json" },
|
||||
body: JSON.stringify({ username, password })
|
||||
});
|
||||
const data = await res.json();
|
||||
if (!res.ok) throw new Error(data.message || "Login failed");
|
||||
|
||||
const tokenStr = data.data?.token || data.token;
|
||||
localStorage.setItem("adminToken", tokenStr);
|
||||
setToken(tokenStr);
|
||||
fetchData(tokenStr, activeTab);
|
||||
} catch (err: any) {
|
||||
setLoginError(err.message);
|
||||
}
|
||||
};
|
||||
|
||||
const handleLogout = () => {
|
||||
localStorage.removeItem("adminToken");
|
||||
setToken(null);
|
||||
};
|
||||
|
||||
const fetchData = async (authToken: string, tab: "stores" | "skus") => {
|
||||
try {
|
||||
const res = await fetch(`/api/v1/master/${tab}`, {
|
||||
headers: { "Authorization": `Bearer ${authToken}` }
|
||||
});
|
||||
if (res.status === 401 || res.status === 403) {
|
||||
handleLogout();
|
||||
return;
|
||||
}
|
||||
const data = await res.json();
|
||||
if (res.ok) {
|
||||
if (tab === "stores") setStores(data.data || []);
|
||||
else setSkus(data.data || []);
|
||||
}
|
||||
} catch (err) {
|
||||
console.error(err);
|
||||
}
|
||||
};
|
||||
|
||||
if (!token) {
|
||||
return (
|
||||
<div className="min-h-screen bg-[radial-gradient(ellipse_at_top,_var(--tw-gradient-stops))] from-slate-900 via-slate-950 to-slate-950 flex items-center justify-center p-4">
|
||||
<div className="bg-slate-900/40 border border-slate-800/80 shadow-2xl backdrop-blur-md rounded-2xl p-8 w-full max-w-md">
|
||||
<h1 className="text-2xl font-extrabold bg-gradient-to-r from-teal-400 to-emerald-400 bg-clip-text text-transparent text-center mb-6">
|
||||
Admin Login
|
||||
</h1>
|
||||
<form onSubmit={handleLogin} className="space-y-5">
|
||||
<div>
|
||||
<label className="block text-xs font-semibold text-slate-400 uppercase tracking-wider mb-2">Username</label>
|
||||
<input
|
||||
type="text"
|
||||
className="w-full bg-slate-950/60 border border-slate-800 text-slate-100 placeholder-slate-600 rounded-xl p-3 focus:outline-none focus:border-teal-500/60 focus:ring-1 focus:ring-teal-500/60 transition-all duration-200 text-sm"
|
||||
value={username}
|
||||
onChange={e => setUsername(e.target.value)}
|
||||
placeholder="Enter admin username"
|
||||
/>
|
||||
</div>
|
||||
<div>
|
||||
<label className="block text-xs font-semibold text-slate-400 uppercase tracking-wider mb-2">Password</label>
|
||||
<input
|
||||
type="password"
|
||||
className="w-full bg-slate-950/60 border border-slate-800 text-slate-100 placeholder-slate-650 rounded-xl p-3 focus:outline-none focus:border-teal-500/60 focus:ring-1 focus:ring-teal-500/60 transition-all duration-200 text-sm"
|
||||
value={password}
|
||||
onChange={e => setPassword(e.target.value)}
|
||||
placeholder="••••••••"
|
||||
/>
|
||||
</div>
|
||||
{loginError && (
|
||||
<div className="bg-rose-950/30 border border-rose-800/40 p-3 rounded-xl text-xs text-rose-450 flex items-center gap-2">
|
||||
<span>⚠️</span>
|
||||
<span>{loginError}</span>
|
||||
</div>
|
||||
)}
|
||||
<button
|
||||
type="submit"
|
||||
className="w-full bg-teal-600 hover:bg-teal-500 text-slate-950 font-bold p-3 rounded-xl transition-all duration-200 shadow-lg shadow-teal-900/20 text-sm cursor-pointer"
|
||||
>
|
||||
Log In
|
||||
</button>
|
||||
</form>
|
||||
</div>
|
||||
</div>
|
||||
);
|
||||
}
|
||||
|
||||
return (
|
||||
<div className="min-h-screen bg-[radial-gradient(ellipse_at_top,_var(--tw-gradient-stops))] from-slate-900 via-slate-950 to-slate-950 p-8 text-slate-100">
|
||||
<div className="max-w-6xl mx-auto">
|
||||
<div className="flex justify-between items-center mb-8 border-b border-slate-800/60 pb-4">
|
||||
<h1 className="text-2xl font-extrabold bg-gradient-to-r from-teal-400 to-emerald-400 bg-clip-text text-transparent flex items-center gap-2">
|
||||
<span>⚙️</span> Master Data Management
|
||||
</h1>
|
||||
<button
|
||||
onClick={handleLogout}
|
||||
className="text-slate-400 hover:text-slate-100 bg-slate-900/60 hover:bg-slate-900 border border-slate-850 px-4 py-2 rounded-xl text-xs font-semibold transition-all duration-200 cursor-pointer"
|
||||
>
|
||||
Logout
|
||||
</button>
|
||||
</div>
|
||||
|
||||
<div className="flex space-x-2 mb-6 border-b border-slate-800/60 pb-px">
|
||||
<button
|
||||
className={`pb-2.5 px-4 text-sm font-semibold transition-all border-b-2 -mb-px cursor-pointer ${
|
||||
activeTab === 'stores'
|
||||
? 'border-teal-500 text-teal-400'
|
||||
: 'border-transparent text-slate-400 hover:text-slate-200'
|
||||
}`}
|
||||
onClick={() => setActiveTab('stores')}
|
||||
>
|
||||
Stores
|
||||
</button>
|
||||
<button
|
||||
className={`pb-2.5 px-4 text-sm font-semibold transition-all border-b-2 -mb-px cursor-pointer ${
|
||||
activeTab === 'skus'
|
||||
? 'border-teal-500 text-teal-400'
|
||||
: 'border-transparent text-slate-400 hover:text-slate-200'
|
||||
}`}
|
||||
onClick={() => setActiveTab('skus')}
|
||||
>
|
||||
SKUs
|
||||
</button>
|
||||
</div>
|
||||
|
||||
<div className="bg-slate-900/40 border border-slate-850 rounded-2xl p-6 shadow-xl backdrop-blur-md">
|
||||
{activeTab === 'stores' && <StoreManager stores={stores} token={token} onRefresh={() => fetchData(token, 'stores')} />}
|
||||
{activeTab === 'skus' && <SkuManager skus={skus} token={token} onRefresh={() => fetchData(token, 'skus')} />}
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
);
|
||||
}
|
||||
|
||||
function StoreManager({ stores, token, onRefresh }: { stores: any[], token: string, onRefresh: () => void }) {
|
||||
const [isAdding, setIsAdding] = useState(false);
|
||||
const [form, setForm] = useState({ kode_toko: "", nama_toko: "", alamat: "" });
|
||||
const [error, setError] = useState("");
|
||||
|
||||
const handleSubmit = async (e: React.FormEvent) => {
|
||||
e.preventDefault();
|
||||
setError("");
|
||||
try {
|
||||
const res = await fetch("/api/v1/master/stores", {
|
||||
method: "POST",
|
||||
headers: { "Content-Type": "application/json", "Authorization": `Bearer ${token}` },
|
||||
body: JSON.stringify(form)
|
||||
});
|
||||
const data = await res.json();
|
||||
if (!res.ok) throw new Error(data.message);
|
||||
setIsAdding(false);
|
||||
setForm({ kode_toko: "", nama_toko: "", alamat: "" });
|
||||
onRefresh();
|
||||
} catch (err: any) {
|
||||
setError(err.message);
|
||||
}
|
||||
};
|
||||
|
||||
const handleDelete = async (kode: string) => {
|
||||
if (!confirm(`Delete store ${kode}?`)) return;
|
||||
try {
|
||||
const res = await fetch(`/api/v1/master/stores/${kode}`, {
|
||||
method: "DELETE",
|
||||
headers: { "Authorization": `Bearer ${token}` }
|
||||
});
|
||||
if (!res.ok) {
|
||||
const data = await res.json();
|
||||
throw new Error(data.message);
|
||||
}
|
||||
onRefresh();
|
||||
} catch (err: any) {
|
||||
alert(err.message);
|
||||
}
|
||||
};
|
||||
|
||||
return (
|
||||
<div>
|
||||
<div className="flex justify-between items-center mb-6">
|
||||
<h2 className="text-lg font-bold text-slate-200 flex items-center gap-2">
|
||||
<span>🏪</span> Store Master
|
||||
</h2>
|
||||
<button
|
||||
onClick={() => setIsAdding(true)}
|
||||
className="bg-teal-600 hover:bg-teal-500 text-slate-950 px-4 py-2 rounded-xl text-xs font-bold transition-all duration-200 shadow-md shadow-teal-900/10 cursor-pointer"
|
||||
>
|
||||
+ Add Store
|
||||
</button>
|
||||
</div>
|
||||
|
||||
{isAdding && (
|
||||
<form onSubmit={handleSubmit} className="mb-6 bg-slate-950/40 p-5 rounded-xl border border-slate-800/60">
|
||||
<h3 className="text-xs font-bold text-slate-450 uppercase tracking-wider mb-4">
|
||||
Add New Store <span className="text-[10px] text-teal-500 font-normal lowercase">(Will auto-generate account with "123" password)</span>
|
||||
</h3>
|
||||
<div className="grid grid-cols-1 md:grid-cols-3 gap-4 mb-4">
|
||||
<input
|
||||
placeholder="Kode Toko"
|
||||
className="bg-slate-900/60 border border-slate-800/80 text-slate-100 placeholder-slate-600 rounded-lg p-2.5 text-xs focus:outline-none focus:border-teal-500/60"
|
||||
value={form.kode_toko}
|
||||
onChange={e => setForm({...form, kode_toko: e.target.value})}
|
||||
required
|
||||
/>
|
||||
<input
|
||||
placeholder="Nama Toko"
|
||||
className="bg-slate-900/60 border border-slate-800/80 text-slate-100 placeholder-slate-600 rounded-lg p-2.5 text-xs focus:outline-none focus:border-teal-500/60"
|
||||
value={form.nama_toko}
|
||||
onChange={e => setForm({...form, nama_toko: e.target.value})}
|
||||
required
|
||||
/>
|
||||
<input
|
||||
placeholder="Alamat"
|
||||
className="bg-slate-900/60 border border-slate-800/80 text-slate-100 placeholder-slate-600 rounded-lg p-2.5 text-xs focus:outline-none focus:border-teal-500/60"
|
||||
value={form.alamat}
|
||||
onChange={e => setForm({...form, alamat: e.target.value})}
|
||||
/>
|
||||
</div>
|
||||
{error && <div className="text-rose-450 text-xs mb-3 font-semibold">⚠️ {error}</div>}
|
||||
<div className="flex space-x-2">
|
||||
<button type="submit" className="bg-emerald-600 hover:bg-emerald-500 text-slate-950 font-bold px-4 py-2 rounded-lg text-xs transition-colors cursor-pointer">Save</button>
|
||||
<button type="button" onClick={() => setIsAdding(false)} className="bg-slate-800 hover:bg-slate-750 text-slate-300 px-4 py-2 rounded-lg text-xs transition-colors cursor-pointer">Cancel</button>
|
||||
</div>
|
||||
</form>
|
||||
)}
|
||||
|
||||
<div className="overflow-x-auto rounded-xl border border-slate-800/60">
|
||||
<table className="w-full text-left text-xs border-collapse">
|
||||
<thead className="bg-slate-950/60 border-b border-slate-800/80 text-slate-400">
|
||||
<tr>
|
||||
<th className="p-3.5 font-bold font-mono text-[10px] tracking-wider uppercase">Kode Toko</th>
|
||||
<th className="p-3.5 font-bold font-mono text-[10px] tracking-wider uppercase">Nama Toko</th>
|
||||
<th className="p-3.5 font-bold font-mono text-[10px] tracking-wider uppercase">Alamat</th>
|
||||
<th className="p-3.5 font-bold font-mono text-[10px] tracking-wider uppercase w-24">Actions</th>
|
||||
</tr>
|
||||
</thead>
|
||||
<tbody>
|
||||
{stores.map(s => (
|
||||
<tr key={s.kode_toko} className="border-b border-slate-850/50 hover:bg-slate-900/30 transition-colors">
|
||||
<td className="p-3.5 font-semibold text-slate-200 font-mono">{s.kode_toko}</td>
|
||||
<td className="p-3.5 text-slate-300 font-medium">{s.nama_toko}</td>
|
||||
<td className="p-3.5 text-slate-400 truncate max-w-xs">{s.alamat}</td>
|
||||
<td className="p-3.5">
|
||||
<button
|
||||
onClick={() => handleDelete(s.kode_toko)}
|
||||
className="text-rose-400 hover:text-rose-355 transition-colors font-bold cursor-pointer font-mono"
|
||||
>
|
||||
Delete
|
||||
</button>
|
||||
</td>
|
||||
</tr>
|
||||
))}
|
||||
{stores.length === 0 && (
|
||||
<tr><td colSpan={4} className="p-6 text-center text-slate-500">No stores found.</td></tr>
|
||||
)}
|
||||
</tbody>
|
||||
</table>
|
||||
</div>
|
||||
</div>
|
||||
);
|
||||
}
|
||||
|
||||
function SkuManager({ skus, token, onRefresh }: { skus: any[], token: string, onRefresh: () => void }) {
|
||||
const [isAdding, setIsAdding] = useState(false);
|
||||
const [form, setForm] = useState({ no_sku: "", nama_item: "", jenis_outer: "", standar_jumlah: "1" });
|
||||
const [error, setError] = useState("");
|
||||
|
||||
const handleSubmit = async (e: React.FormEvent) => {
|
||||
e.preventDefault();
|
||||
setError("");
|
||||
try {
|
||||
const payload = { ...form, standar_jumlah: parseInt(form.standar_jumlah) || 1 };
|
||||
const res = await fetch("/api/v1/master/skus", {
|
||||
method: "POST",
|
||||
headers: { "Content-Type": "application/json", "Authorization": `Bearer ${token}` },
|
||||
body: JSON.stringify(payload)
|
||||
});
|
||||
const data = await res.json();
|
||||
if (!res.ok) throw new Error(data.message);
|
||||
setIsAdding(false);
|
||||
setForm({ no_sku: "", nama_item: "", jenis_outer: "", standar_jumlah: "1" });
|
||||
onRefresh();
|
||||
} catch (err: any) {
|
||||
setError(err.message);
|
||||
}
|
||||
};
|
||||
|
||||
const handleDelete = async (kode: string) => {
|
||||
if (!confirm(`Delete SKU ${kode}?`)) return;
|
||||
try {
|
||||
const res = await fetch(`/api/v1/master/skus/${kode}`, {
|
||||
method: "DELETE",
|
||||
headers: { "Authorization": `Bearer ${token}` }
|
||||
});
|
||||
if (!res.ok) {
|
||||
const data = await res.json();
|
||||
throw new Error(data.message);
|
||||
}
|
||||
onRefresh();
|
||||
} catch (err: any) {
|
||||
alert(err.message);
|
||||
}
|
||||
};
|
||||
|
||||
return (
|
||||
<div>
|
||||
<div className="flex justify-between items-center mb-6">
|
||||
<h2 className="text-lg font-bold text-slate-200 flex items-center gap-2">
|
||||
<span>📦</span> SKU Master
|
||||
</h2>
|
||||
<button
|
||||
onClick={() => setIsAdding(true)}
|
||||
className="bg-teal-600 hover:bg-teal-500 text-slate-950 px-4 py-2 rounded-xl text-xs font-bold transition-all duration-200 shadow-md shadow-teal-900/10 cursor-pointer"
|
||||
>
|
||||
+ Add SKU
|
||||
</button>
|
||||
</div>
|
||||
|
||||
{isAdding && (
|
||||
<form onSubmit={handleSubmit} className="mb-6 bg-slate-950/40 p-5 rounded-xl border border-slate-800/60">
|
||||
<h3 className="text-xs font-bold text-slate-450 uppercase tracking-wider mb-4 font-mono">Add New SKU</h3>
|
||||
<div className="grid grid-cols-2 md:grid-cols-4 gap-4 mb-4">
|
||||
<input
|
||||
placeholder="No SKU"
|
||||
className="bg-slate-900/60 border border-slate-800/80 text-slate-100 placeholder-slate-600 rounded-lg p-2.5 text-xs focus:outline-none focus:border-teal-500/60"
|
||||
value={form.no_sku}
|
||||
onChange={e => setForm({...form, no_sku: e.target.value})}
|
||||
required
|
||||
/>
|
||||
<input
|
||||
placeholder="Nama Item"
|
||||
className="bg-slate-900/60 border border-slate-800/80 text-slate-100 placeholder-slate-600 rounded-lg p-2.5 text-xs focus:outline-none focus:border-teal-500/60"
|
||||
value={form.nama_item}
|
||||
onChange={e => setForm({...form, nama_item: e.target.value})}
|
||||
required
|
||||
/>
|
||||
<input
|
||||
placeholder="Jenis Outer (e.g. DUS)"
|
||||
className="bg-slate-900/60 border border-slate-800/80 text-slate-100 placeholder-slate-600 rounded-lg p-2.5 text-xs focus:outline-none focus:border-teal-500/60"
|
||||
value={form.jenis_outer}
|
||||
onChange={e => setForm({...form, jenis_outer: e.target.value})}
|
||||
/>
|
||||
<input
|
||||
type="number"
|
||||
placeholder="Std Qty"
|
||||
className="bg-slate-900/60 border border-slate-800/80 text-slate-100 placeholder-slate-600 rounded-lg p-2.5 text-xs focus:outline-none focus:border-teal-500/60"
|
||||
value={form.standar_jumlah}
|
||||
onChange={e => setForm({...form, standar_jumlah: e.target.value})}
|
||||
/>
|
||||
</div>
|
||||
{error && <div className="text-rose-450 text-xs mb-3 font-semibold">⚠️ {error}</div>}
|
||||
<div className="flex space-x-2">
|
||||
<button type="submit" className="bg-emerald-600 hover:bg-emerald-500 text-slate-950 font-bold px-4 py-2 rounded-lg text-xs transition-colors cursor-pointer">Save</button>
|
||||
<button type="button" onClick={() => setIsAdding(false)} className="bg-slate-800 hover:bg-slate-750 text-slate-300 px-4 py-2 rounded-lg text-xs transition-colors cursor-pointer">Cancel</button>
|
||||
</div>
|
||||
</form>
|
||||
)}
|
||||
|
||||
<div className="overflow-x-auto rounded-xl border border-slate-800/60">
|
||||
<table className="w-full text-left text-xs border-collapse">
|
||||
<thead className="bg-slate-950/60 border-b border-slate-800/80 text-slate-400">
|
||||
<tr>
|
||||
<th className="p-3.5 font-bold font-mono text-[10px] tracking-wider uppercase">No SKU</th>
|
||||
<th className="p-3.5 font-bold font-mono text-[10px] tracking-wider uppercase">Nama Item</th>
|
||||
<th className="p-3.5 font-bold font-mono text-[10px] tracking-wider uppercase">Outer</th>
|
||||
<th className="p-3.5 font-bold font-mono text-[10px] tracking-wider uppercase">Std Qty</th>
|
||||
<th className="p-3.5 font-bold font-mono text-[10px] tracking-wider uppercase w-24">Actions</th>
|
||||
</tr>
|
||||
</thead>
|
||||
<tbody>
|
||||
{skus.map(s => (
|
||||
<tr key={s.no_sku} className="border-b border-slate-850/50 hover:bg-slate-900/30 transition-colors">
|
||||
<td className="p-3.5 font-semibold text-slate-200 font-mono">{s.no_sku}</td>
|
||||
<td className="p-3.5 text-slate-300 font-medium">{s.nama_item}</td>
|
||||
<td className="p-3.5 text-slate-400 font-mono">{s.jenis_outer}</td>
|
||||
<td className="p-3.5 text-slate-400 font-mono">{s.standar_jumlah}</td>
|
||||
<td className="p-3.5">
|
||||
<button
|
||||
onClick={() => handleDelete(s.no_sku)}
|
||||
className="text-rose-400 hover:text-rose-350 transition-colors font-bold cursor-pointer font-mono"
|
||||
>
|
||||
Delete
|
||||
</button>
|
||||
</td>
|
||||
</tr>
|
||||
))}
|
||||
{skus.length === 0 && (
|
||||
<tr><td colSpan={5} className="p-6 text-center text-slate-500">No SKUs found.</td></tr>
|
||||
)}
|
||||
</tbody>
|
||||
</table>
|
||||
</div>
|
||||
</div>
|
||||
);
|
||||
}
|
||||
@@ -1,265 +1,265 @@
|
||||
import { NextRequest, NextResponse } from "next/server";
|
||||
import fs from "fs";
|
||||
import path from "path";
|
||||
import { Client } from "@gradio/client";
|
||||
import { query } from "../../../db";
|
||||
import { errorResponse } from "@/utils/api-error";
|
||||
|
||||
export const maxDuration = 120; // Allow up to 120 seconds for slow model inference
|
||||
|
||||
export async function GET(req: NextRequest) {
|
||||
try {
|
||||
const { searchParams } = new URL(req.url);
|
||||
const action = searchParams.get("action") || "list";
|
||||
const runId = searchParams.get("runId");
|
||||
const imageType = searchParams.get("imageType"); // 'do', 'product' or null for all
|
||||
|
||||
if (runId) {
|
||||
const runRes = await query(`
|
||||
SELECT id, image_path, engine, status, ocr_result, time_elapsed_ms, image_type, created_at
|
||||
FROM arena_runs
|
||||
WHERE id = $1
|
||||
`, [parseInt(runId)]);
|
||||
|
||||
if (runRes.rowCount === 0) {
|
||||
return errorResponse(404, "Run not found");
|
||||
}
|
||||
return NextResponse.json({ success: true, run: runRes.rows[0] });
|
||||
}
|
||||
|
||||
if (action === "stats") {
|
||||
let queryText = `
|
||||
SELECT
|
||||
engine,
|
||||
COUNT(*)::integer as total_runs,
|
||||
COUNT(CASE WHEN status = 'done' THEN 1 END)::integer as success_runs,
|
||||
COUNT(CASE WHEN status = 'failed' THEN 1 END)::integer as failed_runs,
|
||||
ROUND(AVG(CASE WHEN status = 'done' THEN time_elapsed_ms END))::integer as avg_time_ms,
|
||||
MIN(CASE WHEN status = 'done' THEN time_elapsed_ms END)::integer as min_time_ms,
|
||||
MAX(CASE WHEN status = 'done' THEN time_elapsed_ms END)::integer as max_time_ms
|
||||
FROM arena_runs
|
||||
`;
|
||||
const params: any[] = [];
|
||||
if (imageType === "do" || imageType === "product") {
|
||||
queryText += ` WHERE image_type = $1`;
|
||||
params.push(imageType);
|
||||
}
|
||||
queryText += ` GROUP BY engine`;
|
||||
|
||||
const statsRes = await query(queryText, params);
|
||||
return NextResponse.json({ success: true, stats: statsRes.rows });
|
||||
}
|
||||
|
||||
const limit = parseInt(searchParams.get("limit") || "50");
|
||||
let queryText = `
|
||||
SELECT id, image_path, engine, status, time_elapsed_ms, image_type, created_at
|
||||
FROM arena_runs
|
||||
`;
|
||||
const params: any[] = [];
|
||||
if (imageType === "do" || imageType === "product") {
|
||||
queryText += ` WHERE image_type = $1`;
|
||||
params.push(imageType);
|
||||
}
|
||||
queryText += ` ORDER BY created_at DESC LIMIT $${params.length + 1}`;
|
||||
params.push(limit);
|
||||
|
||||
const runsRes = await query(queryText, params);
|
||||
return NextResponse.json({ success: true, runs: runsRes.rows });
|
||||
} catch (error: any) {
|
||||
console.error("Failed to fetch arena runs/stats:", error);
|
||||
return errorResponse(500, error.message);
|
||||
}
|
||||
}
|
||||
|
||||
export async function POST(req: NextRequest) {
|
||||
const startTime = Date.now();
|
||||
let engine: string | undefined;
|
||||
let image: string | undefined;
|
||||
let imageType = "do";
|
||||
try {
|
||||
const body = await req.json().catch(() => ({}));
|
||||
engine = body.engine;
|
||||
image = body.image;
|
||||
|
||||
if (!engine || !image) {
|
||||
return errorResponse(400, "Missing engine or image");
|
||||
}
|
||||
|
||||
imageType = body.imageType || "do";
|
||||
if (typeof image === "string") {
|
||||
if (image.startsWith("/produk-pfm/") || image.includes("produk-pfm") || image.includes("Product")) {
|
||||
imageType = "product";
|
||||
} else if (image.startsWith("/do-pfm/") || image.includes("do-pfm")) {
|
||||
imageType = "do";
|
||||
}
|
||||
}
|
||||
|
||||
let imageBuffer: Buffer;
|
||||
let base64Image = "";
|
||||
|
||||
// 1. Resolve image (local file or base64)
|
||||
if (typeof image === "string" && (image.startsWith("/do-pfm/") || image.startsWith("/produk-pfm/"))) {
|
||||
// Resolve path in public folder
|
||||
const cleanPath = image.startsWith("/") ? image.slice(1) : image;
|
||||
const filePath = path.join(process.cwd(), "public", cleanPath);
|
||||
|
||||
if (!fs.existsSync(filePath)) {
|
||||
return errorResponse(404, `File not found on server: ${image}`);
|
||||
}
|
||||
imageBuffer = fs.readFileSync(filePath);
|
||||
base64Image = `data:image/jpeg;base64,${imageBuffer.toString("base64")}`;
|
||||
} else if (typeof image === "string" && image.startsWith("data:")) {
|
||||
// Base64 data URI
|
||||
base64Image = image;
|
||||
const base64Data = image.split(",")[1];
|
||||
imageBuffer = Buffer.from(base64Data, "base64");
|
||||
} else if (typeof image === "string") {
|
||||
// Raw base64 string
|
||||
base64Image = `data:image/jpeg;base64,${image}`;
|
||||
imageBuffer = Buffer.from(image, "base64");
|
||||
} else {
|
||||
return errorResponse(400, "Invalid image format");
|
||||
}
|
||||
|
||||
let outputText = "";
|
||||
|
||||
// 2. Route to the requested OCR engine
|
||||
if (engine === "deepseek") {
|
||||
const blob = new Blob([new Uint8Array(imageBuffer)], { type: "image/jpeg" });
|
||||
const gradioUrl = process.env.DEEPSEEK_GRADIO_URL || "http://host.docker.internal:7873/v2/";
|
||||
const client = await Client.connect(gradioUrl);
|
||||
const result = await client.predict(2, [blob, "Default", "Markdown", ""]);
|
||||
const data = result.data as any[];
|
||||
outputText = data[1] || data[0] || "";
|
||||
|
||||
} else if (engine === "lightonocr") {
|
||||
const url = process.env.LIGHTONOCR_API_URL || "http://host.docker.internal:7678/layout-parsing";
|
||||
const res = await fetch(url, {
|
||||
method: "POST",
|
||||
headers: { "Content-Type": "application/json" },
|
||||
body: JSON.stringify({
|
||||
file: base64Image,
|
||||
useLayoutDetection: false
|
||||
})
|
||||
});
|
||||
if (!res.ok) {
|
||||
throw new Error(`LightOnOCR backend error: ${res.status} ${await res.text()}`);
|
||||
}
|
||||
const data = await res.json();
|
||||
outputText = data.result?.layoutParsingResults?.[0]?.markdown?.text || "";
|
||||
|
||||
} else if (engine === "nemotron") {
|
||||
const url = process.env.NEMOTRON_API_URL || "http://host.docker.internal:8009/layout-parsing";
|
||||
const res = await fetch(url, {
|
||||
method: "POST",
|
||||
headers: { "Content-Type": "application/json" },
|
||||
body: JSON.stringify({
|
||||
file: base64Image,
|
||||
model: "Multilingual (en, zh, ja, ko, ru, …)",
|
||||
merge_level: "layout"
|
||||
})
|
||||
});
|
||||
if (!res.ok) {
|
||||
throw new Error(`Nemotron backend error: ${res.status} ${await res.text()}`);
|
||||
}
|
||||
const data = await res.json();
|
||||
outputText = data.result?.layoutParsingResults?.[0]?.markdown?.text || "";
|
||||
|
||||
} else if (engine === "paddle") {
|
||||
const url = process.env.PIPELINE_URL || "http://paddleocr-pipeline-api:8090/layout-parsing";
|
||||
const rawB64 = base64Image.includes(",") ? base64Image.split(",")[1] : base64Image;
|
||||
const res = await fetch(url, {
|
||||
method: "POST",
|
||||
headers: { "Content-Type": "application/json" },
|
||||
body: JSON.stringify({
|
||||
file: rawB64,
|
||||
matchHistoryJob: false,
|
||||
useLayoutDetection: true,
|
||||
fileType: 1,
|
||||
useDocUnwarping: false,
|
||||
useDocOrientationClassify: false
|
||||
})
|
||||
});
|
||||
if (!res.ok) {
|
||||
throw new Error(`PaddleOCR backend error: ${res.status} ${await res.text()}`);
|
||||
}
|
||||
const data = await res.json();
|
||||
const pipelineResult = data.result || data;
|
||||
outputText = pipelineResult?.layoutParsingResults?.[0]?.markdown?.text || "";
|
||||
|
||||
} else if (engine === "dots") {
|
||||
// Calling python API directly
|
||||
const url = process.env.DOTS_API_URL || "http://host.docker.internal:7872/layout-parsing";
|
||||
const res = await fetch(url, {
|
||||
method: "POST",
|
||||
headers: { "Content-Type": "application/json" },
|
||||
body: JSON.stringify({
|
||||
file: base64Image,
|
||||
promptLabel: "ocr",
|
||||
useLayoutDetection: true
|
||||
})
|
||||
});
|
||||
if (!res.ok) {
|
||||
throw new Error(`Dots OCR backend error: ${res.status} ${await res.text()}`);
|
||||
}
|
||||
const data = await res.json();
|
||||
outputText = data.result?.layoutParsingResults?.[0]?.markdown?.text || "";
|
||||
|
||||
} else if (engine === "glm") {
|
||||
const gradioUrl = process.env.GLM_GRADIO_URL || "http://host.docker.internal:7875/";
|
||||
const client = await Client.connect(gradioUrl);
|
||||
const result = await client.predict(2, ["Text", base64Image, 1024, 60]);
|
||||
const data = result.data as any[];
|
||||
outputText = data[0] || "";
|
||||
|
||||
} else {
|
||||
return errorResponse(400, `Unknown engine: ${engine}`);
|
||||
}
|
||||
|
||||
const elapsedMs = Date.now() - startTime;
|
||||
|
||||
// Record successful run
|
||||
try {
|
||||
const loggedImagePath = (typeof image === "string" && image.startsWith("data:"))
|
||||
? `[Base64 Upload: ${image.length} chars]`
|
||||
: (typeof image === "string" && image.length > 500)
|
||||
? `[Raw Base64: ${image.length} chars]`
|
||||
: image;
|
||||
await query(
|
||||
`INSERT INTO arena_runs (image_path, engine, status, ocr_result, time_elapsed_ms, image_type)
|
||||
VALUES ($1, $2, $3, $4, $5, $6)`,
|
||||
[loggedImagePath, engine, "done", outputText, elapsedMs, imageType]
|
||||
);
|
||||
} catch (dbErr) {
|
||||
console.error("Failed to log success to arena_runs:", dbErr);
|
||||
}
|
||||
|
||||
return NextResponse.json({
|
||||
success: true,
|
||||
text: outputText,
|
||||
elapsedMs
|
||||
});
|
||||
|
||||
} catch (error: any) {
|
||||
console.error("OCR Arena proxy error:", error);
|
||||
const elapsedMs = Date.now() - startTime;
|
||||
|
||||
// Record failed run
|
||||
try {
|
||||
const loggedImagePath = (typeof image === "string" && image.startsWith("data:"))
|
||||
? `[Base64 Upload: ${image.length} chars]`
|
||||
: (typeof image === "string" && image.length > 500)
|
||||
? `[Raw Base64: ${image.length} chars]`
|
||||
: image;
|
||||
await query(
|
||||
`INSERT INTO arena_runs (image_path, engine, status, ocr_result, time_elapsed_ms, image_type)
|
||||
VALUES ($1, $2, $3, $4, $5, $6)`,
|
||||
[loggedImagePath || "unknown", engine || "unknown", "failed", error.message || "Unknown error", elapsedMs, imageType]
|
||||
);
|
||||
} catch (dbErr) {
|
||||
console.error("Failed to log failure to arena_runs:", dbErr);
|
||||
}
|
||||
|
||||
return errorResponse(500, error.message || "Failed to process OCR request");
|
||||
}
|
||||
}
|
||||
import { NextRequest, NextResponse } from "next/server";
|
||||
import fs from "fs";
|
||||
import path from "path";
|
||||
import { Client } from "@gradio/client";
|
||||
import { query } from "../../../db";
|
||||
import { errorResponse } from "@/utils/api-error";
|
||||
|
||||
export const maxDuration = 120; // Allow up to 120 seconds for slow model inference
|
||||
|
||||
export async function GET(req: NextRequest) {
|
||||
try {
|
||||
const { searchParams } = new URL(req.url);
|
||||
const action = searchParams.get("action") || "list";
|
||||
const runId = searchParams.get("runId");
|
||||
const imageType = searchParams.get("imageType"); // 'do', 'product' or null for all
|
||||
|
||||
if (runId) {
|
||||
const runRes = await query(`
|
||||
SELECT id, image_path, engine, status, ocr_result, time_elapsed_ms, image_type, created_at
|
||||
FROM arena_runs
|
||||
WHERE id = $1
|
||||
`, [parseInt(runId)]);
|
||||
|
||||
if (runRes.rowCount === 0) {
|
||||
return errorResponse(404, "Run not found");
|
||||
}
|
||||
return NextResponse.json({ success: true, run: runRes.rows[0] });
|
||||
}
|
||||
|
||||
if (action === "stats") {
|
||||
let queryText = `
|
||||
SELECT
|
||||
engine,
|
||||
COUNT(*)::integer as total_runs,
|
||||
COUNT(CASE WHEN status = 'done' THEN 1 END)::integer as success_runs,
|
||||
COUNT(CASE WHEN status = 'failed' THEN 1 END)::integer as failed_runs,
|
||||
ROUND(AVG(CASE WHEN status = 'done' THEN time_elapsed_ms END))::integer as avg_time_ms,
|
||||
MIN(CASE WHEN status = 'done' THEN time_elapsed_ms END)::integer as min_time_ms,
|
||||
MAX(CASE WHEN status = 'done' THEN time_elapsed_ms END)::integer as max_time_ms
|
||||
FROM arena_runs
|
||||
`;
|
||||
const params: any[] = [];
|
||||
if (imageType === "do" || imageType === "product") {
|
||||
queryText += ` WHERE image_type = $1`;
|
||||
params.push(imageType);
|
||||
}
|
||||
queryText += ` GROUP BY engine`;
|
||||
|
||||
const statsRes = await query(queryText, params);
|
||||
return NextResponse.json({ success: true, stats: statsRes.rows });
|
||||
}
|
||||
|
||||
const limit = parseInt(searchParams.get("limit") || "50");
|
||||
let queryText = `
|
||||
SELECT id, image_path, engine, status, time_elapsed_ms, image_type, created_at
|
||||
FROM arena_runs
|
||||
`;
|
||||
const params: any[] = [];
|
||||
if (imageType === "do" || imageType === "product") {
|
||||
queryText += ` WHERE image_type = $1`;
|
||||
params.push(imageType);
|
||||
}
|
||||
queryText += ` ORDER BY created_at DESC LIMIT $${params.length + 1}`;
|
||||
params.push(limit);
|
||||
|
||||
const runsRes = await query(queryText, params);
|
||||
return NextResponse.json({ success: true, runs: runsRes.rows });
|
||||
} catch (error: any) {
|
||||
console.error("Failed to fetch arena runs/stats:", error);
|
||||
return errorResponse(500, error.message);
|
||||
}
|
||||
}
|
||||
|
||||
export async function POST(req: NextRequest) {
|
||||
const startTime = Date.now();
|
||||
let engine: string | undefined;
|
||||
let image: string | undefined;
|
||||
let imageType = "do";
|
||||
try {
|
||||
const body = await req.json().catch(() => ({}));
|
||||
engine = body.engine;
|
||||
image = body.image;
|
||||
|
||||
if (!engine || !image) {
|
||||
return errorResponse(400, "Missing engine or image");
|
||||
}
|
||||
|
||||
imageType = body.imageType || "do";
|
||||
if (typeof image === "string") {
|
||||
if (image.startsWith("/produk-pfm/") || image.includes("produk-pfm") || image.includes("Product")) {
|
||||
imageType = "product";
|
||||
} else if (image.startsWith("/do-pfm/") || image.includes("do-pfm")) {
|
||||
imageType = "do";
|
||||
}
|
||||
}
|
||||
|
||||
let imageBuffer: Buffer;
|
||||
let base64Image = "";
|
||||
|
||||
// 1. Resolve image (local file or base64)
|
||||
if (typeof image === "string" && (image.startsWith("/do-pfm/") || image.startsWith("/produk-pfm/"))) {
|
||||
// Resolve path in public folder
|
||||
const cleanPath = image.startsWith("/") ? image.slice(1) : image;
|
||||
const filePath = path.join(process.cwd(), "public", cleanPath);
|
||||
|
||||
if (!fs.existsSync(filePath)) {
|
||||
return errorResponse(404, `File not found on server: ${image}`);
|
||||
}
|
||||
imageBuffer = fs.readFileSync(filePath);
|
||||
base64Image = `data:image/jpeg;base64,${imageBuffer.toString("base64")}`;
|
||||
} else if (typeof image === "string" && image.startsWith("data:")) {
|
||||
// Base64 data URI
|
||||
base64Image = image;
|
||||
const base64Data = image.split(",")[1];
|
||||
imageBuffer = Buffer.from(base64Data, "base64");
|
||||
} else if (typeof image === "string") {
|
||||
// Raw base64 string
|
||||
base64Image = `data:image/jpeg;base64,${image}`;
|
||||
imageBuffer = Buffer.from(image, "base64");
|
||||
} else {
|
||||
return errorResponse(400, "Invalid image format");
|
||||
}
|
||||
|
||||
let outputText = "";
|
||||
|
||||
// 2. Route to the requested OCR engine
|
||||
if (engine === "deepseek") {
|
||||
const blob = new Blob([new Uint8Array(imageBuffer)], { type: "image/jpeg" });
|
||||
const gradioUrl = process.env.DEEPSEEK_GRADIO_URL || "http://host.docker.internal:7873/v2/";
|
||||
const client = await Client.connect(gradioUrl);
|
||||
const result = await client.predict(2, [blob, "Default", "Markdown", ""]);
|
||||
const data = result.data as any[];
|
||||
outputText = data[1] || data[0] || "";
|
||||
|
||||
} else if (engine === "lightonocr") {
|
||||
const url = process.env.LIGHTONOCR_API_URL || "http://host.docker.internal:7678/layout-parsing";
|
||||
const res = await fetch(url, {
|
||||
method: "POST",
|
||||
headers: { "Content-Type": "application/json" },
|
||||
body: JSON.stringify({
|
||||
file: base64Image,
|
||||
useLayoutDetection: false
|
||||
})
|
||||
});
|
||||
if (!res.ok) {
|
||||
throw new Error(`LightOnOCR backend error: ${res.status} ${await res.text()}`);
|
||||
}
|
||||
const data = await res.json();
|
||||
outputText = data.result?.layoutParsingResults?.[0]?.markdown?.text || "";
|
||||
|
||||
} else if (engine === "nemotron") {
|
||||
const url = process.env.NEMOTRON_API_URL || "http://host.docker.internal:8009/layout-parsing";
|
||||
const res = await fetch(url, {
|
||||
method: "POST",
|
||||
headers: { "Content-Type": "application/json" },
|
||||
body: JSON.stringify({
|
||||
file: base64Image,
|
||||
model: "Multilingual (en, zh, ja, ko, ru, …)",
|
||||
merge_level: "layout"
|
||||
})
|
||||
});
|
||||
if (!res.ok) {
|
||||
throw new Error(`Nemotron backend error: ${res.status} ${await res.text()}`);
|
||||
}
|
||||
const data = await res.json();
|
||||
outputText = data.result?.layoutParsingResults?.[0]?.markdown?.text || "";
|
||||
|
||||
} else if (engine === "paddle") {
|
||||
const url = process.env.PIPELINE_URL || "http://paddleocr-pipeline-api:8090/layout-parsing";
|
||||
const rawB64 = base64Image.includes(",") ? base64Image.split(",")[1] : base64Image;
|
||||
const res = await fetch(url, {
|
||||
method: "POST",
|
||||
headers: { "Content-Type": "application/json" },
|
||||
body: JSON.stringify({
|
||||
file: rawB64,
|
||||
matchHistoryJob: false,
|
||||
useLayoutDetection: true,
|
||||
fileType: 1,
|
||||
useDocUnwarping: false,
|
||||
useDocOrientationClassify: false
|
||||
})
|
||||
});
|
||||
if (!res.ok) {
|
||||
throw new Error(`PaddleOCR backend error: ${res.status} ${await res.text()}`);
|
||||
}
|
||||
const data = await res.json();
|
||||
const pipelineResult = data.result || data;
|
||||
outputText = pipelineResult?.layoutParsingResults?.[0]?.markdown?.text || "";
|
||||
|
||||
} else if (engine === "dots") {
|
||||
// Calling python API directly
|
||||
const url = process.env.DOTS_API_URL || "http://host.docker.internal:7872/layout-parsing";
|
||||
const res = await fetch(url, {
|
||||
method: "POST",
|
||||
headers: { "Content-Type": "application/json" },
|
||||
body: JSON.stringify({
|
||||
file: base64Image,
|
||||
promptLabel: "ocr",
|
||||
useLayoutDetection: true
|
||||
})
|
||||
});
|
||||
if (!res.ok) {
|
||||
throw new Error(`Dots OCR backend error: ${res.status} ${await res.text()}`);
|
||||
}
|
||||
const data = await res.json();
|
||||
outputText = data.result?.layoutParsingResults?.[0]?.markdown?.text || "";
|
||||
|
||||
} else if (engine === "glm") {
|
||||
const gradioUrl = process.env.GLM_GRADIO_URL || "http://host.docker.internal:7875/";
|
||||
const client = await Client.connect(gradioUrl);
|
||||
const result = await client.predict(2, ["Text", base64Image, 1024, 60]);
|
||||
const data = result.data as any[];
|
||||
outputText = data[0] || "";
|
||||
|
||||
} else {
|
||||
return errorResponse(400, `Unknown engine: ${engine}`);
|
||||
}
|
||||
|
||||
const elapsedMs = Date.now() - startTime;
|
||||
|
||||
// Record successful run
|
||||
try {
|
||||
const loggedImagePath = (typeof image === "string" && image.startsWith("data:"))
|
||||
? `[Base64 Upload: ${image.length} chars]`
|
||||
: (typeof image === "string" && image.length > 500)
|
||||
? `[Raw Base64: ${image.length} chars]`
|
||||
: image;
|
||||
await query(
|
||||
`INSERT INTO arena_runs (image_path, engine, status, ocr_result, time_elapsed_ms, image_type)
|
||||
VALUES ($1, $2, $3, $4, $5, $6)`,
|
||||
[loggedImagePath, engine, "done", outputText, elapsedMs, imageType]
|
||||
);
|
||||
} catch (dbErr) {
|
||||
console.error("Failed to log success to arena_runs:", dbErr);
|
||||
}
|
||||
|
||||
return NextResponse.json({
|
||||
success: true,
|
||||
text: outputText,
|
||||
elapsedMs
|
||||
});
|
||||
|
||||
} catch (error: any) {
|
||||
console.error("OCR Arena proxy error:", error);
|
||||
const elapsedMs = Date.now() - startTime;
|
||||
|
||||
// Record failed run
|
||||
try {
|
||||
const loggedImagePath = (typeof image === "string" && image.startsWith("data:"))
|
||||
? `[Base64 Upload: ${image.length} chars]`
|
||||
: (typeof image === "string" && image.length > 500)
|
||||
? `[Raw Base64: ${image.length} chars]`
|
||||
: image;
|
||||
await query(
|
||||
`INSERT INTO arena_runs (image_path, engine, status, ocr_result, time_elapsed_ms, image_type)
|
||||
VALUES ($1, $2, $3, $4, $5, $6)`,
|
||||
[loggedImagePath || "unknown", engine || "unknown", "failed", error.message || "Unknown error", elapsedMs, imageType]
|
||||
);
|
||||
} catch (dbErr) {
|
||||
console.error("Failed to log failure to arena_runs:", dbErr);
|
||||
}
|
||||
|
||||
return errorResponse(500, error.message || "Failed to process OCR request");
|
||||
}
|
||||
}
|
||||
@@ -1,56 +1,56 @@
|
||||
import { NextRequest, NextResponse } from "next/server";
|
||||
import fs from "fs";
|
||||
import path from "path";
|
||||
import { query } from "../../../db";
|
||||
import crypto from "crypto";
|
||||
import { errorResponse } from "@/utils/api-error";
|
||||
|
||||
const UPLOADS_DIR = "/uploads";
|
||||
const PUBLIC_DIR = path.join(process.cwd(), "public/do-pfm");
|
||||
|
||||
export async function POST(req: NextRequest) {
|
||||
try {
|
||||
const { filename, image } = await req.json();
|
||||
|
||||
if (!filename || !image) {
|
||||
return errorResponse(400, "Filename and image base64 data are required");
|
||||
}
|
||||
|
||||
const safeFile = path.basename(filename);
|
||||
|
||||
const isSample = fs.existsSync(path.join(PUBLIC_DIR, safeFile));
|
||||
const filePath = isSample
|
||||
? path.join(PUBLIC_DIR, safeFile)
|
||||
: path.join(UPLOADS_DIR, safeFile);
|
||||
|
||||
const base64Data = image.replace(/^data:image\/\w+;base64,/, "");
|
||||
const buffer = Buffer.from(base64Data, "base64");
|
||||
|
||||
// Write file to disk
|
||||
fs.writeFileSync(filePath, buffer);
|
||||
console.log(`Cropped file saved successfully at ${filePath}`);
|
||||
|
||||
// Update database fields
|
||||
const fileHash = crypto.createHash("sha256").update(buffer).digest("hex");
|
||||
const stats = fs.statSync(filePath);
|
||||
|
||||
// Update document to unparsed state since layout changes
|
||||
await query(
|
||||
"UPDATE documents SET size = $1, file_hash = $2, parsed = false, layout_parsing_result = NULL WHERE filename = $3",
|
||||
[stats.size, fileHash, filename]
|
||||
);
|
||||
|
||||
// Clear old items for this document
|
||||
const docRes = await query("SELECT id FROM documents WHERE filename = $1", [filename]);
|
||||
if (docRes.rowCount && docRes.rowCount > 0) {
|
||||
const docId = docRes.rows[0].id;
|
||||
await query("DELETE FROM ocr_items WHERE document_id = $1", [docId]);
|
||||
}
|
||||
|
||||
return NextResponse.json({ success: true });
|
||||
} catch (error: unknown) {
|
||||
console.error("Error cropping file:", error);
|
||||
const message = error instanceof Error ? error.message : "Internal server error";
|
||||
return errorResponse(500, message);
|
||||
}
|
||||
}
|
||||
import { NextRequest, NextResponse } from "next/server";
|
||||
import fs from "fs";
|
||||
import path from "path";
|
||||
import { query } from "../../../db";
|
||||
import crypto from "crypto";
|
||||
import { errorResponse } from "@/utils/api-error";
|
||||
|
||||
const UPLOADS_DIR = "/uploads";
|
||||
const PUBLIC_DIR = path.join(process.cwd(), "public/do-pfm");
|
||||
|
||||
export async function POST(req: NextRequest) {
|
||||
try {
|
||||
const { filename, image } = await req.json();
|
||||
|
||||
if (!filename || !image) {
|
||||
return errorResponse(400, "Filename and image base64 data are required");
|
||||
}
|
||||
|
||||
const safeFile = path.basename(filename);
|
||||
|
||||
const isSample = fs.existsSync(path.join(PUBLIC_DIR, safeFile));
|
||||
const filePath = isSample
|
||||
? path.join(PUBLIC_DIR, safeFile)
|
||||
: path.join(UPLOADS_DIR, safeFile);
|
||||
|
||||
const base64Data = image.replace(/^data:image\/\w+;base64,/, "");
|
||||
const buffer = Buffer.from(base64Data, "base64");
|
||||
|
||||
// Write file to disk
|
||||
fs.writeFileSync(filePath, buffer);
|
||||
console.log(`Cropped file saved successfully at ${filePath}`);
|
||||
|
||||
// Update database fields
|
||||
const fileHash = crypto.createHash("sha256").update(buffer).digest("hex");
|
||||
const stats = fs.statSync(filePath);
|
||||
|
||||
// Update document to unparsed state since layout changes
|
||||
await query(
|
||||
"UPDATE documents SET size = $1, file_hash = $2, parsed = false, layout_parsing_result = NULL WHERE filename = $3",
|
||||
[stats.size, fileHash, filename]
|
||||
);
|
||||
|
||||
// Clear old items for this document
|
||||
const docRes = await query("SELECT id FROM documents WHERE filename = $1", [filename]);
|
||||
if (docRes.rowCount && docRes.rowCount > 0) {
|
||||
const docId = docRes.rows[0].id;
|
||||
await query("DELETE FROM ocr_items WHERE document_id = $1", [docId]);
|
||||
}
|
||||
|
||||
return NextResponse.json({ success: true });
|
||||
} catch (error: unknown) {
|
||||
console.error("Error cropping file:", error);
|
||||
const message = error instanceof Error ? error.message : "Internal server error";
|
||||
return errorResponse(500, message);
|
||||
}
|
||||
}
|
||||
@@ -1,38 +1,38 @@
|
||||
import { NextRequest, NextResponse } from "next/server";
|
||||
import { query } from "../../../../../db";
|
||||
import { errorResponse } from "@/utils/api-error";
|
||||
|
||||
export const dynamic = "force-dynamic";
|
||||
|
||||
export async function GET(
|
||||
req: NextRequest,
|
||||
{ params }: { params: Promise<{ id: string }> | { id: string } }
|
||||
) {
|
||||
try {
|
||||
// Handle both Promise and synchronous params for Next.js version compatibility
|
||||
const resolvedParams = await params;
|
||||
const { id } = resolvedParams;
|
||||
const docId = parseInt(id, 10);
|
||||
|
||||
if (isNaN(docId)) {
|
||||
return errorResponse(400, "Invalid document ID");
|
||||
}
|
||||
|
||||
const res = await query(
|
||||
"SELECT filename, processing_logs FROM documents WHERE id = $1",
|
||||
[docId]
|
||||
);
|
||||
|
||||
if (res.rowCount === 0 || !res.rows[0]) {
|
||||
return errorResponse(404, "Document not found");
|
||||
}
|
||||
|
||||
return NextResponse.json({
|
||||
filename: res.rows[0].filename,
|
||||
processing_logs: res.rows[0].processing_logs || null
|
||||
});
|
||||
} catch (error: any) {
|
||||
console.error("Error fetching document logs:", error);
|
||||
return errorResponse(500, error.message);
|
||||
}
|
||||
}
|
||||
import { NextRequest, NextResponse } from "next/server";
|
||||
import { query } from "../../../../../db";
|
||||
import { errorResponse } from "@/utils/api-error";
|
||||
|
||||
export const dynamic = "force-dynamic";
|
||||
|
||||
export async function GET(
|
||||
req: NextRequest,
|
||||
{ params }: { params: Promise<{ id: string }> | { id: string } }
|
||||
) {
|
||||
try {
|
||||
// Handle both Promise and synchronous params for Next.js version compatibility
|
||||
const resolvedParams = await params;
|
||||
const { id } = resolvedParams;
|
||||
const docId = parseInt(id, 10);
|
||||
|
||||
if (isNaN(docId)) {
|
||||
return errorResponse(400, "Invalid document ID");
|
||||
}
|
||||
|
||||
const res = await query(
|
||||
"SELECT filename, processing_logs FROM documents WHERE id = $1",
|
||||
[docId]
|
||||
);
|
||||
|
||||
if (res.rowCount === 0 || !res.rows[0]) {
|
||||
return errorResponse(404, "Document not found");
|
||||
}
|
||||
|
||||
return NextResponse.json({
|
||||
filename: res.rows[0].filename,
|
||||
processing_logs: res.rows[0].processing_logs || null
|
||||
});
|
||||
} catch (error: any) {
|
||||
console.error("Error fetching document logs:", error);
|
||||
return errorResponse(500, error.message);
|
||||
}
|
||||
}
|
||||
@@ -1,47 +1,47 @@
|
||||
import { NextRequest, NextResponse } from "next/server";
|
||||
import fs from "fs";
|
||||
import path from "path";
|
||||
import { errorResponse } from "@/utils/api-error";
|
||||
|
||||
const UPLOADS_DIR = "/uploads";
|
||||
|
||||
export async function GET(req: NextRequest) {
|
||||
try {
|
||||
const filename = req.nextUrl.searchParams.get("file");
|
||||
if (!filename) {
|
||||
return errorResponse(400, "File name is required");
|
||||
}
|
||||
|
||||
const safeFile = path.basename(filename);
|
||||
const filePath = path.join(UPLOADS_DIR, safeFile);
|
||||
|
||||
if (!fs.existsSync(filePath)) {
|
||||
return errorResponse(404, "File not found");
|
||||
}
|
||||
|
||||
// Determine content type based on extension
|
||||
const ext = path.extname(safeFile).toLowerCase();
|
||||
let contentType = "application/octet-stream";
|
||||
if (ext === ".jpg" || ext === ".jpeg") {
|
||||
contentType = "image/jpeg";
|
||||
} else if (ext === ".png") {
|
||||
contentType = "image/png";
|
||||
} else if (ext === ".gif") {
|
||||
contentType = "image/gif";
|
||||
} else if (ext === ".pdf") {
|
||||
contentType = "application/pdf";
|
||||
}
|
||||
|
||||
const fileBuffer = fs.readFileSync(filePath);
|
||||
return new Response(fileBuffer, {
|
||||
headers: {
|
||||
"Content-Type": contentType,
|
||||
"Cache-Control": "public, max-age=31536000, immutable"
|
||||
}
|
||||
});
|
||||
} catch (error: unknown) {
|
||||
console.error("Error serving file from uploads:", error);
|
||||
const message = error instanceof Error ? error.message : "Internal server error";
|
||||
return errorResponse(500, message);
|
||||
}
|
||||
}
|
||||
import { NextRequest, NextResponse } from "next/server";
|
||||
import fs from "fs";
|
||||
import path from "path";
|
||||
import { errorResponse } from "@/utils/api-error";
|
||||
|
||||
const UPLOADS_DIR = "/uploads";
|
||||
|
||||
export async function GET(req: NextRequest) {
|
||||
try {
|
||||
const filename = req.nextUrl.searchParams.get("file");
|
||||
if (!filename) {
|
||||
return errorResponse(400, "File name is required");
|
||||
}
|
||||
|
||||
const safeFile = path.basename(filename);
|
||||
const filePath = path.join(UPLOADS_DIR, safeFile);
|
||||
|
||||
if (!fs.existsSync(filePath)) {
|
||||
return errorResponse(404, "File not found");
|
||||
}
|
||||
|
||||
// Determine content type based on extension
|
||||
const ext = path.extname(safeFile).toLowerCase();
|
||||
let contentType = "application/octet-stream";
|
||||
if (ext === ".jpg" || ext === ".jpeg") {
|
||||
contentType = "image/jpeg";
|
||||
} else if (ext === ".png") {
|
||||
contentType = "image/png";
|
||||
} else if (ext === ".gif") {
|
||||
contentType = "image/gif";
|
||||
} else if (ext === ".pdf") {
|
||||
contentType = "application/pdf";
|
||||
}
|
||||
|
||||
const fileBuffer = fs.readFileSync(filePath);
|
||||
return new Response(fileBuffer, {
|
||||
headers: {
|
||||
"Content-Type": contentType,
|
||||
"Cache-Control": "public, max-age=31536000, immutable"
|
||||
}
|
||||
});
|
||||
} catch (error: unknown) {
|
||||
console.error("Error serving file from uploads:", error);
|
||||
const message = error instanceof Error ? error.message : "Internal server error";
|
||||
return errorResponse(500, message);
|
||||
}
|
||||
}
|
||||
@@ -1,159 +1,159 @@
|
||||
import { NextRequest, NextResponse } from "next/server";
|
||||
import {
|
||||
getGpuInfo,
|
||||
getContainerStatus,
|
||||
manageContainer,
|
||||
recreateContainer,
|
||||
getEnvSettings,
|
||||
saveEnvSettings,
|
||||
getProcessName,
|
||||
unloadOtherEngines
|
||||
} from "../../../utils/docker";
|
||||
import { errorResponse } from "@/utils/api-error";
|
||||
|
||||
export async function GET(req: NextRequest) {
|
||||
try {
|
||||
const gpus = await getGpuInfo();
|
||||
const settings = await getEnvSettings();
|
||||
|
||||
const containers = {
|
||||
nginx: await getContainerStatus("paddleocr-nginx"),
|
||||
vllmServer: await getContainerStatus("paddleocr-vllm-server"),
|
||||
pipelineApi: await getContainerStatus("paddleocr-pipeline-api"),
|
||||
gradioUi: await getContainerStatus("paddleocr-gradio-ui"),
|
||||
pfmWebApp: await getContainerStatus("paddleocr-pfm-web-app"),
|
||||
db: await getContainerStatus("paddleocr-db")
|
||||
};
|
||||
|
||||
return NextResponse.json({
|
||||
success: true,
|
||||
gpus,
|
||||
settings,
|
||||
containers
|
||||
});
|
||||
} catch (error: any) {
|
||||
console.error("Failed to fetch GPU/container status:", error);
|
||||
return errorResponse(500, error.message);
|
||||
}
|
||||
}
|
||||
|
||||
export async function POST(req: NextRequest) {
|
||||
try {
|
||||
const body = await req.json().catch(() => ({}));
|
||||
const { action } = body;
|
||||
|
||||
if (action === "kill") {
|
||||
const pid = parseInt(body.pid);
|
||||
if (!pid || isNaN(pid)) {
|
||||
return errorResponse(400, "Invalid PID");
|
||||
}
|
||||
|
||||
// Check if process is protected (same rules as admin_panel.py)
|
||||
const procName = getProcessName(pid);
|
||||
const procNameLower = procName.toLowerCase();
|
||||
const protectedKeywords = ["rustdesk", "xorg", "nginx", "systemd", "dockerd", "python3", "node"];
|
||||
if (anyKeywordMatch(procNameLower, protectedKeywords)) {
|
||||
return errorResponse(403, `Operation Denied: Process ${pid} (${procName || "system"}) is protected and cannot be killed.`);
|
||||
}
|
||||
|
||||
try {
|
||||
process.kill(pid, 9);
|
||||
return NextResponse.json({ success: true, message: `Successfully killed process ${pid}` });
|
||||
} catch (err: any) {
|
||||
return errorResponse(500, `Failed to kill process: ${err.message}`);
|
||||
}
|
||||
}
|
||||
|
||||
if (action === "container") {
|
||||
const { containerName, containerAction } = body;
|
||||
const validActions = ["start", "stop", "restart"];
|
||||
const validContainers = [
|
||||
"paddleocr-nginx",
|
||||
"paddleocr-vllm-server",
|
||||
"paddleocr-pipeline-api",
|
||||
"paddleocr-gradio-ui",
|
||||
"paddleocr-pfm-web-app",
|
||||
"paddleocr-db"
|
||||
];
|
||||
|
||||
if (!validActions.includes(containerAction) || !validContainers.includes(containerName)) {
|
||||
return errorResponse(400, "Invalid container name or action");
|
||||
}
|
||||
|
||||
// Prevent self-stopping nextjs app accidentally through UI
|
||||
if (containerName === "paddleocr-pfm-web-app" && containerAction === "stop") {
|
||||
return errorResponse(400, "Cannot stop the active web application container itself.");
|
||||
}
|
||||
|
||||
await manageContainer(containerName, containerAction);
|
||||
return NextResponse.json({
|
||||
success: true,
|
||||
message: `Command 'docker-compose ${containerAction} ${containerName.replace("paddleocr-", "")}' executed successfully.`
|
||||
});
|
||||
}
|
||||
|
||||
if (action === "saveSettings") {
|
||||
const { cudaDevices } = body;
|
||||
if (typeof cudaDevices !== "string" || cudaDevices.trim() === "") {
|
||||
return errorResponse(400, "Invalid GPU allocation settings");
|
||||
}
|
||||
|
||||
const cleanCuda = cudaDevices.trim();
|
||||
await saveEnvSettings(cleanCuda);
|
||||
|
||||
// Recreate GPU containers to apply env settings
|
||||
try {
|
||||
await recreateContainer("paddleocr-vllm-server", cleanCuda);
|
||||
} catch (err: any) {
|
||||
console.error("Failed to recreate vllm-server container:", err);
|
||||
}
|
||||
|
||||
try {
|
||||
await recreateContainer("paddleocr-pipeline-api", cleanCuda);
|
||||
} catch (err: any) {
|
||||
console.error("Failed to recreate pipeline-api container:", err);
|
||||
}
|
||||
|
||||
return NextResponse.json({
|
||||
success: true,
|
||||
message: `GPU settings updated to device index ${cleanCuda}. Core services recreated successfully.`
|
||||
});
|
||||
}
|
||||
|
||||
if (action === "unload") {
|
||||
const { stopped, failed } = await unloadOtherEngines();
|
||||
if (stopped.length === 0 && failed.length === 0) {
|
||||
return NextResponse.json({
|
||||
success: true,
|
||||
message: "All other OCR engines are already stopped/unloaded."
|
||||
});
|
||||
}
|
||||
|
||||
let msg = "";
|
||||
if (stopped.length > 0) {
|
||||
msg += `Successfully stopped/unloaded: ${stopped.join(", ")}. `;
|
||||
}
|
||||
if (failed.length > 0) {
|
||||
msg += `Failed to stop: ${failed.join(", ")}.`;
|
||||
}
|
||||
|
||||
return NextResponse.json({
|
||||
success: failed.length === 0,
|
||||
message: msg.trim(),
|
||||
error: failed.length > 0 ? `Failed to stop some containers: ${failed.join(", ")}` : undefined
|
||||
});
|
||||
}
|
||||
|
||||
return errorResponse(400, "Invalid API action");
|
||||
} catch (error: any) {
|
||||
console.error("GPU API POST error:", error);
|
||||
return errorResponse(500, error.message);
|
||||
}
|
||||
}
|
||||
|
||||
function anyKeywordMatch(str: string, keywords: string[]): boolean {
|
||||
for (const kw of keywords) {
|
||||
if (str.includes(kw)) return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
import { NextRequest, NextResponse } from "next/server";
|
||||
import {
|
||||
getGpuInfo,
|
||||
getContainerStatus,
|
||||
manageContainer,
|
||||
recreateContainer,
|
||||
getEnvSettings,
|
||||
saveEnvSettings,
|
||||
getProcessName,
|
||||
unloadOtherEngines
|
||||
} from "../../../utils/docker";
|
||||
import { errorResponse } from "@/utils/api-error";
|
||||
|
||||
export async function GET(req: NextRequest) {
|
||||
try {
|
||||
const gpus = await getGpuInfo();
|
||||
const settings = await getEnvSettings();
|
||||
|
||||
const containers = {
|
||||
nginx: await getContainerStatus("paddleocr-nginx"),
|
||||
vllmServer: await getContainerStatus("paddleocr-vllm-server"),
|
||||
pipelineApi: await getContainerStatus("paddleocr-pipeline-api"),
|
||||
gradioUi: await getContainerStatus("paddleocr-gradio-ui"),
|
||||
pfmWebApp: await getContainerStatus("paddleocr-pfm-web-app"),
|
||||
db: await getContainerStatus("paddleocr-db")
|
||||
};
|
||||
|
||||
return NextResponse.json({
|
||||
success: true,
|
||||
gpus,
|
||||
settings,
|
||||
containers
|
||||
});
|
||||
} catch (error: any) {
|
||||
console.error("Failed to fetch GPU/container status:", error);
|
||||
return errorResponse(500, error.message);
|
||||
}
|
||||
}
|
||||
|
||||
export async function POST(req: NextRequest) {
|
||||
try {
|
||||
const body = await req.json().catch(() => ({}));
|
||||
const { action } = body;
|
||||
|
||||
if (action === "kill") {
|
||||
const pid = parseInt(body.pid);
|
||||
if (!pid || isNaN(pid)) {
|
||||
return errorResponse(400, "Invalid PID");
|
||||
}
|
||||
|
||||
// Check if process is protected (same rules as admin_panel.py)
|
||||
const procName = getProcessName(pid);
|
||||
const procNameLower = procName.toLowerCase();
|
||||
const protectedKeywords = ["rustdesk", "xorg", "nginx", "systemd", "dockerd", "python3", "node"];
|
||||
if (anyKeywordMatch(procNameLower, protectedKeywords)) {
|
||||
return errorResponse(403, `Operation Denied: Process ${pid} (${procName || "system"}) is protected and cannot be killed.`);
|
||||
}
|
||||
|
||||
try {
|
||||
process.kill(pid, 9);
|
||||
return NextResponse.json({ success: true, message: `Successfully killed process ${pid}` });
|
||||
} catch (err: any) {
|
||||
return errorResponse(500, `Failed to kill process: ${err.message}`);
|
||||
}
|
||||
}
|
||||
|
||||
if (action === "container") {
|
||||
const { containerName, containerAction } = body;
|
||||
const validActions = ["start", "stop", "restart"];
|
||||
const validContainers = [
|
||||
"paddleocr-nginx",
|
||||
"paddleocr-vllm-server",
|
||||
"paddleocr-pipeline-api",
|
||||
"paddleocr-gradio-ui",
|
||||
"paddleocr-pfm-web-app",
|
||||
"paddleocr-db"
|
||||
];
|
||||
|
||||
if (!validActions.includes(containerAction) || !validContainers.includes(containerName)) {
|
||||
return errorResponse(400, "Invalid container name or action");
|
||||
}
|
||||
|
||||
// Prevent self-stopping nextjs app accidentally through UI
|
||||
if (containerName === "paddleocr-pfm-web-app" && containerAction === "stop") {
|
||||
return errorResponse(400, "Cannot stop the active web application container itself.");
|
||||
}
|
||||
|
||||
await manageContainer(containerName, containerAction);
|
||||
return NextResponse.json({
|
||||
success: true,
|
||||
message: `Command 'docker-compose ${containerAction} ${containerName.replace("paddleocr-", "")}' executed successfully.`
|
||||
});
|
||||
}
|
||||
|
||||
if (action === "saveSettings") {
|
||||
const { cudaDevices } = body;
|
||||
if (typeof cudaDevices !== "string" || cudaDevices.trim() === "") {
|
||||
return errorResponse(400, "Invalid GPU allocation settings");
|
||||
}
|
||||
|
||||
const cleanCuda = cudaDevices.trim();
|
||||
await saveEnvSettings(cleanCuda);
|
||||
|
||||
// Recreate GPU containers to apply env settings
|
||||
try {
|
||||
await recreateContainer("paddleocr-vllm-server", cleanCuda);
|
||||
} catch (err: any) {
|
||||
console.error("Failed to recreate vllm-server container:", err);
|
||||
}
|
||||
|
||||
try {
|
||||
await recreateContainer("paddleocr-pipeline-api", cleanCuda);
|
||||
} catch (err: any) {
|
||||
console.error("Failed to recreate pipeline-api container:", err);
|
||||
}
|
||||
|
||||
return NextResponse.json({
|
||||
success: true,
|
||||
message: `GPU settings updated to device index ${cleanCuda}. Core services recreated successfully.`
|
||||
});
|
||||
}
|
||||
|
||||
if (action === "unload") {
|
||||
const { stopped, failed } = await unloadOtherEngines();
|
||||
if (stopped.length === 0 && failed.length === 0) {
|
||||
return NextResponse.json({
|
||||
success: true,
|
||||
message: "All other OCR engines are already stopped/unloaded."
|
||||
});
|
||||
}
|
||||
|
||||
let msg = "";
|
||||
if (stopped.length > 0) {
|
||||
msg += `Successfully stopped/unloaded: ${stopped.join(", ")}. `;
|
||||
}
|
||||
if (failed.length > 0) {
|
||||
msg += `Failed to stop: ${failed.join(", ")}.`;
|
||||
}
|
||||
|
||||
return NextResponse.json({
|
||||
success: failed.length === 0,
|
||||
message: msg.trim(),
|
||||
error: failed.length > 0 ? `Failed to stop some containers: ${failed.join(", ")}` : undefined
|
||||
});
|
||||
}
|
||||
|
||||
return errorResponse(400, "Invalid API action");
|
||||
} catch (error: any) {
|
||||
console.error("GPU API POST error:", error);
|
||||
return errorResponse(500, error.message);
|
||||
}
|
||||
}
|
||||
|
||||
function anyKeywordMatch(str: string, keywords: string[]): boolean {
|
||||
for (const kw of keywords) {
|
||||
if (str.includes(kw)) return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
@@ -1,202 +1,202 @@
|
||||
import { NextRequest, NextResponse } from "next/server";
|
||||
import { query, cleanupAndReindexItems } from "../../../db";
|
||||
import fs from "fs";
|
||||
import path from "path";
|
||||
import { errorResponse } from "@/utils/api-error";
|
||||
|
||||
const UPLOADS_DIR = "/uploads";
|
||||
|
||||
export async function GET(req: NextRequest) {
|
||||
try {
|
||||
const fileParam = req.nextUrl.searchParams.get("file");
|
||||
|
||||
if (fileParam) {
|
||||
const safeFile = path.basename(fileParam);
|
||||
|
||||
// 1. Try to load from database first
|
||||
const docRes = await query(
|
||||
"SELECT id, layout_parsing_result, metadata FROM documents WHERE filename = $1",
|
||||
[safeFile]
|
||||
);
|
||||
|
||||
if (docRes.rowCount && docRes.rowCount > 0) {
|
||||
const doc = docRes.rows[0];
|
||||
const docId = doc.id;
|
||||
const pipelineResult = doc.layout_parsing_result;
|
||||
|
||||
// Clean up and re-index invalid items first
|
||||
await cleanupAndReindexItems(docId);
|
||||
|
||||
// Fetch items
|
||||
const itemsRes = await query(
|
||||
`SELECT row_index,
|
||||
kode_barang, nama_barang, banyak, jumlah,
|
||||
is_flagged, remark
|
||||
FROM ocr_items
|
||||
WHERE document_id = $1
|
||||
ORDER BY row_index`,
|
||||
[docId]
|
||||
);
|
||||
|
||||
const items = itemsRes.rows.map(row => ({
|
||||
kodeBarang: row.kode_barang,
|
||||
namaBarang: row.nama_barang,
|
||||
banyak: row.banyak,
|
||||
jumlah: row.jumlah
|
||||
}));
|
||||
|
||||
const flagged: Record<number, boolean> = {};
|
||||
const remarks: Record<number, string> = {};
|
||||
|
||||
itemsRes.rows.forEach(row => {
|
||||
if (row.is_flagged) {
|
||||
flagged[row.row_index] = true;
|
||||
}
|
||||
if (row.remark && row.remark.trim()) {
|
||||
remarks[row.row_index] = row.remark;
|
||||
}
|
||||
});
|
||||
|
||||
return NextResponse.json({
|
||||
errorCode: 0,
|
||||
errorMsg: "Success",
|
||||
result: pipelineResult,
|
||||
items,
|
||||
flagged,
|
||||
remarks,
|
||||
headerRemark: (doc.metadata as any)?.headerRemark || ""
|
||||
});
|
||||
}
|
||||
|
||||
// 2. Fallback to filesystem
|
||||
const jsonPath = path.join(UPLOADS_DIR, `${safeFile}.json`);
|
||||
if (fs.existsSync(jsonPath)) {
|
||||
const jsonData = fs.readFileSync(jsonPath, "utf8");
|
||||
const data = JSON.parse(jsonData);
|
||||
return NextResponse.json({
|
||||
errorCode: 0,
|
||||
errorMsg: "Success",
|
||||
result: data.result || data
|
||||
});
|
||||
}
|
||||
|
||||
return errorResponse(404, "Document not found");
|
||||
}
|
||||
|
||||
// List view: return history list from DB
|
||||
const showAll = req.nextUrl.searchParams.get("all") === "true";
|
||||
|
||||
let listRes;
|
||||
if (showAll) {
|
||||
listRes = await query(
|
||||
`SELECT id, filename, upload_time, size, parsed, is_sample, metadata,
|
||||
(SELECT COUNT(*) FROM ocr_items WHERE document_id = documents.id) as total_items,
|
||||
(SELECT COUNT(*) FROM ocr_items WHERE document_id = documents.id AND is_flagged = TRUE) as flagged_items
|
||||
FROM documents
|
||||
ORDER BY upload_time DESC`
|
||||
);
|
||||
} else {
|
||||
listRes = await query(
|
||||
`SELECT id, filename, upload_time, size, parsed, is_sample, metadata,
|
||||
(SELECT COUNT(*) FROM ocr_items WHERE document_id = documents.id) as total_items,
|
||||
(SELECT COUNT(*) FROM ocr_items WHERE document_id = documents.id AND is_flagged = TRUE) as flagged_items
|
||||
FROM documents
|
||||
WHERE is_sample = FALSE
|
||||
ORDER BY upload_time DESC`
|
||||
);
|
||||
}
|
||||
|
||||
const history = listRes.rows.map(row => ({
|
||||
id: row.id,
|
||||
filename: row.filename,
|
||||
uploadTime: row.upload_time.toISOString(),
|
||||
size: row.size,
|
||||
parsed: row.parsed,
|
||||
isSample: row.is_sample,
|
||||
metadata: row.metadata,
|
||||
totalItems: parseInt(row.total_items || "0"),
|
||||
flaggedItems: parseInt(row.flagged_items || "0")
|
||||
}));
|
||||
|
||||
return NextResponse.json({ history });
|
||||
} catch (error: unknown) {
|
||||
console.error("Error in history API route:", error);
|
||||
const message = error instanceof Error ? error.message : "Internal server error";
|
||||
return errorResponse(500, message);
|
||||
}
|
||||
}
|
||||
|
||||
export async function DELETE(req: NextRequest) {
|
||||
try {
|
||||
const { filename } = await req.json();
|
||||
if (!filename) {
|
||||
return errorResponse(400, "Filename is required");
|
||||
}
|
||||
|
||||
const safeFile = path.basename(filename);
|
||||
|
||||
// Check if it exists and get its status
|
||||
const checkRes = await query(
|
||||
"SELECT id, is_sample FROM documents WHERE filename = $1",
|
||||
[safeFile]
|
||||
);
|
||||
|
||||
if (checkRes.rowCount && checkRes.rowCount > 0) {
|
||||
const doc = checkRes.rows[0];
|
||||
const isSample = doc.is_sample;
|
||||
|
||||
// Delete from DB (cascading delete will remove ocr_items)
|
||||
await query("DELETE FROM documents WHERE filename = $1", [safeFile]);
|
||||
|
||||
// If it is a custom upload, clean up files from /uploads directory
|
||||
if (!isSample) {
|
||||
const imagePath = path.join(UPLOADS_DIR, safeFile);
|
||||
const jsonPath = path.join(UPLOADS_DIR, `${safeFile}.json`);
|
||||
|
||||
if (fs.existsSync(imagePath)) {
|
||||
fs.unlinkSync(imagePath);
|
||||
}
|
||||
if (fs.existsSync(jsonPath)) {
|
||||
fs.unlinkSync(jsonPath);
|
||||
}
|
||||
}
|
||||
|
||||
return NextResponse.json({ success: true });
|
||||
}
|
||||
|
||||
return errorResponse(404, "Document not found");
|
||||
} catch (error: unknown) {
|
||||
console.error("Error in DELETE history API route:", error);
|
||||
const message = error instanceof Error ? error.message : "Internal server error";
|
||||
return errorResponse(500, message);
|
||||
}
|
||||
}
|
||||
|
||||
export async function POST(req: NextRequest) {
|
||||
try {
|
||||
const { filename, remark } = await req.json();
|
||||
if (!filename) {
|
||||
return errorResponse(400, "Filename is required");
|
||||
}
|
||||
|
||||
const safeFile = path.basename(filename);
|
||||
|
||||
const valueJson = JSON.stringify(remark || "");
|
||||
const updateRes = await query(
|
||||
`UPDATE documents
|
||||
SET metadata = jsonb_set(coalesce(metadata, '{}'::jsonb), '{headerRemark}', $1::jsonb)
|
||||
WHERE filename = $2`,
|
||||
[valueJson, safeFile]
|
||||
);
|
||||
|
||||
if (updateRes.rowCount && updateRes.rowCount > 0) {
|
||||
return NextResponse.json({ success: true });
|
||||
}
|
||||
|
||||
return errorResponse(404, "Document not found");
|
||||
} catch (error: unknown) {
|
||||
console.error("Error in POST history API route:", error);
|
||||
const message = error instanceof Error ? error.message : "Internal server error";
|
||||
return errorResponse(500, message);
|
||||
}
|
||||
}
|
||||
import { NextRequest, NextResponse } from "next/server";
|
||||
import { query, cleanupAndReindexItems } from "../../../db";
|
||||
import fs from "fs";
|
||||
import path from "path";
|
||||
import { errorResponse } from "@/utils/api-error";
|
||||
|
||||
const UPLOADS_DIR = "/uploads";
|
||||
|
||||
export async function GET(req: NextRequest) {
|
||||
try {
|
||||
const fileParam = req.nextUrl.searchParams.get("file");
|
||||
|
||||
if (fileParam) {
|
||||
const safeFile = path.basename(fileParam);
|
||||
|
||||
// 1. Try to load from database first
|
||||
const docRes = await query(
|
||||
"SELECT id, layout_parsing_result, metadata FROM documents WHERE filename = $1",
|
||||
[safeFile]
|
||||
);
|
||||
|
||||
if (docRes.rowCount && docRes.rowCount > 0) {
|
||||
const doc = docRes.rows[0];
|
||||
const docId = doc.id;
|
||||
const pipelineResult = doc.layout_parsing_result;
|
||||
|
||||
// Clean up and re-index invalid items first
|
||||
await cleanupAndReindexItems(docId);
|
||||
|
||||
// Fetch items
|
||||
const itemsRes = await query(
|
||||
`SELECT row_index,
|
||||
kode_barang, nama_barang, banyak, jumlah,
|
||||
is_flagged, remark
|
||||
FROM ocr_items
|
||||
WHERE document_id = $1
|
||||
ORDER BY row_index`,
|
||||
[docId]
|
||||
);
|
||||
|
||||
const items = itemsRes.rows.map(row => ({
|
||||
kodeBarang: row.kode_barang,
|
||||
namaBarang: row.nama_barang,
|
||||
banyak: row.banyak,
|
||||
jumlah: row.jumlah
|
||||
}));
|
||||
|
||||
const flagged: Record<number, boolean> = {};
|
||||
const remarks: Record<number, string> = {};
|
||||
|
||||
itemsRes.rows.forEach(row => {
|
||||
if (row.is_flagged) {
|
||||
flagged[row.row_index] = true;
|
||||
}
|
||||
if (row.remark && row.remark.trim()) {
|
||||
remarks[row.row_index] = row.remark;
|
||||
}
|
||||
});
|
||||
|
||||
return NextResponse.json({
|
||||
errorCode: 0,
|
||||
errorMsg: "Success",
|
||||
result: pipelineResult,
|
||||
items,
|
||||
flagged,
|
||||
remarks,
|
||||
headerRemark: (doc.metadata as any)?.headerRemark || ""
|
||||
});
|
||||
}
|
||||
|
||||
// 2. Fallback to filesystem
|
||||
const jsonPath = path.join(UPLOADS_DIR, `${safeFile}.json`);
|
||||
if (fs.existsSync(jsonPath)) {
|
||||
const jsonData = fs.readFileSync(jsonPath, "utf8");
|
||||
const data = JSON.parse(jsonData);
|
||||
return NextResponse.json({
|
||||
errorCode: 0,
|
||||
errorMsg: "Success",
|
||||
result: data.result || data
|
||||
});
|
||||
}
|
||||
|
||||
return errorResponse(404, "Document not found");
|
||||
}
|
||||
|
||||
// List view: return history list from DB
|
||||
const showAll = req.nextUrl.searchParams.get("all") === "true";
|
||||
|
||||
let listRes;
|
||||
if (showAll) {
|
||||
listRes = await query(
|
||||
`SELECT id, filename, upload_time, size, parsed, is_sample, metadata,
|
||||
(SELECT COUNT(*) FROM ocr_items WHERE document_id = documents.id) as total_items,
|
||||
(SELECT COUNT(*) FROM ocr_items WHERE document_id = documents.id AND is_flagged = TRUE) as flagged_items
|
||||
FROM documents
|
||||
ORDER BY upload_time DESC`
|
||||
);
|
||||
} else {
|
||||
listRes = await query(
|
||||
`SELECT id, filename, upload_time, size, parsed, is_sample, metadata,
|
||||
(SELECT COUNT(*) FROM ocr_items WHERE document_id = documents.id) as total_items,
|
||||
(SELECT COUNT(*) FROM ocr_items WHERE document_id = documents.id AND is_flagged = TRUE) as flagged_items
|
||||
FROM documents
|
||||
WHERE is_sample = FALSE
|
||||
ORDER BY upload_time DESC`
|
||||
);
|
||||
}
|
||||
|
||||
const history = listRes.rows.map(row => ({
|
||||
id: row.id,
|
||||
filename: row.filename,
|
||||
uploadTime: row.upload_time.toISOString(),
|
||||
size: row.size,
|
||||
parsed: row.parsed,
|
||||
isSample: row.is_sample,
|
||||
metadata: row.metadata,
|
||||
totalItems: parseInt(row.total_items || "0"),
|
||||
flaggedItems: parseInt(row.flagged_items || "0")
|
||||
}));
|
||||
|
||||
return NextResponse.json({ history });
|
||||
} catch (error: unknown) {
|
||||
console.error("Error in history API route:", error);
|
||||
const message = error instanceof Error ? error.message : "Internal server error";
|
||||
return errorResponse(500, message);
|
||||
}
|
||||
}
|
||||
|
||||
export async function DELETE(req: NextRequest) {
|
||||
try {
|
||||
const { filename } = await req.json();
|
||||
if (!filename) {
|
||||
return errorResponse(400, "Filename is required");
|
||||
}
|
||||
|
||||
const safeFile = path.basename(filename);
|
||||
|
||||
// Check if it exists and get its status
|
||||
const checkRes = await query(
|
||||
"SELECT id, is_sample FROM documents WHERE filename = $1",
|
||||
[safeFile]
|
||||
);
|
||||
|
||||
if (checkRes.rowCount && checkRes.rowCount > 0) {
|
||||
const doc = checkRes.rows[0];
|
||||
const isSample = doc.is_sample;
|
||||
|
||||
// Delete from DB (cascading delete will remove ocr_items)
|
||||
await query("DELETE FROM documents WHERE filename = $1", [safeFile]);
|
||||
|
||||
// If it is a custom upload, clean up files from /uploads directory
|
||||
if (!isSample) {
|
||||
const imagePath = path.join(UPLOADS_DIR, safeFile);
|
||||
const jsonPath = path.join(UPLOADS_DIR, `${safeFile}.json`);
|
||||
|
||||
if (fs.existsSync(imagePath)) {
|
||||
fs.unlinkSync(imagePath);
|
||||
}
|
||||
if (fs.existsSync(jsonPath)) {
|
||||
fs.unlinkSync(jsonPath);
|
||||
}
|
||||
}
|
||||
|
||||
return NextResponse.json({ success: true });
|
||||
}
|
||||
|
||||
return errorResponse(404, "Document not found");
|
||||
} catch (error: unknown) {
|
||||
console.error("Error in DELETE history API route:", error);
|
||||
const message = error instanceof Error ? error.message : "Internal server error";
|
||||
return errorResponse(500, message);
|
||||
}
|
||||
}
|
||||
|
||||
export async function POST(req: NextRequest) {
|
||||
try {
|
||||
const { filename, remark } = await req.json();
|
||||
if (!filename) {
|
||||
return errorResponse(400, "Filename is required");
|
||||
}
|
||||
|
||||
const safeFile = path.basename(filename);
|
||||
|
||||
const valueJson = JSON.stringify(remark || "");
|
||||
const updateRes = await query(
|
||||
`UPDATE documents
|
||||
SET metadata = jsonb_set(coalesce(metadata, '{}'::jsonb), '{headerRemark}', $1::jsonb)
|
||||
WHERE filename = $2`,
|
||||
[valueJson, safeFile]
|
||||
);
|
||||
|
||||
if (updateRes.rowCount && updateRes.rowCount > 0) {
|
||||
return NextResponse.json({ success: true });
|
||||
}
|
||||
|
||||
return errorResponse(404, "Document not found");
|
||||
} catch (error: unknown) {
|
||||
console.error("Error in POST history API route:", error);
|
||||
const message = error instanceof Error ? error.message : "Internal server error";
|
||||
return errorResponse(500, message);
|
||||
}
|
||||
}
|
||||
@@ -1,23 +1,23 @@
|
||||
import { NextResponse } from "next/server";
|
||||
import fs from "fs";
|
||||
import path from "path";
|
||||
import { errorResponse } from "@/utils/api-error";
|
||||
|
||||
export async function GET() {
|
||||
try {
|
||||
const dirPath = path.join(process.cwd(), "..", "sources", "test-images");
|
||||
if (!fs.existsSync(dirPath)) {
|
||||
return NextResponse.json({ files: [] });
|
||||
}
|
||||
const files = fs.readdirSync(dirPath).filter(file => {
|
||||
const ext = path.extname(file).toLowerCase();
|
||||
return ext === ".jpg" || ext === ".jpeg" || ext === ".png";
|
||||
});
|
||||
// Sort files to keep consistent ordering in UI
|
||||
files.sort();
|
||||
return NextResponse.json({ files });
|
||||
} catch (error: any) {
|
||||
console.error("Error reading test-images directory:", error);
|
||||
return errorResponse(500, error.message);
|
||||
}
|
||||
}
|
||||
import { NextResponse } from "next/server";
|
||||
import fs from "fs";
|
||||
import path from "path";
|
||||
import { errorResponse } from "@/utils/api-error";
|
||||
|
||||
export async function GET() {
|
||||
try {
|
||||
const dirPath = path.join(process.cwd(), "..", "sources", "test-images");
|
||||
if (!fs.existsSync(dirPath)) {
|
||||
return NextResponse.json({ files: [] });
|
||||
}
|
||||
const files = fs.readdirSync(dirPath).filter(file => {
|
||||
const ext = path.extname(file).toLowerCase();
|
||||
return ext === ".jpg" || ext === ".jpeg" || ext === ".png";
|
||||
});
|
||||
// Sort files to keep consistent ordering in UI
|
||||
files.sort();
|
||||
return NextResponse.json({ files });
|
||||
} catch (error: any) {
|
||||
console.error("Error reading test-images directory:", error);
|
||||
return errorResponse(500, error.message);
|
||||
}
|
||||
}
|
||||
@@ -1,238 +1,238 @@
|
||||
import { NextRequest, NextResponse } from "next/server";
|
||||
import fs from "fs";
|
||||
import path from "path";
|
||||
import crypto from "crypto";
|
||||
import { query } from "@/db";
|
||||
import { errorResponse } from "@/utils/api-error";
|
||||
|
||||
// Separate from DO manual_labels.json - product scan ground truth only
|
||||
const LABELS_PATH = path.join(process.cwd(), "..", "sources", "product_manual_labels.json");
|
||||
|
||||
interface ProductScanLabel {
|
||||
filename: string;
|
||||
no_sku: string;
|
||||
nama_item: string;
|
||||
expiry_date: string;
|
||||
top1_confidence: number | null;
|
||||
notes: string;
|
||||
saved_at: string;
|
||||
}
|
||||
|
||||
function sanitizeFilename(filename: string): string {
|
||||
let cleaned = filename.replace(/\\/g, "/");
|
||||
while (cleaned.startsWith("/")) {
|
||||
cleaned = cleaned.substring(1);
|
||||
}
|
||||
return cleaned.replace(/\.\.\//g, "");
|
||||
}
|
||||
|
||||
function normalizeDateString(dateStr: string): string {
|
||||
if (!dateStr) return "";
|
||||
const trimmed = dateStr.trim();
|
||||
|
||||
// Pattern 1: d Month YYYY (e.g. 7 June 2026)
|
||||
const textPattern = /^(\d{1,2})\s+([a-zA-Z]+)\s+(\d{4})$/;
|
||||
const tm = trimmed.match(textPattern);
|
||||
if (tm) {
|
||||
const day = tm[1].padStart(2, "0");
|
||||
const month = tm[2].charAt(0).toUpperCase() + tm[2].slice(1).toLowerCase();
|
||||
const year = tm[3];
|
||||
return `${day} ${month} ${year}`;
|
||||
}
|
||||
|
||||
// Pattern 2: d/m/YYYY or d-m-YYYY or d.m.YYYY (e.g. 7/6/2026)
|
||||
const digitPattern = /^(\d{1,2})([-./])(\d{1,2})\2(\d{2,4})$/;
|
||||
const dm = trimmed.match(digitPattern);
|
||||
if (dm) {
|
||||
const day = dm[1].padStart(2, "0");
|
||||
const month = dm[3].padStart(2, "0");
|
||||
let year = dm[4];
|
||||
if (year.length === 2) {
|
||||
year = "20" + year;
|
||||
}
|
||||
return `${day}/${month}/${year}`;
|
||||
}
|
||||
|
||||
return trimmed;
|
||||
}
|
||||
|
||||
let purged = false;
|
||||
function readLabels(): ProductScanLabel[] {
|
||||
if (!fs.existsSync(LABELS_PATH)) {
|
||||
return [];
|
||||
}
|
||||
const raw = fs.readFileSync(LABELS_PATH, "utf8");
|
||||
if (!raw.trim()) return [];
|
||||
let labels: ProductScanLabel[] = JSON.parse(raw);
|
||||
|
||||
// Cleanup phantom uploaded-* entries once
|
||||
if (!purged) {
|
||||
const valid = labels.filter((l) => !l.filename.startsWith("uploaded-"));
|
||||
if (valid.length !== labels.length) {
|
||||
writeLabels(valid);
|
||||
labels = valid;
|
||||
}
|
||||
purged = true;
|
||||
}
|
||||
return labels;
|
||||
}
|
||||
|
||||
function writeLabels(labels: ProductScanLabel[]) {
|
||||
const dir = path.dirname(LABELS_PATH);
|
||||
if (!fs.existsSync(dir)) {
|
||||
fs.mkdirSync(dir, { recursive: true });
|
||||
}
|
||||
fs.writeFileSync(LABELS_PATH, JSON.stringify(labels, null, 2), "utf8");
|
||||
}
|
||||
|
||||
export async function GET(req: NextRequest) {
|
||||
try {
|
||||
const { searchParams } = new URL(req.url);
|
||||
const filename = searchParams.get("filename");
|
||||
|
||||
if (!filename) {
|
||||
const labels = readLabels();
|
||||
return NextResponse.json(labels);
|
||||
}
|
||||
|
||||
const safeFilename = sanitizeFilename(filename);
|
||||
const labels = readLabels();
|
||||
const existing = labels.find((l) => l.filename === safeFilename);
|
||||
|
||||
// Inferred values from filename/directory structure
|
||||
let inferredSku = "";
|
||||
let inferredNamaItem = "";
|
||||
const parts = safeFilename.split("/");
|
||||
if (parts.length > 1) {
|
||||
const folderName = parts[0];
|
||||
const match = folderName.match(/^(\d{8})/);
|
||||
if (match) {
|
||||
inferredSku = match[1];
|
||||
} else if (/^\d{8}$/.test(folderName)) {
|
||||
inferredSku = folderName;
|
||||
}
|
||||
}
|
||||
|
||||
if (inferredSku) {
|
||||
try {
|
||||
const dbRes = await query("SELECT nama_item FROM sku_master WHERE no_sku = $1", [inferredSku]);
|
||||
if (dbRes.rowCount && dbRes.rowCount > 0) {
|
||||
inferredNamaItem = dbRes.rows[0].nama_item;
|
||||
}
|
||||
} catch (dbErr) {
|
||||
console.error("Failed to query sku_master for manual label:", dbErr);
|
||||
}
|
||||
}
|
||||
|
||||
// Inferred expiry date from sibling files in the same parent directory
|
||||
let siblingExpiry = "";
|
||||
let parentFolder = "";
|
||||
if (parts.length > 1) {
|
||||
parentFolder = parts.slice(0, -1).join("/");
|
||||
}
|
||||
if (parentFolder) {
|
||||
const sibling = labels.find(
|
||||
(l) => l.filename.startsWith(parentFolder + "/") && l.expiry_date
|
||||
);
|
||||
if (sibling) {
|
||||
siblingExpiry = sibling.expiry_date;
|
||||
}
|
||||
}
|
||||
|
||||
if (existing) {
|
||||
return NextResponse.json({
|
||||
...existing,
|
||||
no_sku: existing.no_sku || inferredSku,
|
||||
nama_item: existing.nama_item || inferredNamaItem,
|
||||
expiry_date: existing.expiry_date || siblingExpiry
|
||||
});
|
||||
}
|
||||
|
||||
// Return empty default state if not found, with inferred metadata
|
||||
return NextResponse.json({
|
||||
filename: safeFilename,
|
||||
no_sku: inferredSku,
|
||||
nama_item: inferredNamaItem,
|
||||
expiry_date: siblingExpiry,
|
||||
top1_confidence: null,
|
||||
notes: "",
|
||||
saved_at: ""
|
||||
});
|
||||
} catch (err: unknown) {
|
||||
console.error("Error in GET manual-label-scan:", err);
|
||||
const message = err instanceof Error ? err.message : "Failed to load product label";
|
||||
return errorResponse(500, message);
|
||||
}
|
||||
}
|
||||
|
||||
export async function POST(req: NextRequest) {
|
||||
try {
|
||||
const body = await req.json();
|
||||
const { filename, image } = body;
|
||||
|
||||
let safeFilename = sanitizeFilename(filename || "unknown.jpg");
|
||||
|
||||
if (image && image.startsWith("data:image/")) {
|
||||
const base64Data = image.replace(/^data:image\/\w+;base64,/, "");
|
||||
const buffer = Buffer.from(base64Data, "base64");
|
||||
const hash = crypto.createHash("md5").update(buffer).digest("hex");
|
||||
const ext = image.match(/^data:image\/(\w+);base64,/)?.[1] || "jpg";
|
||||
safeFilename = `${hash}.${ext}`;
|
||||
|
||||
const saveDir = path.join(process.cwd(), "..", "sources", "product-test-images");
|
||||
if (!fs.existsSync(saveDir)) {
|
||||
fs.mkdirSync(saveDir, { recursive: true });
|
||||
}
|
||||
fs.writeFileSync(path.join(saveDir, safeFilename), buffer);
|
||||
}
|
||||
|
||||
if (!safeFilename || safeFilename === "unknown.jpg") {
|
||||
return errorResponse(400, "Filename or valid image is required in request body");
|
||||
}
|
||||
|
||||
const labels = readLabels();
|
||||
const index = labels.findIndex((l) => l.filename === safeFilename);
|
||||
|
||||
const entry: ProductScanLabel = {
|
||||
filename: safeFilename,
|
||||
no_sku: body.no_sku || "",
|
||||
nama_item: body.nama_item || "",
|
||||
expiry_date: normalizeDateString(body.expiry_date || ""),
|
||||
top1_confidence: typeof body.top1_confidence === "number" ? body.top1_confidence : null,
|
||||
notes: body.notes || "",
|
||||
saved_at: new Date().toISOString()
|
||||
};
|
||||
|
||||
if (index >= 0) {
|
||||
labels[index] = entry;
|
||||
} else {
|
||||
labels.push(entry);
|
||||
}
|
||||
|
||||
writeLabels(labels);
|
||||
|
||||
return NextResponse.json({ success: true, filePath: LABELS_PATH, entry });
|
||||
} catch (err: unknown) {
|
||||
console.error("Error in POST manual-label-scan:", err);
|
||||
const message = err instanceof Error ? err.message : "Failed to save product label";
|
||||
return errorResponse(500, message);
|
||||
}
|
||||
}
|
||||
|
||||
export async function DELETE(req: NextRequest) {
|
||||
try {
|
||||
const { searchParams } = new URL(req.url);
|
||||
const filename = searchParams.get("filename");
|
||||
if (!filename) return errorResponse(400, "Filename parameter is required");
|
||||
|
||||
const safeFilename = sanitizeFilename(filename);
|
||||
const labels = readLabels();
|
||||
const filtered = labels.filter((l) => l.filename !== safeFilename);
|
||||
|
||||
writeLabels(filtered);
|
||||
return NextResponse.json({ success: true });
|
||||
} catch (err: unknown) {
|
||||
const message = err instanceof Error ? err.message : "Failed to delete label";
|
||||
return errorResponse(500, message);
|
||||
}
|
||||
}
|
||||
import { NextRequest, NextResponse } from "next/server";
|
||||
import fs from "fs";
|
||||
import path from "path";
|
||||
import crypto from "crypto";
|
||||
import { query } from "@/db";
|
||||
import { errorResponse } from "@/utils/api-error";
|
||||
|
||||
// Separate from DO manual_labels.json - product scan ground truth only
|
||||
const LABELS_PATH = path.join(process.cwd(), "..", "sources", "product_manual_labels.json");
|
||||
|
||||
interface ProductScanLabel {
|
||||
filename: string;
|
||||
no_sku: string;
|
||||
nama_item: string;
|
||||
expiry_date: string;
|
||||
top1_confidence: number | null;
|
||||
notes: string;
|
||||
saved_at: string;
|
||||
}
|
||||
|
||||
function sanitizeFilename(filename: string): string {
|
||||
let cleaned = filename.replace(/\\/g, "/");
|
||||
while (cleaned.startsWith("/")) {
|
||||
cleaned = cleaned.substring(1);
|
||||
}
|
||||
return cleaned.replace(/\.\.\//g, "");
|
||||
}
|
||||
|
||||
function normalizeDateString(dateStr: string): string {
|
||||
if (!dateStr) return "";
|
||||
const trimmed = dateStr.trim();
|
||||
|
||||
// Pattern 1: d Month YYYY (e.g. 7 June 2026)
|
||||
const textPattern = /^(\d{1,2})\s+([a-zA-Z]+)\s+(\d{4})$/;
|
||||
const tm = trimmed.match(textPattern);
|
||||
if (tm) {
|
||||
const day = tm[1].padStart(2, "0");
|
||||
const month = tm[2].charAt(0).toUpperCase() + tm[2].slice(1).toLowerCase();
|
||||
const year = tm[3];
|
||||
return `${day} ${month} ${year}`;
|
||||
}
|
||||
|
||||
// Pattern 2: d/m/YYYY or d-m-YYYY or d.m.YYYY (e.g. 7/6/2026)
|
||||
const digitPattern = /^(\d{1,2})([-./])(\d{1,2})\2(\d{2,4})$/;
|
||||
const dm = trimmed.match(digitPattern);
|
||||
if (dm) {
|
||||
const day = dm[1].padStart(2, "0");
|
||||
const month = dm[3].padStart(2, "0");
|
||||
let year = dm[4];
|
||||
if (year.length === 2) {
|
||||
year = "20" + year;
|
||||
}
|
||||
return `${day}/${month}/${year}`;
|
||||
}
|
||||
|
||||
return trimmed;
|
||||
}
|
||||
|
||||
let purged = false;
|
||||
function readLabels(): ProductScanLabel[] {
|
||||
if (!fs.existsSync(LABELS_PATH)) {
|
||||
return [];
|
||||
}
|
||||
const raw = fs.readFileSync(LABELS_PATH, "utf8");
|
||||
if (!raw.trim()) return [];
|
||||
let labels: ProductScanLabel[] = JSON.parse(raw);
|
||||
|
||||
// Cleanup phantom uploaded-* entries once
|
||||
if (!purged) {
|
||||
const valid = labels.filter((l) => !l.filename.startsWith("uploaded-"));
|
||||
if (valid.length !== labels.length) {
|
||||
writeLabels(valid);
|
||||
labels = valid;
|
||||
}
|
||||
purged = true;
|
||||
}
|
||||
return labels;
|
||||
}
|
||||
|
||||
function writeLabels(labels: ProductScanLabel[]) {
|
||||
const dir = path.dirname(LABELS_PATH);
|
||||
if (!fs.existsSync(dir)) {
|
||||
fs.mkdirSync(dir, { recursive: true });
|
||||
}
|
||||
fs.writeFileSync(LABELS_PATH, JSON.stringify(labels, null, 2), "utf8");
|
||||
}
|
||||
|
||||
export async function GET(req: NextRequest) {
|
||||
try {
|
||||
const { searchParams } = new URL(req.url);
|
||||
const filename = searchParams.get("filename");
|
||||
|
||||
if (!filename) {
|
||||
const labels = readLabels();
|
||||
return NextResponse.json(labels);
|
||||
}
|
||||
|
||||
const safeFilename = sanitizeFilename(filename);
|
||||
const labels = readLabels();
|
||||
const existing = labels.find((l) => l.filename === safeFilename);
|
||||
|
||||
// Inferred values from filename/directory structure
|
||||
let inferredSku = "";
|
||||
let inferredNamaItem = "";
|
||||
const parts = safeFilename.split("/");
|
||||
if (parts.length > 1) {
|
||||
const folderName = parts[0];
|
||||
const match = folderName.match(/^(\d{8})/);
|
||||
if (match) {
|
||||
inferredSku = match[1];
|
||||
} else if (/^\d{8}$/.test(folderName)) {
|
||||
inferredSku = folderName;
|
||||
}
|
||||
}
|
||||
|
||||
if (inferredSku) {
|
||||
try {
|
||||
const dbRes = await query("SELECT nama_item FROM sku_master WHERE no_sku = $1", [inferredSku]);
|
||||
if (dbRes.rowCount && dbRes.rowCount > 0) {
|
||||
inferredNamaItem = dbRes.rows[0].nama_item;
|
||||
}
|
||||
} catch (dbErr) {
|
||||
console.error("Failed to query sku_master for manual label:", dbErr);
|
||||
}
|
||||
}
|
||||
|
||||
// Inferred expiry date from sibling files in the same parent directory
|
||||
let siblingExpiry = "";
|
||||
let parentFolder = "";
|
||||
if (parts.length > 1) {
|
||||
parentFolder = parts.slice(0, -1).join("/");
|
||||
}
|
||||
if (parentFolder) {
|
||||
const sibling = labels.find(
|
||||
(l) => l.filename.startsWith(parentFolder + "/") && l.expiry_date
|
||||
);
|
||||
if (sibling) {
|
||||
siblingExpiry = sibling.expiry_date;
|
||||
}
|
||||
}
|
||||
|
||||
if (existing) {
|
||||
return NextResponse.json({
|
||||
...existing,
|
||||
no_sku: existing.no_sku || inferredSku,
|
||||
nama_item: existing.nama_item || inferredNamaItem,
|
||||
expiry_date: existing.expiry_date || siblingExpiry
|
||||
});
|
||||
}
|
||||
|
||||
// Return empty default state if not found, with inferred metadata
|
||||
return NextResponse.json({
|
||||
filename: safeFilename,
|
||||
no_sku: inferredSku,
|
||||
nama_item: inferredNamaItem,
|
||||
expiry_date: siblingExpiry,
|
||||
top1_confidence: null,
|
||||
notes: "",
|
||||
saved_at: ""
|
||||
});
|
||||
} catch (err: unknown) {
|
||||
console.error("Error in GET manual-label-scan:", err);
|
||||
const message = err instanceof Error ? err.message : "Failed to load product label";
|
||||
return errorResponse(500, message);
|
||||
}
|
||||
}
|
||||
|
||||
export async function POST(req: NextRequest) {
|
||||
try {
|
||||
const body = await req.json();
|
||||
const { filename, image } = body;
|
||||
|
||||
let safeFilename = sanitizeFilename(filename || "unknown.jpg");
|
||||
|
||||
if (image && image.startsWith("data:image/")) {
|
||||
const base64Data = image.replace(/^data:image\/\w+;base64,/, "");
|
||||
const buffer = Buffer.from(base64Data, "base64");
|
||||
const hash = crypto.createHash("md5").update(buffer).digest("hex");
|
||||
const ext = image.match(/^data:image\/(\w+);base64,/)?.[1] || "jpg";
|
||||
safeFilename = `${hash}.${ext}`;
|
||||
|
||||
const saveDir = path.join(process.cwd(), "..", "sources", "product-test-images");
|
||||
if (!fs.existsSync(saveDir)) {
|
||||
fs.mkdirSync(saveDir, { recursive: true });
|
||||
}
|
||||
fs.writeFileSync(path.join(saveDir, safeFilename), buffer);
|
||||
}
|
||||
|
||||
if (!safeFilename || safeFilename === "unknown.jpg") {
|
||||
return errorResponse(400, "Filename or valid image is required in request body");
|
||||
}
|
||||
|
||||
const labels = readLabels();
|
||||
const index = labels.findIndex((l) => l.filename === safeFilename);
|
||||
|
||||
const entry: ProductScanLabel = {
|
||||
filename: safeFilename,
|
||||
no_sku: body.no_sku || "",
|
||||
nama_item: body.nama_item || "",
|
||||
expiry_date: normalizeDateString(body.expiry_date || ""),
|
||||
top1_confidence: typeof body.top1_confidence === "number" ? body.top1_confidence : null,
|
||||
notes: body.notes || "",
|
||||
saved_at: new Date().toISOString()
|
||||
};
|
||||
|
||||
if (index >= 0) {
|
||||
labels[index] = entry;
|
||||
} else {
|
||||
labels.push(entry);
|
||||
}
|
||||
|
||||
writeLabels(labels);
|
||||
|
||||
return NextResponse.json({ success: true, filePath: LABELS_PATH, entry });
|
||||
} catch (err: unknown) {
|
||||
console.error("Error in POST manual-label-scan:", err);
|
||||
const message = err instanceof Error ? err.message : "Failed to save product label";
|
||||
return errorResponse(500, message);
|
||||
}
|
||||
}
|
||||
|
||||
export async function DELETE(req: NextRequest) {
|
||||
try {
|
||||
const { searchParams } = new URL(req.url);
|
||||
const filename = searchParams.get("filename");
|
||||
if (!filename) return errorResponse(400, "Filename parameter is required");
|
||||
|
||||
const safeFilename = sanitizeFilename(filename);
|
||||
const labels = readLabels();
|
||||
const filtered = labels.filter((l) => l.filename !== safeFilename);
|
||||
|
||||
writeLabels(filtered);
|
||||
return NextResponse.json({ success: true });
|
||||
} catch (err: unknown) {
|
||||
const message = err instanceof Error ? err.message : "Failed to delete label";
|
||||
return errorResponse(500, message);
|
||||
}
|
||||
}
|
||||
@@ -1,147 +1,147 @@
|
||||
import { NextRequest, NextResponse } from "next/server";
|
||||
import { query } from "../../../db";
|
||||
import fs from "fs";
|
||||
import path from "path";
|
||||
import { errorResponse } from "@/utils/api-error";
|
||||
|
||||
const LABELS_PATH = path.join(process.cwd(), "..", "sources", "manual_labels.json");
|
||||
|
||||
function readLabels(): any[] {
|
||||
if (!fs.existsSync(LABELS_PATH)) {
|
||||
return [];
|
||||
}
|
||||
const raw = fs.readFileSync(LABELS_PATH, "utf8");
|
||||
return raw.trim() ? JSON.parse(raw) : [];
|
||||
}
|
||||
|
||||
function writeLabels(labels: any[]) {
|
||||
fs.writeFileSync(LABELS_PATH, JSON.stringify(labels, null, 2), "utf8");
|
||||
}
|
||||
|
||||
const SCALAR_FIELDS = ["noPO", "noSO", "noDO", "tanggal", "customer", "store", "alamat", "plat"] as const;
|
||||
|
||||
// Fetches the latest automated parser result for a filename, in the same
|
||||
// shape as a manual_labels.json entry, so it can be used as fill-in data.
|
||||
async function fetchLatestParsed(safeFilename: string): Promise<Record<string, any> | null> {
|
||||
try {
|
||||
const docRes = await query(
|
||||
"SELECT id, metadata FROM documents WHERE filename = $1",
|
||||
[safeFilename]
|
||||
);
|
||||
|
||||
if (!docRes.rowCount || docRes.rowCount === 0) return null;
|
||||
|
||||
const doc = docRes.rows[0];
|
||||
const meta = doc.metadata || {};
|
||||
|
||||
const itemsRes = await query(
|
||||
"SELECT kode_barang, nama_barang, banyak, jumlah FROM ocr_items WHERE document_id = $1 ORDER BY row_index",
|
||||
[doc.id]
|
||||
);
|
||||
|
||||
return {
|
||||
noPO: meta.noPO || "",
|
||||
noSO: meta.noSO || "",
|
||||
noDO: meta.noDO || "",
|
||||
tanggal: meta.tanggal || "",
|
||||
customer: meta.customerInfo || "",
|
||||
store: meta.orderUntuk || "",
|
||||
alamat: meta.alamat || "",
|
||||
plat: meta.platTruk || "",
|
||||
items: itemsRes.rows.map(row => ({
|
||||
kodeBarang: row.kode_barang || "",
|
||||
namaBarang: row.nama_barang || "",
|
||||
banyak: row.banyak || "",
|
||||
jumlah: row.jumlah || ""
|
||||
}))
|
||||
};
|
||||
} catch (dbErr) {
|
||||
console.error("DB fallback failed inside manual-label GET:", dbErr);
|
||||
return null;
|
||||
}
|
||||
}
|
||||
|
||||
export async function GET(req: NextRequest) {
|
||||
try {
|
||||
const { searchParams } = new URL(req.url);
|
||||
const filename = searchParams.get("filename");
|
||||
|
||||
if (!filename) {
|
||||
return errorResponse(400, "Filename parameter is required");
|
||||
}
|
||||
|
||||
const safeFilename = path.basename(filename);
|
||||
const labels = readLabels();
|
||||
const existing = labels.find(l => l.filename === safeFilename);
|
||||
const latest = await fetchLatestParsed(safeFilename);
|
||||
|
||||
if (existing) {
|
||||
// Never overwrite a field the user already corrected manually - only
|
||||
// fill in whatever is still blank, using the latest AI/DB parse.
|
||||
const merged = { ...existing, filename: safeFilename };
|
||||
if (latest) {
|
||||
for (const field of SCALAR_FIELDS) {
|
||||
if (!merged[field]) merged[field] = latest[field];
|
||||
}
|
||||
if (!merged.items || merged.items.length === 0) {
|
||||
merged.items = latest.items;
|
||||
}
|
||||
}
|
||||
// aiPredicted is the raw AI value for every field, always included
|
||||
// (even when a manual value already exists) so the UI can show what
|
||||
// the AI actually predicted next to the current/manual value.
|
||||
return NextResponse.json({ ...merged, aiPredicted: latest });
|
||||
}
|
||||
|
||||
if (latest) {
|
||||
return NextResponse.json({ filename, ...latest, aiPredicted: latest });
|
||||
}
|
||||
|
||||
// Return empty default state if not found anywhere
|
||||
return NextResponse.json({
|
||||
filename,
|
||||
noPO: "",
|
||||
noSO: "",
|
||||
noDO: "",
|
||||
tanggal: "",
|
||||
customer: "",
|
||||
store: "",
|
||||
alamat: "",
|
||||
plat: "",
|
||||
items: [],
|
||||
aiPredicted: null
|
||||
});
|
||||
} catch (err: any) {
|
||||
console.error("Error in GET manual-label:", err);
|
||||
return errorResponse(500, err.message || "Failed to load manual label");
|
||||
}
|
||||
}
|
||||
|
||||
export async function POST(req: NextRequest) {
|
||||
try {
|
||||
const body = await req.json();
|
||||
const { filename } = body;
|
||||
|
||||
if (!filename) {
|
||||
return errorResponse(400, "Filename is required in request body");
|
||||
}
|
||||
|
||||
const safeFilename = path.basename(filename);
|
||||
const labels = readLabels();
|
||||
const index = labels.findIndex(l => l.filename === safeFilename);
|
||||
const entry = { ...body, filename: safeFilename };
|
||||
|
||||
if (index >= 0) {
|
||||
labels[index] = entry;
|
||||
} else {
|
||||
labels.push(entry);
|
||||
}
|
||||
|
||||
writeLabels(labels);
|
||||
|
||||
return NextResponse.json({ success: true, filePath: LABELS_PATH });
|
||||
} catch (err: any) {
|
||||
console.error("Error in POST manual-label:", err);
|
||||
return errorResponse(500, err.message || "Failed to save manual label");
|
||||
}
|
||||
}
|
||||
import { NextRequest, NextResponse } from "next/server";
|
||||
import { query } from "../../../db";
|
||||
import fs from "fs";
|
||||
import path from "path";
|
||||
import { errorResponse } from "@/utils/api-error";
|
||||
|
||||
const LABELS_PATH = path.join(process.cwd(), "..", "sources", "manual_labels.json");
|
||||
|
||||
function readLabels(): any[] {
|
||||
if (!fs.existsSync(LABELS_PATH)) {
|
||||
return [];
|
||||
}
|
||||
const raw = fs.readFileSync(LABELS_PATH, "utf8");
|
||||
return raw.trim() ? JSON.parse(raw) : [];
|
||||
}
|
||||
|
||||
function writeLabels(labels: any[]) {
|
||||
fs.writeFileSync(LABELS_PATH, JSON.stringify(labels, null, 2), "utf8");
|
||||
}
|
||||
|
||||
const SCALAR_FIELDS = ["noPO", "noSO", "noDO", "tanggal", "customer", "store", "alamat", "plat"] as const;
|
||||
|
||||
// Fetches the latest automated parser result for a filename, in the same
|
||||
// shape as a manual_labels.json entry, so it can be used as fill-in data.
|
||||
async function fetchLatestParsed(safeFilename: string): Promise<Record<string, any> | null> {
|
||||
try {
|
||||
const docRes = await query(
|
||||
"SELECT id, metadata FROM documents WHERE filename = $1",
|
||||
[safeFilename]
|
||||
);
|
||||
|
||||
if (!docRes.rowCount || docRes.rowCount === 0) return null;
|
||||
|
||||
const doc = docRes.rows[0];
|
||||
const meta = doc.metadata || {};
|
||||
|
||||
const itemsRes = await query(
|
||||
"SELECT kode_barang, nama_barang, banyak, jumlah FROM ocr_items WHERE document_id = $1 ORDER BY row_index",
|
||||
[doc.id]
|
||||
);
|
||||
|
||||
return {
|
||||
noPO: meta.noPO || "",
|
||||
noSO: meta.noSO || "",
|
||||
noDO: meta.noDO || "",
|
||||
tanggal: meta.tanggal || "",
|
||||
customer: meta.customerInfo || "",
|
||||
store: meta.orderUntuk || "",
|
||||
alamat: meta.alamat || "",
|
||||
plat: meta.platTruk || "",
|
||||
items: itemsRes.rows.map(row => ({
|
||||
kodeBarang: row.kode_barang || "",
|
||||
namaBarang: row.nama_barang || "",
|
||||
banyak: row.banyak || "",
|
||||
jumlah: row.jumlah || ""
|
||||
}))
|
||||
};
|
||||
} catch (dbErr) {
|
||||
console.error("DB fallback failed inside manual-label GET:", dbErr);
|
||||
return null;
|
||||
}
|
||||
}
|
||||
|
||||
export async function GET(req: NextRequest) {
|
||||
try {
|
||||
const { searchParams } = new URL(req.url);
|
||||
const filename = searchParams.get("filename");
|
||||
|
||||
if (!filename) {
|
||||
return errorResponse(400, "Filename parameter is required");
|
||||
}
|
||||
|
||||
const safeFilename = path.basename(filename);
|
||||
const labels = readLabels();
|
||||
const existing = labels.find(l => l.filename === safeFilename);
|
||||
const latest = await fetchLatestParsed(safeFilename);
|
||||
|
||||
if (existing) {
|
||||
// Never overwrite a field the user already corrected manually - only
|
||||
// fill in whatever is still blank, using the latest AI/DB parse.
|
||||
const merged = { ...existing, filename: safeFilename };
|
||||
if (latest) {
|
||||
for (const field of SCALAR_FIELDS) {
|
||||
if (!merged[field]) merged[field] = latest[field];
|
||||
}
|
||||
if (!merged.items || merged.items.length === 0) {
|
||||
merged.items = latest.items;
|
||||
}
|
||||
}
|
||||
// aiPredicted is the raw AI value for every field, always included
|
||||
// (even when a manual value already exists) so the UI can show what
|
||||
// the AI actually predicted next to the current/manual value.
|
||||
return NextResponse.json({ ...merged, aiPredicted: latest });
|
||||
}
|
||||
|
||||
if (latest) {
|
||||
return NextResponse.json({ filename, ...latest, aiPredicted: latest });
|
||||
}
|
||||
|
||||
// Return empty default state if not found anywhere
|
||||
return NextResponse.json({
|
||||
filename,
|
||||
noPO: "",
|
||||
noSO: "",
|
||||
noDO: "",
|
||||
tanggal: "",
|
||||
customer: "",
|
||||
store: "",
|
||||
alamat: "",
|
||||
plat: "",
|
||||
items: [],
|
||||
aiPredicted: null
|
||||
});
|
||||
} catch (err: any) {
|
||||
console.error("Error in GET manual-label:", err);
|
||||
return errorResponse(500, err.message || "Failed to load manual label");
|
||||
}
|
||||
}
|
||||
|
||||
export async function POST(req: NextRequest) {
|
||||
try {
|
||||
const body = await req.json();
|
||||
const { filename } = body;
|
||||
|
||||
if (!filename) {
|
||||
return errorResponse(400, "Filename is required in request body");
|
||||
}
|
||||
|
||||
const safeFilename = path.basename(filename);
|
||||
const labels = readLabels();
|
||||
const index = labels.findIndex(l => l.filename === safeFilename);
|
||||
const entry = { ...body, filename: safeFilename };
|
||||
|
||||
if (index >= 0) {
|
||||
labels[index] = entry;
|
||||
} else {
|
||||
labels.push(entry);
|
||||
}
|
||||
|
||||
writeLabels(labels);
|
||||
|
||||
return NextResponse.json({ success: true, filePath: LABELS_PATH });
|
||||
} catch (err: any) {
|
||||
console.error("Error in POST manual-label:", err);
|
||||
return errorResponse(500, err.message || "Failed to save manual label");
|
||||
}
|
||||
}
|
||||
File diff suppressed because it is too large.
Load diff
@@ -1,54 +1,54 @@
|
||||
import { NextRequest, NextResponse } from "next/server";
|
||||
import fs from "fs";
|
||||
import path from "path";
|
||||
import { errorResponse } from "@/utils/api-error";
|
||||
|
||||
export const dynamic = "force-dynamic";
|
||||
|
||||
export async function GET(req: NextRequest) {
|
||||
try {
|
||||
const filename = req.nextUrl.searchParams.get("filename");
|
||||
const dirPath = path.join(process.cwd(), "..", "sources", "product-test-images-fixed");
|
||||
|
||||
// File serving mode
|
||||
if (filename) {
|
||||
const safeFile = path.basename(filename);
|
||||
const filePath = path.join(dirPath, safeFile);
|
||||
|
||||
if (!fs.existsSync(filePath)) {
|
||||
return errorResponse(404, "File not found");
|
||||
}
|
||||
|
||||
const ext = path.extname(safeFile).toLowerCase();
|
||||
let contentType = "application/octet-stream";
|
||||
if (ext === ".jpg" || ext === ".jpeg") contentType = "image/jpeg";
|
||||
else if (ext === ".png") contentType = "image/png";
|
||||
else if (ext === ".webp") contentType = "image/webp";
|
||||
|
||||
const fileBuffer = fs.readFileSync(filePath);
|
||||
return new Response(fileBuffer, {
|
||||
headers: {
|
||||
"Content-Type": contentType,
|
||||
"Cache-Control": "public, max-age=31536000, immutable"
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
// List mode
|
||||
if (!fs.existsSync(dirPath)) {
|
||||
return NextResponse.json({ files: [] });
|
||||
}
|
||||
|
||||
const files = fs.readdirSync(dirPath).filter((file) => {
|
||||
const ext = path.extname(file).toLowerCase();
|
||||
return [".jpg", ".jpeg", ".png", ".webp"].includes(ext);
|
||||
});
|
||||
|
||||
files.sort();
|
||||
return NextResponse.json({ files });
|
||||
} catch (error: unknown) {
|
||||
console.error("Error in product-images API:", error);
|
||||
const message = error instanceof Error ? error.message : "Internal server error";
|
||||
return errorResponse(500, message);
|
||||
}
|
||||
}
|
||||
import { NextRequest, NextResponse } from "next/server";
|
||||
import fs from "fs";
|
||||
import path from "path";
|
||||
import { errorResponse } from "@/utils/api-error";
|
||||
|
||||
export const dynamic = "force-dynamic";
|
||||
|
||||
export async function GET(req: NextRequest) {
|
||||
try {
|
||||
const filename = req.nextUrl.searchParams.get("filename");
|
||||
const dirPath = path.join(process.cwd(), "..", "sources", "product-test-images-fixed");
|
||||
|
||||
// File serving mode
|
||||
if (filename) {
|
||||
const safeFile = path.basename(filename);
|
||||
const filePath = path.join(dirPath, safeFile);
|
||||
|
||||
if (!fs.existsSync(filePath)) {
|
||||
return errorResponse(404, "File not found");
|
||||
}
|
||||
|
||||
const ext = path.extname(safeFile).toLowerCase();
|
||||
let contentType = "application/octet-stream";
|
||||
if (ext === ".jpg" || ext === ".jpeg") contentType = "image/jpeg";
|
||||
else if (ext === ".png") contentType = "image/png";
|
||||
else if (ext === ".webp") contentType = "image/webp";
|
||||
|
||||
const fileBuffer = fs.readFileSync(filePath);
|
||||
return new Response(fileBuffer, {
|
||||
headers: {
|
||||
"Content-Type": contentType,
|
||||
"Cache-Control": "public, max-age=31536000, immutable"
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
// List mode
|
||||
if (!fs.existsSync(dirPath)) {
|
||||
return NextResponse.json({ files: [] });
|
||||
}
|
||||
|
||||
const files = fs.readdirSync(dirPath).filter((file) => {
|
||||
const ext = path.extname(file).toLowerCase();
|
||||
return [".jpg", ".jpeg", ".png", ".webp"].includes(ext);
|
||||
});
|
||||
|
||||
files.sort();
|
||||
return NextResponse.json({ files });
|
||||
} catch (error: unknown) {
|
||||
console.error("Error in product-images API:", error);
|
||||
const message = error instanceof Error ? error.message : "Internal server error";
|
||||
return errorResponse(500, message);
|
||||
}
|
||||
}
|
||||
@@ -1,86 +1,86 @@
|
||||
import { NextRequest, NextResponse } from "next/server";
|
||||
import fs from "fs";
|
||||
import path from "path";
|
||||
import { errorResponse } from "@/utils/api-error";
|
||||
|
||||
// Serves the most recent accuracy-check-scan.mts detail dump
|
||||
// (sources/product_scan_detail_*.json) so the manual-label-scan page can show
|
||||
// what the AI actually predicted for a given Validation Set image by default,
|
||||
// without re-running the pipeline live for every image browsed. This is the
|
||||
// same predicted value the accuracy harness scores against ground truth -
|
||||
// not a fresh scan, so it reflects the last batch test run.
|
||||
const SOURCES_DIR = path.join(process.cwd(), "..", "sources");
|
||||
|
||||
interface DetailCheck {
|
||||
field: string;
|
||||
match: boolean;
|
||||
expected: string;
|
||||
predicted: string;
|
||||
}
|
||||
|
||||
interface DetailValidationItem {
|
||||
filename: string;
|
||||
method?: string;
|
||||
confidence?: number;
|
||||
checks: DetailCheck[];
|
||||
}
|
||||
|
||||
interface DetailDump {
|
||||
timestamp: string;
|
||||
validation: DetailValidationItem[];
|
||||
}
|
||||
|
||||
function findLatestDump(): { path: string; data: DetailDump } | null {
|
||||
if (!fs.existsSync(SOURCES_DIR)) return null;
|
||||
const candidates = fs
|
||||
.readdirSync(SOURCES_DIR)
|
||||
.filter((f) => /^product_scan_detail_.*\.json$/.test(f))
|
||||
.map((f) => {
|
||||
const p = path.join(SOURCES_DIR, f);
|
||||
return { path: p, mtime: fs.statSync(p).mtimeMs };
|
||||
})
|
||||
.sort((a, b) => b.mtime - a.mtime);
|
||||
|
||||
if (candidates.length === 0) return null;
|
||||
const latest = candidates[0];
|
||||
const data = JSON.parse(fs.readFileSync(latest.path, "utf8"));
|
||||
return { path: latest.path, data };
|
||||
}
|
||||
|
||||
export async function GET(req: NextRequest) {
|
||||
try {
|
||||
const { searchParams } = new URL(req.url);
|
||||
const filename = searchParams.get("filename");
|
||||
|
||||
const latest = findLatestDump();
|
||||
if (!latest) {
|
||||
return NextResponse.json({ available: false });
|
||||
}
|
||||
|
||||
if (!filename) {
|
||||
return NextResponse.json({ available: true, timestamp: latest.data.timestamp });
|
||||
}
|
||||
|
||||
const item = latest.data.validation.find((v) => v.filename === filename);
|
||||
if (!item) {
|
||||
return NextResponse.json({ available: true, timestamp: latest.data.timestamp, found: false });
|
||||
}
|
||||
|
||||
const byField = Object.fromEntries(item.checks.map((c) => [c.field, c]));
|
||||
|
||||
return NextResponse.json({
|
||||
available: true,
|
||||
found: true,
|
||||
timestamp: latest.data.timestamp,
|
||||
method: item.method,
|
||||
confidence: item.confidence,
|
||||
no_sku: byField.no_sku?.predicted,
|
||||
nama_item: byField.nama_item?.predicted,
|
||||
expiry_date: byField.expiry_date?.predicted
|
||||
});
|
||||
} catch (err: unknown) {
|
||||
console.error("Error in product-scan-results API:", err);
|
||||
const message = err instanceof Error ? err.message : "Internal server error";
|
||||
return errorResponse(500, message);
|
||||
}
|
||||
}
|
||||
import { NextRequest, NextResponse } from "next/server";
|
||||
import fs from "fs";
|
||||
import path from "path";
|
||||
import { errorResponse } from "@/utils/api-error";
|
||||
|
||||
// Serves the most recent accuracy-check-scan.mts detail dump
|
||||
// (sources/product_scan_detail_*.json) so the manual-label-scan page can show
|
||||
// what the AI actually predicted for a given Validation Set image by default,
|
||||
// without re-running the pipeline live for every image browsed. This is the
|
||||
// same predicted value the accuracy harness scores against ground truth -
|
||||
// not a fresh scan, so it reflects the last batch test run.
|
||||
const SOURCES_DIR = path.join(process.cwd(), "..", "sources");
|
||||
|
||||
interface DetailCheck {
|
||||
field: string;
|
||||
match: boolean;
|
||||
expected: string;
|
||||
predicted: string;
|
||||
}
|
||||
|
||||
interface DetailValidationItem {
|
||||
filename: string;
|
||||
method?: string;
|
||||
confidence?: number;
|
||||
checks: DetailCheck[];
|
||||
}
|
||||
|
||||
interface DetailDump {
|
||||
timestamp: string;
|
||||
validation: DetailValidationItem[];
|
||||
}
|
||||
|
||||
function findLatestDump(): { path: string; data: DetailDump } | null {
|
||||
if (!fs.existsSync(SOURCES_DIR)) return null;
|
||||
const candidates = fs
|
||||
.readdirSync(SOURCES_DIR)
|
||||
.filter((f) => /^product_scan_detail_.*\.json$/.test(f))
|
||||
.map((f) => {
|
||||
const p = path.join(SOURCES_DIR, f);
|
||||
return { path: p, mtime: fs.statSync(p).mtimeMs };
|
||||
})
|
||||
.sort((a, b) => b.mtime - a.mtime);
|
||||
|
||||
if (candidates.length === 0) return null;
|
||||
const latest = candidates[0];
|
||||
const data = JSON.parse(fs.readFileSync(latest.path, "utf8"));
|
||||
return { path: latest.path, data };
|
||||
}
|
||||
|
||||
export async function GET(req: NextRequest) {
|
||||
try {
|
||||
const { searchParams } = new URL(req.url);
|
||||
const filename = searchParams.get("filename");
|
||||
|
||||
const latest = findLatestDump();
|
||||
if (!latest) {
|
||||
return NextResponse.json({ available: false });
|
||||
}
|
||||
|
||||
if (!filename) {
|
||||
return NextResponse.json({ available: true, timestamp: latest.data.timestamp });
|
||||
}
|
||||
|
||||
const item = latest.data.validation.find((v) => v.filename === filename);
|
||||
if (!item) {
|
||||
return NextResponse.json({ available: true, timestamp: latest.data.timestamp, found: false });
|
||||
}
|
||||
|
||||
const byField = Object.fromEntries(item.checks.map((c) => [c.field, c]));
|
||||
|
||||
return NextResponse.json({
|
||||
available: true,
|
||||
found: true,
|
||||
timestamp: latest.data.timestamp,
|
||||
method: item.method,
|
||||
confidence: item.confidence,
|
||||
no_sku: byField.no_sku?.predicted,
|
||||
nama_item: byField.nama_item?.predicted,
|
||||
expiry_date: byField.expiry_date?.predicted
|
||||
});
|
||||
} catch (err: unknown) {
|
||||
console.error("Error in product-scan-results API:", err);
|
||||
const message = err instanceof Error ? err.message : "Internal server error";
|
||||
return errorResponse(500, message);
|
||||
}
|
||||
}
|
||||
@@ -1,50 +1,50 @@
|
||||
import { NextRequest, NextResponse } from "next/server";
|
||||
import fs from "fs";
|
||||
import path from "path";
|
||||
import { errorResponse } from "@/utils/api-error";
|
||||
|
||||
export const dynamic = "force-dynamic";
|
||||
|
||||
export async function GET(req: NextRequest) {
|
||||
try {
|
||||
const pfmDir = path.join(process.cwd(), "public", "produk-pfm", "foto-kemasan-v2");
|
||||
if (!fs.existsSync(pfmDir)) {
|
||||
return NextResponse.json({ products: [] });
|
||||
}
|
||||
|
||||
const entries = fs.readdirSync(pfmDir, { withFileTypes: true });
|
||||
const products = [];
|
||||
|
||||
const ignoredNames = ["models", "runs", "yolo_dataset", ".venv", ".venv-api"];
|
||||
|
||||
for (const entry of entries) {
|
||||
if (entry.isDirectory() && !ignoredNames.includes(entry.name)) {
|
||||
const productDirPath = path.join(pfmDir, entry.name);
|
||||
const files = fs.readdirSync(productDirPath);
|
||||
|
||||
// Filter image files
|
||||
const imageExtensions = [".jpg", ".jpeg", ".png", ".webp", ".bmp"];
|
||||
const images = files.filter(f =>
|
||||
imageExtensions.includes(path.extname(f).toLowerCase())
|
||||
);
|
||||
|
||||
if (images.length > 0) {
|
||||
products.push({
|
||||
productName: entry.name,
|
||||
images: images.map(img => `/produk-pfm/foto-kemasan-v2/${entry.name}/${img}`),
|
||||
thumbs: images.map(img => `/produk-pfm/thumbs/${entry.name}/${img}`)
|
||||
});
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Sort products by name
|
||||
products.sort((a, b) => a.productName.localeCompare(b.productName));
|
||||
|
||||
return NextResponse.json({ products });
|
||||
} catch (error: unknown) {
|
||||
console.error("Error fetching produk PFM:", error);
|
||||
const message = error instanceof Error ? error.message : "Internal server error";
|
||||
return errorResponse(500, message);
|
||||
}
|
||||
}
|
||||
import { NextRequest, NextResponse } from "next/server";
|
||||
import fs from "fs";
|
||||
import path from "path";
|
||||
import { errorResponse } from "@/utils/api-error";
|
||||
|
||||
export const dynamic = "force-dynamic";
|
||||
|
||||
export async function GET(req: NextRequest) {
|
||||
try {
|
||||
const pfmDir = path.join(process.cwd(), "public", "produk-pfm", "foto-kemasan-v2");
|
||||
if (!fs.existsSync(pfmDir)) {
|
||||
return NextResponse.json({ products: [] });
|
||||
}
|
||||
|
||||
const entries = fs.readdirSync(pfmDir, { withFileTypes: true });
|
||||
const products = [];
|
||||
|
||||
const ignoredNames = ["models", "runs", "yolo_dataset", ".venv", ".venv-api"];
|
||||
|
||||
for (const entry of entries) {
|
||||
if (entry.isDirectory() && !ignoredNames.includes(entry.name)) {
|
||||
const productDirPath = path.join(pfmDir, entry.name);
|
||||
const files = fs.readdirSync(productDirPath);
|
||||
|
||||
// Filter image files
|
||||
const imageExtensions = [".jpg", ".jpeg", ".png", ".webp", ".bmp"];
|
||||
const images = files.filter(f =>
|
||||
imageExtensions.includes(path.extname(f).toLowerCase())
|
||||
);
|
||||
|
||||
if (images.length > 0) {
|
||||
products.push({
|
||||
productName: entry.name,
|
||||
images: images.map(img => `/produk-pfm/foto-kemasan-v2/${entry.name}/${img}`),
|
||||
thumbs: images.map(img => `/produk-pfm/thumbs/${entry.name}/${img}`)
|
||||
});
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Sort products by name
|
||||
products.sort((a, b) => a.productName.localeCompare(b.productName));
|
||||
|
||||
return NextResponse.json({ products });
|
||||
} catch (error: unknown) {
|
||||
console.error("Error fetching produk PFM:", error);
|
||||
const message = error instanceof Error ? error.message : "Internal server error";
|
||||
return errorResponse(500, message);
|
||||
}
|
||||
}
|
||||
@@ -1,56 +1,56 @@
|
||||
import { NextRequest, NextResponse } from "next/server";
|
||||
import fs from "fs";
|
||||
import path from "path";
|
||||
import { query } from "../../../db";
|
||||
import crypto from "crypto";
|
||||
import { errorResponse } from "@/utils/api-error";
|
||||
|
||||
const UPLOADS_DIR = "/uploads";
|
||||
const PUBLIC_DIR = path.join(process.cwd(), "public/do-pfm");
|
||||
|
||||
export async function POST(req: NextRequest) {
|
||||
try {
|
||||
const { filename, image } = await req.json();
|
||||
|
||||
if (!filename || !image) {
|
||||
return errorResponse(400, "Filename and image base64 data are required");
|
||||
}
|
||||
|
||||
const safeFile = path.basename(filename);
|
||||
|
||||
const isSample = fs.existsSync(path.join(PUBLIC_DIR, safeFile));
|
||||
const filePath = isSample
|
||||
? path.join(PUBLIC_DIR, safeFile)
|
||||
: path.join(UPLOADS_DIR, safeFile);
|
||||
|
||||
const base64Data = image.replace(/^data:image\/\w+;base64,/, "");
|
||||
const buffer = Buffer.from(base64Data, "base64");
|
||||
|
||||
// Write file to disk
|
||||
fs.writeFileSync(filePath, buffer);
|
||||
console.log(`Rotated file saved successfully at ${filePath}`);
|
||||
|
||||
// Update database fields
|
||||
const fileHash = crypto.createHash("sha256").update(buffer).digest("hex");
|
||||
const stats = fs.statSync(filePath);
|
||||
|
||||
// Update document to unparsed state since layout changes
|
||||
await query(
|
||||
"UPDATE documents SET size = $1, file_hash = $2, parsed = false, layout_parsing_result = NULL WHERE filename = $3",
|
||||
[stats.size, fileHash, filename]
|
||||
);
|
||||
|
||||
// Clear old items for this document
|
||||
const docRes = await query("SELECT id FROM documents WHERE filename = $1", [filename]);
|
||||
if (docRes.rowCount && docRes.rowCount > 0) {
|
||||
const docId = docRes.rows[0].id;
|
||||
await query("DELETE FROM ocr_items WHERE document_id = $1", [docId]);
|
||||
}
|
||||
|
||||
return NextResponse.json({ success: true });
|
||||
} catch (error: unknown) {
|
||||
console.error("Error rotating file:", error);
|
||||
const message = error instanceof Error ? error.message : "Internal server error";
|
||||
return errorResponse(500, message);
|
||||
}
|
||||
}
|
||||
import { NextRequest, NextResponse } from "next/server";
|
||||
import fs from "fs";
|
||||
import path from "path";
|
||||
import { query } from "../../../db";
|
||||
import crypto from "crypto";
|
||||
import { errorResponse } from "@/utils/api-error";
|
||||
|
||||
const UPLOADS_DIR = "/uploads";
|
||||
const PUBLIC_DIR = path.join(process.cwd(), "public/do-pfm");
|
||||
|
||||
export async function POST(req: NextRequest) {
|
||||
try {
|
||||
const { filename, image } = await req.json();
|
||||
|
||||
if (!filename || !image) {
|
||||
return errorResponse(400, "Filename and image base64 data are required");
|
||||
}
|
||||
|
||||
const safeFile = path.basename(filename);
|
||||
|
||||
const isSample = fs.existsSync(path.join(PUBLIC_DIR, safeFile));
|
||||
const filePath = isSample
|
||||
? path.join(PUBLIC_DIR, safeFile)
|
||||
: path.join(UPLOADS_DIR, safeFile);
|
||||
|
||||
const base64Data = image.replace(/^data:image\/\w+;base64,/, "");
|
||||
const buffer = Buffer.from(base64Data, "base64");
|
||||
|
||||
// Write file to disk
|
||||
fs.writeFileSync(filePath, buffer);
|
||||
console.log(`Rotated file saved successfully at ${filePath}`);
|
||||
|
||||
// Update database fields
|
||||
const fileHash = crypto.createHash("sha256").update(buffer).digest("hex");
|
||||
const stats = fs.statSync(filePath);
|
||||
|
||||
// Update document to unparsed state since layout changes
|
||||
await query(
|
||||
"UPDATE documents SET size = $1, file_hash = $2, parsed = false, layout_parsing_result = NULL WHERE filename = $3",
|
||||
[stats.size, fileHash, filename]
|
||||
);
|
||||
|
||||
// Clear old items for this document
|
||||
const docRes = await query("SELECT id FROM documents WHERE filename = $1", [filename]);
|
||||
if (docRes.rowCount && docRes.rowCount > 0) {
|
||||
const docId = docRes.rows[0].id;
|
||||
await query("DELETE FROM ocr_items WHERE document_id = $1", [docId]);
|
||||
}
|
||||
|
||||
return NextResponse.json({ success: true });
|
||||
} catch (error: unknown) {
|
||||
console.error("Error rotating file:", error);
|
||||
const message = error instanceof Error ? error.message : "Internal server error";
|
||||
return errorResponse(500, message);
|
||||
}
|
||||
}
|
||||
@@ -1,31 +1,31 @@
|
||||
import { NextRequest, NextResponse } from "next/server";
|
||||
import { errorResponse } from "@/utils/api-error";
|
||||
import { classifyAndMatchProduct, ClassifierError } from "@/utils/product-scan";
|
||||
|
||||
export const dynamic = "force-dynamic";
|
||||
|
||||
export async function POST(req: NextRequest) {
|
||||
try {
|
||||
const body = await req.json();
|
||||
const image_base64 = body.image_base64 || body.image;
|
||||
if (!image_base64) {
|
||||
return errorResponse(400, "Image is required");
|
||||
}
|
||||
|
||||
const result = await classifyAndMatchProduct(image_base64);
|
||||
|
||||
return NextResponse.json({
|
||||
classification: result.classification,
|
||||
ocr: result.ocr,
|
||||
possibleMatches: result.possibleMatches
|
||||
});
|
||||
|
||||
} catch (error: unknown) {
|
||||
console.error("Error in scan-pfm API route:", error);
|
||||
if (error instanceof ClassifierError) {
|
||||
return errorResponse(error.status, error.message);
|
||||
}
|
||||
const message = error instanceof Error ? error.message : "Internal server error";
|
||||
return errorResponse(500, message);
|
||||
}
|
||||
}
|
||||
import { NextRequest, NextResponse } from "next/server";
|
||||
import { errorResponse } from "@/utils/api-error";
|
||||
import { classifyAndMatchProduct, ClassifierError } from "@/utils/product-scan";
|
||||
|
||||
export const dynamic = "force-dynamic";
|
||||
|
||||
export async function POST(req: NextRequest) {
|
||||
try {
|
||||
const body = await req.json();
|
||||
const image_base64 = body.image_base64 || body.image;
|
||||
if (!image_base64) {
|
||||
return errorResponse(400, "Image is required");
|
||||
}
|
||||
|
||||
const result = await classifyAndMatchProduct(image_base64);
|
||||
|
||||
return NextResponse.json({
|
||||
classification: result.classification,
|
||||
ocr: result.ocr,
|
||||
possibleMatches: result.possibleMatches
|
||||
});
|
||||
|
||||
} catch (error: unknown) {
|
||||
console.error("Error in scan-pfm API route:", error);
|
||||
if (error instanceof ClassifierError) {
|
||||
return errorResponse(error.status, error.message);
|
||||
}
|
||||
const message = error instanceof Error ? error.message : "Internal server error";
|
||||
return errorResponse(500, message);
|
||||
}
|
||||
}
|
||||
@@ -1,24 +1,24 @@
|
||||
import { NextRequest, NextResponse } from "next/server";
|
||||
import { query } from "../../../db";
|
||||
import { errorResponse } from "@/utils/api-error";
|
||||
|
||||
export const dynamic = "force-dynamic";
|
||||
|
||||
export async function GET(req: NextRequest) {
|
||||
try {
|
||||
const res = await query(
|
||||
"SELECT no_sku, nama_item FROM sku_master ORDER BY no_sku"
|
||||
);
|
||||
|
||||
const skus = res.rows.map(row => ({
|
||||
no_sku: row.no_sku,
|
||||
nama_item: row.nama_item
|
||||
}));
|
||||
|
||||
return NextResponse.json({ skus });
|
||||
} catch (error: unknown) {
|
||||
console.error("Error in SKUs API route:", error);
|
||||
const message = error instanceof Error ? error.message : "Internal server error";
|
||||
return errorResponse(500, message);
|
||||
}
|
||||
}
|
||||
import { NextRequest, NextResponse } from "next/server";
|
||||
import { query } from "../../../db";
|
||||
import { errorResponse } from "@/utils/api-error";
|
||||
|
||||
export const dynamic = "force-dynamic";
|
||||
|
||||
export async function GET(req: NextRequest) {
|
||||
try {
|
||||
const res = await query(
|
||||
"SELECT no_sku, nama_item FROM sku_master ORDER BY no_sku"
|
||||
);
|
||||
|
||||
const skus = res.rows.map(row => ({
|
||||
no_sku: row.no_sku,
|
||||
nama_item: row.nama_item
|
||||
}));
|
||||
|
||||
return NextResponse.json({ skus });
|
||||
} catch (error: unknown) {
|
||||
console.error("Error in SKUs API route:", error);
|
||||
const message = error instanceof Error ? error.message : "Internal server error";
|
||||
return errorResponse(500, message);
|
||||
}
|
||||
}
|
||||
@@ -1,25 +1,25 @@
|
||||
import { NextRequest, NextResponse } from "next/server";
|
||||
import { query } from "../../../db";
|
||||
import { errorResponse } from "@/utils/api-error";
|
||||
|
||||
export const dynamic = "force-dynamic";
|
||||
|
||||
export async function GET(req: NextRequest) {
|
||||
try {
|
||||
const res = await query(
|
||||
"SELECT kode_toko, nama_toko, alamat FROM store_master ORDER BY nama_toko"
|
||||
);
|
||||
|
||||
const stores = res.rows.map(row => ({
|
||||
kodeToko: row.kode_toko,
|
||||
namaToko: row.nama_toko,
|
||||
alamat: row.alamat
|
||||
}));
|
||||
|
||||
return NextResponse.json({ stores });
|
||||
} catch (error: unknown) {
|
||||
console.error("Error in stores API route:", error);
|
||||
const message = error instanceof Error ? error.message : "Internal server error";
|
||||
return errorResponse(500, message);
|
||||
}
|
||||
}
|
||||
import { NextRequest, NextResponse } from "next/server";
|
||||
import { query } from "../../../db";
|
||||
import { errorResponse } from "@/utils/api-error";
|
||||
|
||||
export const dynamic = "force-dynamic";
|
||||
|
||||
export async function GET(req: NextRequest) {
|
||||
try {
|
||||
const res = await query(
|
||||
"SELECT kode_toko, nama_toko, alamat FROM store_master ORDER BY nama_toko"
|
||||
);
|
||||
|
||||
const stores = res.rows.map(row => ({
|
||||
kodeToko: row.kode_toko,
|
||||
namaToko: row.nama_toko,
|
||||
alamat: row.alamat
|
||||
}));
|
||||
|
||||
return NextResponse.json({ stores });
|
||||
} catch (error: unknown) {
|
||||
console.error("Error in stores API route:", error);
|
||||
const message = error instanceof Error ? error.message : "Internal server error";
|
||||
return errorResponse(500, message);
|
||||
}
|
||||
}
|
||||
@@ -1,233 +1,233 @@
|
||||
import { NextRequest, NextResponse } from "next/server";
|
||||
import { query } from "../../../db";
|
||||
import { correctVisualDigits } from "../../../utils/parser";
|
||||
|
||||
export const dynamic = "force-dynamic";
|
||||
|
||||
function levenshteinDistance(s1: string, s2: string): number {
|
||||
const len1 = s1.length;
|
||||
const len2 = s2.length;
|
||||
const matrix = Array.from({ length: len1 + 1 }, () => new Array(len2 + 1).fill(0));
|
||||
|
||||
for (let i = 0; i <= len1; i++) matrix[i][0] = i;
|
||||
for (let j = 0; j <= len2; j++) matrix[0][j] = j;
|
||||
|
||||
for (let i = 1; i <= len1; i++) {
|
||||
for (let j = 1; j <= len2; j++) {
|
||||
const cost = s1[i - 1] === s2[j - 1] ? 0 : 1;
|
||||
matrix[i][j] = Math.min(
|
||||
matrix[i - 1][j] + 1, // deletion
|
||||
matrix[i][j - 1] + 1, // insertion
|
||||
matrix[i - 1][j - 1] + cost // substitution
|
||||
);
|
||||
}
|
||||
}
|
||||
return matrix[len1][len2];
|
||||
}
|
||||
|
||||
function getStringSimilarity(s1: string, s2: string): number {
|
||||
const clean1 = s1.toLowerCase().replace(/[^a-z0-9]/g, '');
|
||||
const clean2 = s2.toLowerCase().replace(/[^a-z0-9]/g, '');
|
||||
if (!clean1 || !clean2) return 0;
|
||||
const distance = levenshteinDistance(clean1, clean2);
|
||||
const maxLength = Math.max(clean1.length, clean2.length);
|
||||
return (maxLength - distance) / maxLength;
|
||||
}
|
||||
|
||||
// Re-implement cleanDateValue directly so we don't have to deal with exports issues if any
|
||||
const MONTHS_MAP: Record<string, string> = {
|
||||
january: "January", januari: "January", janov: "January", jan: "January",
|
||||
february: "February", februari: "February", feb: "February",
|
||||
march: "March", maret: "March", mar: "March",
|
||||
april: "April", apr: "April",
|
||||
may: "May", mei: "May",
|
||||
june: "June", juni: "June", jun: "June",
|
||||
july: "July", juli: "July", jul: "July",
|
||||
august: "August", agustus: "August", agt: "August", ags: "August", aug: "August",
|
||||
september: "September", sept: "September", sep: "September",
|
||||
oktober: "October", october: "October", okt: "October", oct: "October",
|
||||
november: "November", nopember: "November", nov: "November",
|
||||
desember: "December", december: "December", des: "December", dec: "December"
|
||||
};
|
||||
|
||||
function cleanDateValue(raw: string): string {
|
||||
if (!raw) return "Not Found";
|
||||
const cleaned = raw.trim();
|
||||
if (cleaned === "Not Found" || cleaned === "") return "Not Found";
|
||||
|
||||
const today = new Date();
|
||||
let day: number | null = null;
|
||||
let monthStr: string | null = null;
|
||||
let year: number | null = null;
|
||||
|
||||
const yearMatch = cleaned.match(/\b(20\d{2})\b/);
|
||||
if (yearMatch) {
|
||||
const parsedYear = parseInt(yearMatch[1], 10);
|
||||
if (parsedYear >= 2010 && parsedYear <= 2035) {
|
||||
year = parsedYear;
|
||||
}
|
||||
}
|
||||
|
||||
const lowerRaw = cleaned.toLowerCase();
|
||||
const monthsKeys = Object.keys(MONTHS_MAP);
|
||||
monthsKeys.sort((a, b) => b.length - a.length);
|
||||
|
||||
for (const key of monthsKeys) {
|
||||
if (lowerRaw.includes(key)) {
|
||||
monthStr = MONTHS_MAP[key] || null;
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
let textForDay = cleaned;
|
||||
if (year) {
|
||||
textForDay = textForDay.replace(year.toString(), "");
|
||||
}
|
||||
const dayMatches = textForDay.match(/\b(\d{1,2})\b/g);
|
||||
if (dayMatches) {
|
||||
for (const matchStr of dayMatches) {
|
||||
const parsedDay = parseInt(matchStr, 10);
|
||||
if (parsedDay >= 1 && parsedDay <= 31) {
|
||||
day = parsedDay;
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
const currentYear = today.getFullYear();
|
||||
const currentMonthNames = ["January", "February", "March", "April", "May", "June", "July", "August", "September", "October", "November", "December"];
|
||||
const currentMonth = currentMonthNames[today.getMonth()];
|
||||
const currentDay = today.getDate();
|
||||
|
||||
const finalDay = day !== null ? day : currentDay;
|
||||
const finalMonth = monthStr !== null ? monthStr : currentMonth;
|
||||
const finalYear = year !== null ? year : currentYear;
|
||||
|
||||
return `${finalDay} ${finalMonth} ${finalYear}`;
|
||||
}
|
||||
|
||||
export async function GET(req: NextRequest) {
|
||||
const results: string[] = [];
|
||||
let passed = true;
|
||||
|
||||
const assert = (condition: boolean, desc: string) => {
|
||||
if (condition) {
|
||||
results.push(`[PASS] ${desc}`);
|
||||
} else {
|
||||
results.push(`[FAIL] ${desc}`);
|
||||
passed = false;
|
||||
}
|
||||
};
|
||||
|
||||
// 1. Test Visual Digit Correction
|
||||
const so1 = correctVisualDigits("16O29B7162");
|
||||
assert(so1 === "1602987162", `correctVisualDigits("16O29B7162") -> got "${so1}", expected "1602987162"`);
|
||||
|
||||
const do1 = correctVisualDigits("1602l87");
|
||||
assert(do1 === "1602187", `correctVisualDigits("1602l87") -> got "${do1}", expected "1602187"`);
|
||||
|
||||
const so2 = correctVisualDigits("16O29B7162-OK");
|
||||
assert(so2 === "1602987162", `correctVisualDigits("16O29B7162-OK") -> got "${so2}", expected "1602987162"`);
|
||||
|
||||
// 2. Test Date Lenient Parsing & Fallback Auto-Fill
|
||||
const today = new Date();
|
||||
const currentMonthNames = ["January", "February", "March", "April", "May", "June", "July", "August", "September", "October", "November", "December"];
|
||||
const currentMonth = currentMonthNames[today.getMonth()];
|
||||
const currentDay = today.getDate();
|
||||
const currentYear = today.getFullYear();
|
||||
|
||||
const d1 = cleanDateValue("30-Hv-2026");
|
||||
assert(d1 === `30 ${currentMonth} 2026`, `cleanDateValue("30-Hv-2026") -> got "${d1}", expected "30 ${currentMonth} 2026"`);
|
||||
|
||||
const d2 = cleanDateValue("Hv-Jan-2026");
|
||||
assert(d2 === `${currentDay} January 2026`, `cleanDateValue("Hv-Jan-2026") -> got "${d2}", expected "${currentDay} January 2026"`);
|
||||
|
||||
const d3 = cleanDateValue("30-Jan");
|
||||
assert(d3 === `30 January ${currentYear}`, `cleanDateValue("30-Jan") -> got "${d3}", expected "30 January ${currentYear}"`);
|
||||
|
||||
// 3. Test Two-Way Database SKU Cross-Check
|
||||
try {
|
||||
const skuDbRes = await query("SELECT no_sku, nama_item FROM sku_master");
|
||||
const skuMasterList = skuDbRes.rows.map(row => ({
|
||||
no_sku: row.no_sku.toString().trim(),
|
||||
nama_item: row.nama_item.toString().trim()
|
||||
}));
|
||||
|
||||
// Mock an OCR parsed items list
|
||||
const items = [
|
||||
{
|
||||
kodeBarang: "11048006",
|
||||
namaBarang: "BEBEK PARTING wrong ocr text",
|
||||
banyak: "10 BAG",
|
||||
jumlah: "100000"
|
||||
},
|
||||
{
|
||||
kodeBarang: "Not Found",
|
||||
namaBarang: "CEKER BERKUKU FROZEN PACK",
|
||||
banyak: "20 KRG",
|
||||
jumlah: "200000"
|
||||
},
|
||||
{
|
||||
kodeBarang: "Not Found",
|
||||
namaBarang: "Tanda Tangan Supit",
|
||||
banyak: "Bag. Pengeluaran Barang",
|
||||
jumlah: "Bagian Penjualan"
|
||||
}
|
||||
];
|
||||
|
||||
const checkedItems: typeof items = [];
|
||||
for (const item of items) {
|
||||
const ocrSku = item.kodeBarang ? item.kodeBarang.trim() : "";
|
||||
const ocrName = item.namaBarang ? item.namaBarang.trim() : "";
|
||||
|
||||
const matchedBySku = /^\d{8}$/.test(ocrSku) ? skuMasterList.find(sku => sku.no_sku === ocrSku) : null;
|
||||
|
||||
if (matchedBySku) {
|
||||
item.kodeBarang = matchedBySku.no_sku;
|
||||
item.namaBarang = matchedBySku.nama_item;
|
||||
checkedItems.push(item);
|
||||
} else {
|
||||
let bestMatch: typeof skuMasterList[0] | null = null;
|
||||
let bestScore = 0;
|
||||
|
||||
for (const sku of skuMasterList) {
|
||||
const score = getStringSimilarity(sku.nama_item, ocrName);
|
||||
if (score > bestScore) {
|
||||
bestScore = score;
|
||||
bestMatch = sku;
|
||||
}
|
||||
}
|
||||
|
||||
if (bestMatch && bestScore >= 0.6) {
|
||||
item.kodeBarang = bestMatch.no_sku;
|
||||
item.namaBarang = bestMatch.nama_item;
|
||||
checkedItems.push(item);
|
||||
} else {
|
||||
if (/^\d{8}$/.test(ocrSku)) {
|
||||
checkedItems.push(item);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Verify checkedItems length (noise item discarded)
|
||||
assert(checkedItems.length === 2, `checkedItems length should be 2, got ${checkedItems.length} (noise footer row successfully discarded)`);
|
||||
|
||||
// Verify item 1 description correction
|
||||
assert(checkedItems[0].kodeBarang === "11048006", "Item 1 SKU should remain 11048006");
|
||||
assert(checkedItems[0].namaBarang === "BEBEK PARTING-NEW(*)", `Item 1 name corrected from DB -> got "${checkedItems[0].namaBarang}"`);
|
||||
|
||||
// Verify item 2 SKU fuzzy autocomplete from description
|
||||
assert(checkedItems[1].kodeBarang === "11110059", `Item 2 SKU autocompleted from DB -> got "${checkedItems[1].kodeBarang}"`);
|
||||
assert(checkedItems[1].namaBarang === "CEKER BERKUKU FROZEN PACK 1 KG(*)", `Item 2 name corrected from DB -> got "${checkedItems[1].namaBarang}"`);
|
||||
|
||||
} catch (err: any) {
|
||||
passed = false;
|
||||
results.push(`[ERROR] Database SKU check failed: ${err.message}`);
|
||||
}
|
||||
|
||||
return NextResponse.json({
|
||||
status: passed ? "success" : "failed",
|
||||
results
|
||||
});
|
||||
}
|
||||
import { NextRequest, NextResponse } from "next/server";
|
||||
import { query } from "../../../db";
|
||||
import { correctVisualDigits } from "../../../utils/parser";
|
||||
|
||||
export const dynamic = "force-dynamic";
|
||||
|
||||
function levenshteinDistance(s1: string, s2: string): number {
|
||||
const len1 = s1.length;
|
||||
const len2 = s2.length;
|
||||
const matrix = Array.from({ length: len1 + 1 }, () => new Array(len2 + 1).fill(0));
|
||||
|
||||
for (let i = 0; i <= len1; i++) matrix[i][0] = i;
|
||||
for (let j = 0; j <= len2; j++) matrix[0][j] = j;
|
||||
|
||||
for (let i = 1; i <= len1; i++) {
|
||||
for (let j = 1; j <= len2; j++) {
|
||||
const cost = s1[i - 1] === s2[j - 1] ? 0 : 1;
|
||||
matrix[i][j] = Math.min(
|
||||
matrix[i - 1][j] + 1, // deletion
|
||||
matrix[i][j - 1] + 1, // insertion
|
||||
matrix[i - 1][j - 1] + cost // substitution
|
||||
);
|
||||
}
|
||||
}
|
||||
return matrix[len1][len2];
|
||||
}
|
||||
|
||||
function getStringSimilarity(s1: string, s2: string): number {
|
||||
const clean1 = s1.toLowerCase().replace(/[^a-z0-9]/g, '');
|
||||
const clean2 = s2.toLowerCase().replace(/[^a-z0-9]/g, '');
|
||||
if (!clean1 || !clean2) return 0;
|
||||
const distance = levenshteinDistance(clean1, clean2);
|
||||
const maxLength = Math.max(clean1.length, clean2.length);
|
||||
return (maxLength - distance) / maxLength;
|
||||
}
|
||||
|
||||
// Re-implement cleanDateValue directly so we don't have to deal with exports issues if any
|
||||
const MONTHS_MAP: Record<string, string> = {
|
||||
january: "January", januari: "January", janov: "January", jan: "January",
|
||||
february: "February", februari: "February", feb: "February",
|
||||
march: "March", maret: "March", mar: "March",
|
||||
april: "April", apr: "April",
|
||||
may: "May", mei: "May",
|
||||
june: "June", juni: "June", jun: "June",
|
||||
july: "July", juli: "July", jul: "July",
|
||||
august: "August", agustus: "August", agt: "August", ags: "August", aug: "August",
|
||||
september: "September", sept: "September", sep: "September",
|
||||
oktober: "October", october: "October", okt: "October", oct: "October",
|
||||
november: "November", nopember: "November", nov: "November",
|
||||
desember: "December", december: "December", des: "December", dec: "December"
|
||||
};
|
||||
|
||||
function cleanDateValue(raw: string): string {
|
||||
if (!raw) return "Not Found";
|
||||
const cleaned = raw.trim();
|
||||
if (cleaned === "Not Found" || cleaned === "") return "Not Found";
|
||||
|
||||
const today = new Date();
|
||||
let day: number | null = null;
|
||||
let monthStr: string | null = null;
|
||||
let year: number | null = null;
|
||||
|
||||
const yearMatch = cleaned.match(/\b(20\d{2})\b/);
|
||||
if (yearMatch) {
|
||||
const parsedYear = parseInt(yearMatch[1], 10);
|
||||
if (parsedYear >= 2010 && parsedYear <= 2035) {
|
||||
year = parsedYear;
|
||||
}
|
||||
}
|
||||
|
||||
const lowerRaw = cleaned.toLowerCase();
|
||||
const monthsKeys = Object.keys(MONTHS_MAP);
|
||||
monthsKeys.sort((a, b) => b.length - a.length);
|
||||
|
||||
for (const key of monthsKeys) {
|
||||
if (lowerRaw.includes(key)) {
|
||||
monthStr = MONTHS_MAP[key] || null;
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
let textForDay = cleaned;
|
||||
if (year) {
|
||||
textForDay = textForDay.replace(year.toString(), "");
|
||||
}
|
||||
const dayMatches = textForDay.match(/\b(\d{1,2})\b/g);
|
||||
if (dayMatches) {
|
||||
for (const matchStr of dayMatches) {
|
||||
const parsedDay = parseInt(matchStr, 10);
|
||||
if (parsedDay >= 1 && parsedDay <= 31) {
|
||||
day = parsedDay;
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
const currentYear = today.getFullYear();
|
||||
const currentMonthNames = ["January", "February", "March", "April", "May", "June", "July", "August", "September", "October", "November", "December"];
|
||||
const currentMonth = currentMonthNames[today.getMonth()];
|
||||
const currentDay = today.getDate();
|
||||
|
||||
const finalDay = day !== null ? day : currentDay;
|
||||
const finalMonth = monthStr !== null ? monthStr : currentMonth;
|
||||
const finalYear = year !== null ? year : currentYear;
|
||||
|
||||
return `${finalDay} ${finalMonth} ${finalYear}`;
|
||||
}
|
||||
|
||||
export async function GET(req: NextRequest) {
|
||||
const results: string[] = [];
|
||||
let passed = true;
|
||||
|
||||
const assert = (condition: boolean, desc: string) => {
|
||||
if (condition) {
|
||||
results.push(`[PASS] ${desc}`);
|
||||
} else {
|
||||
results.push(`[FAIL] ${desc}`);
|
||||
passed = false;
|
||||
}
|
||||
};
|
||||
|
||||
// 1. Test Visual Digit Correction
|
||||
const so1 = correctVisualDigits("16O29B7162");
|
||||
assert(so1 === "1602987162", `correctVisualDigits("16O29B7162") -> got "${so1}", expected "1602987162"`);
|
||||
|
||||
const do1 = correctVisualDigits("1602l87");
|
||||
assert(do1 === "1602187", `correctVisualDigits("1602l87") -> got "${do1}", expected "1602187"`);
|
||||
|
||||
const so2 = correctVisualDigits("16O29B7162-OK");
|
||||
assert(so2 === "1602987162", `correctVisualDigits("16O29B7162-OK") -> got "${so2}", expected "1602987162"`);
|
||||
|
||||
// 2. Test Date Lenient Parsing & Fallback Auto-Fill
|
||||
const today = new Date();
|
||||
const currentMonthNames = ["January", "February", "March", "April", "May", "June", "July", "August", "September", "October", "November", "December"];
|
||||
const currentMonth = currentMonthNames[today.getMonth()];
|
||||
const currentDay = today.getDate();
|
||||
const currentYear = today.getFullYear();
|
||||
|
||||
const d1 = cleanDateValue("30-Hv-2026");
|
||||
assert(d1 === `30 ${currentMonth} 2026`, `cleanDateValue("30-Hv-2026") -> got "${d1}", expected "30 ${currentMonth} 2026"`);
|
||||
|
||||
const d2 = cleanDateValue("Hv-Jan-2026");
|
||||
assert(d2 === `${currentDay} January 2026`, `cleanDateValue("Hv-Jan-2026") -> got "${d2}", expected "${currentDay} January 2026"`);
|
||||
|
||||
const d3 = cleanDateValue("30-Jan");
|
||||
assert(d3 === `30 January ${currentYear}`, `cleanDateValue("30-Jan") -> got "${d3}", expected "30 January ${currentYear}"`);
|
||||
|
||||
// 3. Test Two-Way Database SKU Cross-Check
|
||||
try {
|
||||
const skuDbRes = await query("SELECT no_sku, nama_item FROM sku_master");
|
||||
const skuMasterList = skuDbRes.rows.map(row => ({
|
||||
no_sku: row.no_sku.toString().trim(),
|
||||
nama_item: row.nama_item.toString().trim()
|
||||
}));
|
||||
|
||||
// Mock an OCR parsed items list
|
||||
const items = [
|
||||
{
|
||||
kodeBarang: "11048006",
|
||||
namaBarang: "BEBEK PARTING wrong ocr text",
|
||||
banyak: "10 BAG",
|
||||
jumlah: "100000"
|
||||
},
|
||||
{
|
||||
kodeBarang: "Not Found",
|
||||
namaBarang: "CEKER BERKUKU FROZEN PACK",
|
||||
banyak: "20 KRG",
|
||||
jumlah: "200000"
|
||||
},
|
||||
{
|
||||
kodeBarang: "Not Found",
|
||||
namaBarang: "Tanda Tangan Supit",
|
||||
banyak: "Bag. Pengeluaran Barang",
|
||||
jumlah: "Bagian Penjualan"
|
||||
}
|
||||
];
|
||||
|
||||
const checkedItems: typeof items = [];
|
||||
for (const item of items) {
|
||||
const ocrSku = item.kodeBarang ? item.kodeBarang.trim() : "";
|
||||
const ocrName = item.namaBarang ? item.namaBarang.trim() : "";
|
||||
|
||||
const matchedBySku = /^\d{8}$/.test(ocrSku) ? skuMasterList.find(sku => sku.no_sku === ocrSku) : null;
|
||||
|
||||
if (matchedBySku) {
|
||||
item.kodeBarang = matchedBySku.no_sku;
|
||||
item.namaBarang = matchedBySku.nama_item;
|
||||
checkedItems.push(item);
|
||||
} else {
|
||||
let bestMatch: typeof skuMasterList[0] | null = null;
|
||||
let bestScore = 0;
|
||||
|
||||
for (const sku of skuMasterList) {
|
||||
const score = getStringSimilarity(sku.nama_item, ocrName);
|
||||
if (score > bestScore) {
|
||||
bestScore = score;
|
||||
bestMatch = sku;
|
||||
}
|
||||
}
|
||||
|
||||
if (bestMatch && bestScore >= 0.6) {
|
||||
item.kodeBarang = bestMatch.no_sku;
|
||||
item.namaBarang = bestMatch.nama_item;
|
||||
checkedItems.push(item);
|
||||
} else {
|
||||
if (/^\d{8}$/.test(ocrSku)) {
|
||||
checkedItems.push(item);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Verify checkedItems length (noise item discarded)
|
||||
assert(checkedItems.length === 2, `checkedItems length should be 2, got ${checkedItems.length} (noise footer row successfully discarded)`);
|
||||
|
||||
// Verify item 1 description correction
|
||||
assert(checkedItems[0].kodeBarang === "11048006", "Item 1 SKU should remain 11048006");
|
||||
assert(checkedItems[0].namaBarang === "BEBEK PARTING-NEW(*)", `Item 1 name corrected from DB -> got "${checkedItems[0].namaBarang}"`);
|
||||
|
||||
// Verify item 2 SKU fuzzy autocomplete from description
|
||||
assert(checkedItems[1].kodeBarang === "11110059", `Item 2 SKU autocompleted from DB -> got "${checkedItems[1].kodeBarang}"`);
|
||||
assert(checkedItems[1].namaBarang === "CEKER BERKUKU FROZEN PACK 1 KG(*)", `Item 2 name corrected from DB -> got "${checkedItems[1].namaBarang}"`);
|
||||
|
||||
} catch (err: any) {
|
||||
passed = false;
|
||||
results.push(`[ERROR] Database SKU check failed: ${err.message}`);
|
||||
}
|
||||
|
||||
return NextResponse.json({
|
||||
status: passed ? "success" : "failed",
|
||||
results
|
||||
});
|
||||
}
|
||||
@@ -1,72 +1,72 @@
|
||||
import { NextRequest, NextResponse } from "next/server";
|
||||
import { query } from "../../../db";
|
||||
import path from "path";
|
||||
import { errorResponse } from "@/utils/api-error";
|
||||
|
||||
export async function POST(req: NextRequest) {
|
||||
try {
|
||||
const body = await req.json();
|
||||
const { page, rowIndex, action } = body;
|
||||
|
||||
if (!page || rowIndex === undefined || !action) {
|
||||
return errorResponse(400, "Missing required fields");
|
||||
}
|
||||
|
||||
const safeFile = path.basename(page);
|
||||
|
||||
// Get document ID
|
||||
const docRes = await query("SELECT id FROM documents WHERE filename = $1", [safeFile]);
|
||||
if (!docRes.rowCount || docRes.rowCount === 0) {
|
||||
return errorResponse(404, "Document not found in database");
|
||||
}
|
||||
const docId = docRes.rows[0].id;
|
||||
|
||||
if (action === "edit") {
|
||||
const { field, value } = body;
|
||||
if (!field || value === undefined) {
|
||||
return errorResponse(400, "Missing edit parameters");
|
||||
}
|
||||
|
||||
// Map UI field names to database columns
|
||||
let colName = "";
|
||||
if (field === "kodeBarang") {
|
||||
colName = "kode_barang";
|
||||
} else if (field === "banyak") {
|
||||
colName = "banyak";
|
||||
} else if (field === "jumlah") {
|
||||
colName = "jumlah";
|
||||
} else {
|
||||
return errorResponse(400, "Invalid field name");
|
||||
}
|
||||
|
||||
await query(
|
||||
`UPDATE ocr_items
|
||||
SET ${colName} = $1
|
||||
WHERE document_id = $2 AND row_index = $3`,
|
||||
[value, docId, rowIndex]
|
||||
);
|
||||
|
||||
return NextResponse.json({ success: true });
|
||||
} else if (action === "flag") {
|
||||
const { isFlagged, remark } = body;
|
||||
if (isFlagged === undefined || remark === undefined) {
|
||||
return errorResponse(400, "Missing flag parameters");
|
||||
}
|
||||
|
||||
await query(
|
||||
`UPDATE ocr_items
|
||||
SET is_flagged = $1, remark = $2
|
||||
WHERE document_id = $3 AND row_index = $4`,
|
||||
[!!isFlagged, remark, docId, rowIndex]
|
||||
);
|
||||
|
||||
return NextResponse.json({ success: true });
|
||||
} else {
|
||||
return errorResponse(400, "Invalid action");
|
||||
}
|
||||
} catch (error: unknown) {
|
||||
console.error("Error in update-row API route:", error);
|
||||
const message = error instanceof Error ? error.message : "Internal server error";
|
||||
return errorResponse(500, message);
|
||||
}
|
||||
}
|
||||
import { NextRequest, NextResponse } from "next/server";
|
||||
import { query } from "../../../db";
|
||||
import path from "path";
|
||||
import { errorResponse } from "@/utils/api-error";
|
||||
|
||||
export async function POST(req: NextRequest) {
|
||||
try {
|
||||
const body = await req.json();
|
||||
const { page, rowIndex, action } = body;
|
||||
|
||||
if (!page || rowIndex === undefined || !action) {
|
||||
return errorResponse(400, "Missing required fields");
|
||||
}
|
||||
|
||||
const safeFile = path.basename(page);
|
||||
|
||||
// Get document ID
|
||||
const docRes = await query("SELECT id FROM documents WHERE filename = $1", [safeFile]);
|
||||
if (!docRes.rowCount || docRes.rowCount === 0) {
|
||||
return errorResponse(404, "Document not found in database");
|
||||
}
|
||||
const docId = docRes.rows[0].id;
|
||||
|
||||
if (action === "edit") {
|
||||
const { field, value } = body;
|
||||
if (!field || value === undefined) {
|
||||
return errorResponse(400, "Missing edit parameters");
|
||||
}
|
||||
|
||||
// Map UI field names to database columns
|
||||
let colName = "";
|
||||
if (field === "kodeBarang") {
|
||||
colName = "kode_barang";
|
||||
} else if (field === "banyak") {
|
||||
colName = "banyak";
|
||||
} else if (field === "jumlah") {
|
||||
colName = "jumlah";
|
||||
} else {
|
||||
return errorResponse(400, "Invalid field name");
|
||||
}
|
||||
|
||||
await query(
|
||||
`UPDATE ocr_items
|
||||
SET ${colName} = $1
|
||||
WHERE document_id = $2 AND row_index = $3`,
|
||||
[value, docId, rowIndex]
|
||||
);
|
||||
|
||||
return NextResponse.json({ success: true });
|
||||
} else if (action === "flag") {
|
||||
const { isFlagged, remark } = body;
|
||||
if (isFlagged === undefined || remark === undefined) {
|
||||
return errorResponse(400, "Missing flag parameters");
|
||||
}
|
||||
|
||||
await query(
|
||||
`UPDATE ocr_items
|
||||
SET is_flagged = $1, remark = $2
|
||||
WHERE document_id = $3 AND row_index = $4`,
|
||||
[!!isFlagged, remark, docId, rowIndex]
|
||||
);
|
||||
|
||||
return NextResponse.json({ success: true });
|
||||
} else {
|
||||
return errorResponse(400, "Invalid action");
|
||||
}
|
||||
} catch (error: unknown) {
|
||||
console.error("Error in update-row API route:", error);
|
||||
const message = error instanceof Error ? error.message : "Internal server error";
|
||||
return errorResponse(500, message);
|
||||
}
|
||||
}
|
||||
@@ -1,241 +1,241 @@
|
||||
import { NextRequest, NextResponse } from "next/server";
|
||||
import fs from "fs";
|
||||
import path from "path";
|
||||
import crypto from "crypto";
|
||||
import { query, resolveStoreFromText } from "../../../db";
|
||||
import { parseDOMetadata, sanitizeParsedMetadata } from "../../../utils/parser";
|
||||
import { startActiveLog, getActiveLog, clearActiveLog } from "../../../utils/active-log";
|
||||
import { errorResponse } from "@/utils/api-error";
|
||||
|
||||
const UPLOADS_DIR = "/uploads";
|
||||
|
||||
export async function POST(req: NextRequest) {
|
||||
try {
|
||||
// Ensure uploads directory exists
|
||||
if (!fs.existsSync(UPLOADS_DIR)) {
|
||||
fs.mkdirSync(UPLOADS_DIR, { recursive: true });
|
||||
}
|
||||
|
||||
const formData = await req.formData();
|
||||
const file = formData.get("file") as Blob | null;
|
||||
|
||||
if (!file) {
|
||||
return errorResponse(400, "No file uploaded");
|
||||
}
|
||||
|
||||
const originalName = file instanceof File ? file.name : "document.jpg";
|
||||
// Sanitize filename to avoid directory traversal
|
||||
const safeName = path.basename(originalName).replace(/\s+/g, "_");
|
||||
const filename = `${Date.now()}-${safeName}`;
|
||||
const filePath = path.join(UPLOADS_DIR, filename);
|
||||
|
||||
// Save file
|
||||
const arrayBuffer = await file.arrayBuffer();
|
||||
const buffer = Buffer.from(arrayBuffer);
|
||||
|
||||
// Compute hash to check for duplicate content
|
||||
const fileHash = crypto.createHash("sha256").update(buffer).digest("hex");
|
||||
|
||||
|
||||
|
||||
fs.writeFileSync(filePath, buffer);
|
||||
|
||||
// Convert to base64 for pipeline API
|
||||
const b64 = buffer.toString("base64");
|
||||
|
||||
// Form payload
|
||||
const payload = {
|
||||
file: b64,
|
||||
matchHistoryJob: false,
|
||||
useLayoutDetection: true,
|
||||
fileType: 1,
|
||||
useDocUnwarping: false,
|
||||
useDocOrientationClassify: true
|
||||
};
|
||||
|
||||
const pipelineUrl = process.env.PIPELINE_URL || "http://paddleocr-pipeline-api:8090/layout-parsing";
|
||||
console.log(`Forwarding uploaded file ${filename} to pipeline: ${pipelineUrl}`);
|
||||
startActiveLog(filename);
|
||||
|
||||
const response = await fetch(pipelineUrl, {
|
||||
method: "POST",
|
||||
headers: {
|
||||
"Content-Type": "application/json"
|
||||
},
|
||||
body: JSON.stringify(payload)
|
||||
});
|
||||
|
||||
if (!response.ok) {
|
||||
clearActiveLog(filename);
|
||||
const errText = await response.text();
|
||||
return errorResponse(response.status, `Pipeline API error: ${errText}`);
|
||||
}
|
||||
|
||||
let data = await response.json();
|
||||
|
||||
// Check if the image is not straight (tilt > 1.0 degree)
|
||||
const tilt = calculateAverageTilt(data);
|
||||
let unwarped = false;
|
||||
if (tilt > 1.0) {
|
||||
console.log(`Uploaded document ${filename} is not straight (average tilt: ${tilt.toFixed(2)} deg). Re-running with unwarping and orientation classification enabled...`);
|
||||
const unwarpPayload = {
|
||||
...payload,
|
||||
useDocUnwarping: true,
|
||||
useDocOrientationClassify: true
|
||||
};
|
||||
const unwarpResponse = await fetch(pipelineUrl, {
|
||||
method: "POST",
|
||||
headers: {
|
||||
"Content-Type": "application/json"
|
||||
},
|
||||
body: JSON.stringify(unwarpPayload)
|
||||
});
|
||||
if (unwarpResponse.ok) {
|
||||
data = await unwarpResponse.json();
|
||||
console.log(`Document unwarped successfully.`);
|
||||
unwarped = true;
|
||||
} else {
|
||||
console.error(`Unwarping failed with status ${unwarpResponse.status}`);
|
||||
}
|
||||
}
|
||||
|
||||
// Save JSON extraction result
|
||||
const jsonPath = `${filePath}.json`;
|
||||
fs.writeFileSync(jsonPath, JSON.stringify(data, null, 2));
|
||||
|
||||
// Save to PostgreSQL database
|
||||
try {
|
||||
const pipelineResult = data.result || data;
|
||||
pipelineResult.pipeline_info = {
|
||||
tilt,
|
||||
unwarped,
|
||||
original_tilt: tilt
|
||||
};
|
||||
|
||||
const page0 = pipelineResult?.layoutParsingResults?.[0] || {};
|
||||
const markdownText = page0?.markdown?.text || "";
|
||||
const docMetadata = parseDOMetadata(markdownText);
|
||||
|
||||
// Resolve store information using master database
|
||||
const resolvedStore = await resolveStoreFromText(markdownText);
|
||||
(docMetadata as any).orderUntuk = resolvedStore.orderUntuk;
|
||||
(docMetadata as any).alamat = resolvedStore.alamat;
|
||||
|
||||
// Stage 2 Filtering: Sanitize parsed metadata
|
||||
const sanitizedMetadata = sanitizeParsedMetadata(docMetadata as any);
|
||||
|
||||
// Construct client response representation
|
||||
const wrappedResult = {
|
||||
errorCode: 0,
|
||||
errorMsg: "Success",
|
||||
result: pipelineResult
|
||||
};
|
||||
const clientResponse = {
|
||||
filename,
|
||||
result: wrappedResult
|
||||
};
|
||||
|
||||
// Retrieve and finalize active log data
|
||||
const activeLog = getActiveLog(filename);
|
||||
let logsPayload: any = null;
|
||||
if (activeLog && activeLog.filename === filename) {
|
||||
activeLog.ocr_raw = pipelineResult;
|
||||
activeLog.stage_1_output = docMetadata;
|
||||
activeLog.stage_2_output = sanitizedMetadata;
|
||||
activeLog.frontend_response = clientResponse;
|
||||
activeLog.pipeline_info = {
|
||||
tilt,
|
||||
unwarped,
|
||||
original_tilt: tilt
|
||||
};
|
||||
logsPayload = { ...activeLog };
|
||||
}
|
||||
clearActiveLog(filename);
|
||||
|
||||
const insertDocRes = await query(`
|
||||
INSERT INTO documents (filename, upload_time, size, parsed, metadata, layout_parsing_result, is_sample, file_hash, processing_logs)
|
||||
VALUES ($1, $2, $3, $4, $5, $6, $7, $8, $9)
|
||||
RETURNING id
|
||||
`, [
|
||||
filename,
|
||||
new Date(),
|
||||
buffer.length,
|
||||
true,
|
||||
JSON.stringify(sanitizedMetadata),
|
||||
JSON.stringify(pipelineResult),
|
||||
false,
|
||||
fileHash,
|
||||
logsPayload ? JSON.stringify(logsPayload) : null
|
||||
]);
|
||||
|
||||
const docId = insertDocRes.rows[0].id;
|
||||
|
||||
for (let i = 0; i < docMetadata.items.length; i++) {
|
||||
const item = docMetadata.items[i];
|
||||
await query(`
|
||||
INSERT INTO ocr_items (
|
||||
document_id, row_index,
|
||||
kode_barang_original, kode_barang,
|
||||
nama_barang,
|
||||
banyak_original, banyak,
|
||||
jumlah_original, jumlah,
|
||||
is_flagged, remark
|
||||
)
|
||||
VALUES ($1, $2, $3, $3, $4, $5, $5, $6, $6, false, '')
|
||||
ON CONFLICT DO NOTHING
|
||||
`, [
|
||||
docId,
|
||||
i,
|
||||
item.kodeBarang,
|
||||
item.namaBarang,
|
||||
item.banyak,
|
||||
item.jumlah
|
||||
]);
|
||||
}
|
||||
} catch (dbErr) {
|
||||
console.error("Database save failed during upload (falling back to file):", dbErr);
|
||||
}
|
||||
|
||||
const wrappedResult = {
|
||||
errorCode: 0,
|
||||
errorMsg: "Success",
|
||||
result: data.result || data
|
||||
};
|
||||
|
||||
return NextResponse.json({
|
||||
filename,
|
||||
result: wrappedResult
|
||||
});
|
||||
} catch (error: unknown) {
|
||||
console.error("Error in upload API route:", error);
|
||||
const message = error instanceof Error ? error.message : "Internal server error";
|
||||
return errorResponse(500, message);
|
||||
}
|
||||
}
|
||||
|
||||
function getBlockAngle(points: number[][]) {
|
||||
if (!points || points.length < 2) return 0;
|
||||
const p0 = points[0];
|
||||
const p1 = points[1];
|
||||
const dx = p1[0] - p0[0];
|
||||
const dy = p1[1] - p0[1];
|
||||
let angle = Math.atan2(dy, dx) * 180 / Math.PI;
|
||||
if (angle < -45) angle = 90 + angle;
|
||||
if (angle > 45) angle = angle - 90;
|
||||
return Math.abs(angle);
|
||||
}
|
||||
|
||||
function calculateAverageTilt(data: any): number {
|
||||
const results = data?.result?.layoutParsingResults || data?.layoutParsingResults || [];
|
||||
if (results.length === 0) return 0;
|
||||
const list = results[0]?.prunedResult?.parsing_res_list || [];
|
||||
if (list.length === 0) return 0;
|
||||
const angles: number[] = [];
|
||||
for (const block of list) {
|
||||
if (block.block_polygon_points) {
|
||||
angles.push(getBlockAngle(block.block_polygon_points));
|
||||
}
|
||||
}
|
||||
if (angles.length === 0) return 0;
|
||||
return angles.reduce((sum, a) => sum + a, 0) / angles.length;
|
||||
}
|
||||
import { NextRequest, NextResponse } from "next/server";
|
||||
import fs from "fs";
|
||||
import path from "path";
|
||||
import crypto from "crypto";
|
||||
import { query, resolveStoreFromText } from "../../../db";
|
||||
import { parseDOMetadata, sanitizeParsedMetadata } from "../../../utils/parser";
|
||||
import { startActiveLog, getActiveLog, clearActiveLog } from "../../../utils/active-log";
|
||||
import { errorResponse } from "@/utils/api-error";
|
||||
|
||||
const UPLOADS_DIR = "/uploads";
|
||||
|
||||
export async function POST(req: NextRequest) {
|
||||
try {
|
||||
// Ensure uploads directory exists
|
||||
if (!fs.existsSync(UPLOADS_DIR)) {
|
||||
fs.mkdirSync(UPLOADS_DIR, { recursive: true });
|
||||
}
|
||||
|
||||
const formData = await req.formData();
|
||||
const file = formData.get("file") as Blob | null;
|
||||
|
||||
if (!file) {
|
||||
return errorResponse(400, "No file uploaded");
|
||||
}
|
||||
|
||||
const originalName = file instanceof File ? file.name : "document.jpg";
|
||||
// Sanitize filename to avoid directory traversal
|
||||
const safeName = path.basename(originalName).replace(/\s+/g, "_");
|
||||
const filename = `${Date.now()}-${safeName}`;
|
||||
const filePath = path.join(UPLOADS_DIR, filename);
|
||||
|
||||
// Save file
|
||||
const arrayBuffer = await file.arrayBuffer();
|
||||
const buffer = Buffer.from(arrayBuffer);
|
||||
|
||||
// Compute hash to check for duplicate content
|
||||
const fileHash = crypto.createHash("sha256").update(buffer).digest("hex");
|
||||
|
||||
|
||||
|
||||
fs.writeFileSync(filePath, buffer);
|
||||
|
||||
// Convert to base64 for pipeline API
|
||||
const b64 = buffer.toString("base64");
|
||||
|
||||
// Form payload
|
||||
const payload = {
|
||||
file: b64,
|
||||
matchHistoryJob: false,
|
||||
useLayoutDetection: true,
|
||||
fileType: 1,
|
||||
useDocUnwarping: false,
|
||||
useDocOrientationClassify: true
|
||||
};
|
||||
|
||||
const pipelineUrl = process.env.PIPELINE_URL || "http://paddleocr-pipeline-api:8090/layout-parsing";
|
||||
console.log(`Forwarding uploaded file ${filename} to pipeline: ${pipelineUrl}`);
|
||||
startActiveLog(filename);
|
||||
|
||||
const response = await fetch(pipelineUrl, {
|
||||
method: "POST",
|
||||
headers: {
|
||||
"Content-Type": "application/json"
|
||||
},
|
||||
body: JSON.stringify(payload)
|
||||
});
|
||||
|
||||
if (!response.ok) {
|
||||
clearActiveLog(filename);
|
||||
const errText = await response.text();
|
||||
return errorResponse(response.status, `Pipeline API error: ${errText}`);
|
||||
}
|
||||
|
||||
let data = await response.json();
|
||||
|
||||
// Check if the image is not straight (tilt > 1.0 degree)
|
||||
const tilt = calculateAverageTilt(data);
|
||||
let unwarped = false;
|
||||
if (tilt > 1.0) {
|
||||
console.log(`Uploaded document ${filename} is not straight (average tilt: ${tilt.toFixed(2)} deg). Re-running with unwarping and orientation classification enabled...`);
|
||||
const unwarpPayload = {
|
||||
...payload,
|
||||
useDocUnwarping: true,
|
||||
useDocOrientationClassify: true
|
||||
};
|
||||
const unwarpResponse = await fetch(pipelineUrl, {
|
||||
method: "POST",
|
||||
headers: {
|
||||
"Content-Type": "application/json"
|
||||
},
|
||||
body: JSON.stringify(unwarpPayload)
|
||||
});
|
||||
if (unwarpResponse.ok) {
|
||||
data = await unwarpResponse.json();
|
||||
console.log(`Document unwarped successfully.`);
|
||||
unwarped = true;
|
||||
} else {
|
||||
console.error(`Unwarping failed with status ${unwarpResponse.status}`);
|
||||
}
|
||||
}
|
||||
|
||||
// Save JSON extraction result
|
||||
const jsonPath = `${filePath}.json`;
|
||||
fs.writeFileSync(jsonPath, JSON.stringify(data, null, 2));
|
||||
|
||||
// Save to PostgreSQL database
|
||||
try {
|
||||
const pipelineResult = data.result || data;
|
||||
pipelineResult.pipeline_info = {
|
||||
tilt,
|
||||
unwarped,
|
||||
original_tilt: tilt
|
||||
};
|
||||
|
||||
const page0 = pipelineResult?.layoutParsingResults?.[0] || {};
|
||||
const markdownText = page0?.markdown?.text || "";
|
||||
const docMetadata = parseDOMetadata(markdownText);
|
||||
|
||||
// Resolve store information using master database
|
||||
const resolvedStore = await resolveStoreFromText(markdownText);
|
||||
(docMetadata as any).orderUntuk = resolvedStore.orderUntuk;
|
||||
(docMetadata as any).alamat = resolvedStore.alamat;
|
||||
|
||||
// Stage 2 Filtering: Sanitize parsed metadata
|
||||
const sanitizedMetadata = sanitizeParsedMetadata(docMetadata as any);
|
||||
|
||||
// Construct client response representation
|
||||
const wrappedResult = {
|
||||
errorCode: 0,
|
||||
errorMsg: "Success",
|
||||
result: pipelineResult
|
||||
};
|
||||
const clientResponse = {
|
||||
filename,
|
||||
result: wrappedResult
|
||||
};
|
||||
|
||||
// Retrieve and finalize active log data
|
||||
const activeLog = getActiveLog(filename);
|
||||
let logsPayload: any = null;
|
||||
if (activeLog && activeLog.filename === filename) {
|
||||
activeLog.ocr_raw = pipelineResult;
|
||||
activeLog.stage_1_output = docMetadata;
|
||||
activeLog.stage_2_output = sanitizedMetadata;
|
||||
activeLog.frontend_response = clientResponse;
|
||||
activeLog.pipeline_info = {
|
||||
tilt,
|
||||
unwarped,
|
||||
original_tilt: tilt
|
||||
};
|
||||
logsPayload = { ...activeLog };
|
||||
}
|
||||
clearActiveLog(filename);
|
||||
|
||||
const insertDocRes = await query(`
|
||||
INSERT INTO documents (filename, upload_time, size, parsed, metadata, layout_parsing_result, is_sample, file_hash, processing_logs)
|
||||
VALUES ($1, $2, $3, $4, $5, $6, $7, $8, $9)
|
||||
RETURNING id
|
||||
`, [
|
||||
filename,
|
||||
new Date(),
|
||||
buffer.length,
|
||||
true,
|
||||
JSON.stringify(sanitizedMetadata),
|
||||
JSON.stringify(pipelineResult),
|
||||
false,
|
||||
fileHash,
|
||||
logsPayload ? JSON.stringify(logsPayload) : null
|
||||
]);
|
||||
|
||||
const docId = insertDocRes.rows[0].id;
|
||||
|
||||
for (let i = 0; i < docMetadata.items.length; i++) {
|
||||
const item = docMetadata.items[i];
|
||||
await query(`
|
||||
INSERT INTO ocr_items (
|
||||
document_id, row_index,
|
||||
kode_barang_original, kode_barang,
|
||||
nama_barang,
|
||||
banyak_original, banyak,
|
||||
jumlah_original, jumlah,
|
||||
is_flagged, remark
|
||||
)
|
||||
VALUES ($1, $2, $3, $3, $4, $5, $5, $6, $6, false, '')
|
||||
ON CONFLICT DO NOTHING
|
||||
`, [
|
||||
docId,
|
||||
i,
|
||||
item.kodeBarang,
|
||||
item.namaBarang,
|
||||
item.banyak,
|
||||
item.jumlah
|
||||
]);
|
||||
}
|
||||
} catch (dbErr) {
|
||||
console.error("Database save failed during upload (falling back to file):", dbErr);
|
||||
}
|
||||
|
||||
const wrappedResult = {
|
||||
errorCode: 0,
|
||||
errorMsg: "Success",
|
||||
result: data.result || data
|
||||
};
|
||||
|
||||
return NextResponse.json({
|
||||
filename,
|
||||
result: wrappedResult
|
||||
});
|
||||
} catch (error: unknown) {
|
||||
console.error("Error in upload API route:", error);
|
||||
const message = error instanceof Error ? error.message : "Internal server error";
|
||||
return errorResponse(500, message);
|
||||
}
|
||||
}
|
||||
|
||||
function getBlockAngle(points: number[][]) {
|
||||
if (!points || points.length < 2) return 0;
|
||||
const p0 = points[0];
|
||||
const p1 = points[1];
|
||||
const dx = p1[0] - p0[0];
|
||||
const dy = p1[1] - p0[1];
|
||||
let angle = Math.atan2(dy, dx) * 180 / Math.PI;
|
||||
if (angle < -45) angle = 90 + angle;
|
||||
if (angle > 45) angle = angle - 90;
|
||||
return Math.abs(angle);
|
||||
}
|
||||
|
||||
function calculateAverageTilt(data: any): number {
|
||||
const results = data?.result?.layoutParsingResults || data?.layoutParsingResults || [];
|
||||
if (results.length === 0) return 0;
|
||||
const list = results[0]?.prunedResult?.parsing_res_list || [];
|
||||
if (list.length === 0) return 0;
|
||||
const angles: number[] = [];
|
||||
for (const block of list) {
|
||||
if (block.block_polygon_points) {
|
||||
angles.push(getBlockAngle(block.block_polygon_points));
|
||||
}
|
||||
}
|
||||
if (angles.length === 0) return 0;
|
||||
return angles.reduce((sum, a) => sum + a, 0) / angles.length;
|
||||
}
|
||||
@@ -1,75 +1,75 @@
|
||||
import { NextRequest, NextResponse } from "next/server";
|
||||
import bcrypt from "bcryptjs";
|
||||
import { errorResponse } from "@/utils/api-error";
|
||||
import { signAccountToken } from "@/utils/auth";
|
||||
import { query } from "../../../../../db";
|
||||
|
||||
const corsHeaders = {
|
||||
"Access-Control-Allow-Origin": "*",
|
||||
"Access-Control-Allow-Methods": "GET, POST, PUT, DELETE, OPTIONS",
|
||||
"Access-Control-Allow-Headers": "Content-Type, Authorization"
|
||||
};
|
||||
|
||||
export async function OPTIONS() {
|
||||
return new NextResponse(null, { status: 204, headers: corsHeaders });
|
||||
}
|
||||
|
||||
export async function POST(req: NextRequest) {
|
||||
try {
|
||||
const body = await req.json();
|
||||
const { username, password } = body;
|
||||
|
||||
if (!username || !password) {
|
||||
return errorResponse(401, "Invalid username or password", { headers: corsHeaders });
|
||||
}
|
||||
|
||||
// Each account is assigned exactly one store (kode_toko) - the token
|
||||
// carries that assignment so store name/address never need OCR
|
||||
// detection later; whichever account uploads, its own store is used.
|
||||
const accountRes = await query(
|
||||
`SELECT a.id, a.username, a.password, a.role, a.is_active,
|
||||
s.kode_toko, s.nama_toko, s.alamat
|
||||
FROM accounts a
|
||||
LEFT JOIN store_master s ON a.kode_toko = s.kode_toko
|
||||
WHERE a.username = $1`,
|
||||
[username]
|
||||
);
|
||||
|
||||
if (accountRes.rowCount && accountRes.rowCount > 0 && bcrypt.compareSync(password, accountRes.rows[0].password)) {
|
||||
const account = accountRes.rows[0];
|
||||
|
||||
if (!account.is_active) {
|
||||
return errorResponse(401, "Account is disabled", { headers: corsHeaders });
|
||||
}
|
||||
|
||||
const token = signAccountToken({
|
||||
accountId: account.id,
|
||||
username: account.username,
|
||||
kodeToko: account.kode_toko,
|
||||
role: account.role
|
||||
});
|
||||
|
||||
return NextResponse.json({
|
||||
status: "success",
|
||||
message: "Login successful",
|
||||
data: {
|
||||
token,
|
||||
profile: {
|
||||
username: account.username,
|
||||
role: account.role,
|
||||
is_active: account.is_active,
|
||||
kodeToko: account.kode_toko,
|
||||
namaToko: account.nama_toko,
|
||||
alamat: account.alamat
|
||||
}
|
||||
}
|
||||
}, { headers: corsHeaders });
|
||||
}
|
||||
|
||||
return errorResponse(401, "Invalid username or password", { headers: corsHeaders });
|
||||
} catch (error: unknown) {
|
||||
console.error("Error in login API route:", error);
|
||||
const message = error instanceof Error ? error.message : "Internal server error";
|
||||
return errorResponse(500, message, { headers: corsHeaders });
|
||||
}
|
||||
}
|
||||
import { NextRequest, NextResponse } from "next/server";
|
||||
import bcrypt from "bcryptjs";
|
||||
import { errorResponse } from "@/utils/api-error";
|
||||
import { signAccountToken } from "@/utils/auth";
|
||||
import { query } from "../../../../../db";
|
||||
|
||||
const corsHeaders = {
|
||||
"Access-Control-Allow-Origin": "*",
|
||||
"Access-Control-Allow-Methods": "GET, POST, PUT, DELETE, OPTIONS",
|
||||
"Access-Control-Allow-Headers": "Content-Type, Authorization"
|
||||
};
|
||||
|
||||
export async function OPTIONS() {
|
||||
return new NextResponse(null, { status: 204, headers: corsHeaders });
|
||||
}
|
||||
|
||||
export async function POST(req: NextRequest) {
|
||||
try {
|
||||
const body = await req.json();
|
||||
const { username, password } = body;
|
||||
|
||||
if (!username || !password) {
|
||||
return errorResponse(401, "Invalid username or password", { headers: corsHeaders });
|
||||
}
|
||||
|
||||
// Each account is assigned exactly one store (kode_toko) - the token
|
||||
// carries that assignment so store name/address never need OCR
|
||||
// detection later; whichever account uploads, its own store is used.
|
||||
const accountRes = await query(
|
||||
`SELECT a.id, a.username, a.password, a.role, a.is_active,
|
||||
s.kode_toko, s.nama_toko, s.alamat
|
||||
FROM accounts a
|
||||
LEFT JOIN store_master s ON a.kode_toko = s.kode_toko
|
||||
WHERE a.username = $1`,
|
||||
[username]
|
||||
);
|
||||
|
||||
if (accountRes.rowCount && accountRes.rowCount > 0 && bcrypt.compareSync(password, accountRes.rows[0].password)) {
|
||||
const account = accountRes.rows[0];
|
||||
|
||||
if (!account.is_active) {
|
||||
return errorResponse(401, "Account is disabled", { headers: corsHeaders });
|
||||
}
|
||||
|
||||
const token = signAccountToken({
|
||||
accountId: account.id,
|
||||
username: account.username,
|
||||
kodeToko: account.kode_toko,
|
||||
role: account.role
|
||||
});
|
||||
|
||||
return NextResponse.json({
|
||||
status: "success",
|
||||
message: "Login successful",
|
||||
data: {
|
||||
token,
|
||||
profile: {
|
||||
username: account.username,
|
||||
role: account.role,
|
||||
is_active: account.is_active,
|
||||
kodeToko: account.kode_toko,
|
||||
namaToko: account.nama_toko,
|
||||
alamat: account.alamat
|
||||
}
|
||||
}
|
||||
}, { headers: corsHeaders });
|
||||
}
|
||||
|
||||
return errorResponse(401, "Invalid username or password", { headers: corsHeaders });
|
||||
} catch (error: unknown) {
|
||||
console.error("Error in login API route:", error);
|
||||
const message = error instanceof Error ? error.message : "Internal server error";
|
||||
return errorResponse(500, message, { headers: corsHeaders });
|
||||
}
|
||||
}
|
||||
@@ -1,67 +1,67 @@
|
||||
import { NextRequest, NextResponse } from "next/server";
|
||||
import { errorResponse } from "@/utils/api-error";
|
||||
import { getAccountFromAuthHeader } from "@/utils/auth";
|
||||
import { query } from "../../../../../db";
|
||||
|
||||
const corsHeaders = {
|
||||
"Access-Control-Allow-Origin": "*",
|
||||
"Access-Control-Allow-Methods": "GET, POST, PUT, DELETE, OPTIONS",
|
||||
"Access-Control-Allow-Headers": "Content-Type, Authorization"
|
||||
};
|
||||
|
||||
export async function OPTIONS() {
|
||||
return new NextResponse(null, { status: 204, headers: corsHeaders });
|
||||
}
|
||||
|
||||
export async function GET(req: NextRequest) {
|
||||
try {
|
||||
const authHeader = req.headers.get("authorization");
|
||||
const tokenPayload = getAccountFromAuthHeader(authHeader);
|
||||
|
||||
if (!tokenPayload) {
|
||||
return errorResponse(401, "Unauthorized", { headers: corsHeaders });
|
||||
}
|
||||
|
||||
const accountRes = await query(
|
||||
`SELECT a.id, a.username, a.role, a.is_active,
|
||||
s.kode_toko, s.nama_toko, s.alamat
|
||||
FROM accounts a
|
||||
LEFT JOIN store_master s ON a.kode_toko = s.kode_toko
|
||||
WHERE a.id = $1`,
|
||||
[tokenPayload.accountId]
|
||||
);
|
||||
|
||||
if (accountRes.rowCount && accountRes.rowCount > 0) {
|
||||
const account = accountRes.rows[0];
|
||||
|
||||
if (!account.is_active) {
|
||||
return errorResponse(401, "Account is disabled", { headers: corsHeaders });
|
||||
}
|
||||
|
||||
// We extract the token exactly as passed in to echo it back in the same shape as login
|
||||
const token = authHeader?.slice("Bearer ".length).trim();
|
||||
|
||||
return NextResponse.json({
|
||||
status: "success",
|
||||
message: "Profile retrieved successfully",
|
||||
data: {
|
||||
token,
|
||||
profile: {
|
||||
username: account.username,
|
||||
role: account.role,
|
||||
is_active: account.is_active,
|
||||
kodeToko: account.kode_toko,
|
||||
namaToko: account.nama_toko,
|
||||
alamat: account.alamat
|
||||
}
|
||||
}
|
||||
}, { headers: corsHeaders });
|
||||
}
|
||||
|
||||
return errorResponse(401, "Account not found", { headers: corsHeaders });
|
||||
} catch (error: unknown) {
|
||||
console.error("Error in auth/me API route:", error);
|
||||
const message = error instanceof Error ? error.message : "Internal server error";
|
||||
return errorResponse(500, message, { headers: corsHeaders });
|
||||
}
|
||||
}
|
||||
import { NextRequest, NextResponse } from "next/server";
|
||||
import { errorResponse } from "@/utils/api-error";
|
||||
import { getAccountFromAuthHeader } from "@/utils/auth";
|
||||
import { query } from "../../../../../db";
|
||||
|
||||
const corsHeaders = {
|
||||
"Access-Control-Allow-Origin": "*",
|
||||
"Access-Control-Allow-Methods": "GET, POST, PUT, DELETE, OPTIONS",
|
||||
"Access-Control-Allow-Headers": "Content-Type, Authorization"
|
||||
};
|
||||
|
||||
export async function OPTIONS() {
|
||||
return new NextResponse(null, { status: 204, headers: corsHeaders });
|
||||
}
|
||||
|
||||
export async function GET(req: NextRequest) {
|
||||
try {
|
||||
const authHeader = req.headers.get("authorization");
|
||||
const tokenPayload = getAccountFromAuthHeader(authHeader);
|
||||
|
||||
if (!tokenPayload) {
|
||||
return errorResponse(401, "Unauthorized", { headers: corsHeaders });
|
||||
}
|
||||
|
||||
const accountRes = await query(
|
||||
`SELECT a.id, a.username, a.role, a.is_active,
|
||||
s.kode_toko, s.nama_toko, s.alamat
|
||||
FROM accounts a
|
||||
LEFT JOIN store_master s ON a.kode_toko = s.kode_toko
|
||||
WHERE a.id = $1`,
|
||||
[tokenPayload.accountId]
|
||||
);
|
||||
|
||||
if (accountRes.rowCount && accountRes.rowCount > 0) {
|
||||
const account = accountRes.rows[0];
|
||||
|
||||
if (!account.is_active) {
|
||||
return errorResponse(401, "Account is disabled", { headers: corsHeaders });
|
||||
}
|
||||
|
||||
// We extract the token exactly as passed in to echo it back in the same shape as login
|
||||
const token = authHeader?.slice("Bearer ".length).trim();
|
||||
|
||||
return NextResponse.json({
|
||||
status: "success",
|
||||
message: "Profile retrieved successfully",
|
||||
data: {
|
||||
token,
|
||||
profile: {
|
||||
username: account.username,
|
||||
role: account.role,
|
||||
is_active: account.is_active,
|
||||
kodeToko: account.kode_toko,
|
||||
namaToko: account.nama_toko,
|
||||
alamat: account.alamat
|
||||
}
|
||||
}
|
||||
}, { headers: corsHeaders });
|
||||
}
|
||||
|
||||
return errorResponse(401, "Account not found", { headers: corsHeaders });
|
||||
} catch (error: unknown) {
|
||||
console.error("Error in auth/me API route:", error);
|
||||
const message = error instanceof Error ? error.message : "Internal server error";
|
||||
return errorResponse(500, message, { headers: corsHeaders });
|
||||
}
|
||||
}
|
||||
@@ -1,228 +1,228 @@
|
||||
import { NextRequest, NextResponse } from "next/server";
|
||||
import { query, withTransaction } from "../../../../../db";
|
||||
import { errorResponse } from "@/utils/api-error";
|
||||
import { getAccountFromAuthHeader } from "@/utils/auth";
|
||||
import { mapDocumentRow } from "@/utils/document-mapper";
|
||||
|
||||
const corsHeaders = {
|
||||
"Access-Control-Allow-Origin": "*",
|
||||
"Access-Control-Allow-Methods": "GET, POST, PUT, DELETE, OPTIONS",
|
||||
"Access-Control-Allow-Headers": "Content-Type, Authorization"
|
||||
};
|
||||
|
||||
export async function OPTIONS() {
|
||||
return new NextResponse(null, { status: 204, headers: corsHeaders });
|
||||
}
|
||||
|
||||
export async function GET(
|
||||
req: NextRequest,
|
||||
context: { params: Promise<{ id: string }> }
|
||||
) {
|
||||
try {
|
||||
const account = getAccountFromAuthHeader(req.headers.get("authorization"));
|
||||
if (!account) {
|
||||
return errorResponse(401, "Unauthorized", { headers: corsHeaders });
|
||||
}
|
||||
|
||||
const params = await context.params;
|
||||
const docId = parseInt(params.id);
|
||||
if (isNaN(docId)) {
|
||||
return errorResponse(400, "Invalid document ID", { headers: corsHeaders });
|
||||
}
|
||||
|
||||
// Deliberately not filtering on `parsed = true` here (unlike the list route) -
|
||||
// the whole point of this endpoint is to let the poller see pending/failed
|
||||
// documents, not just done ones.
|
||||
const docRes = await query(`
|
||||
SELECT id, filename, upload_time, parsed, is_sample, metadata, latitude, longitude, kode_toko, scan_mode, parse_error, confirmed
|
||||
FROM documents
|
||||
WHERE id = $1
|
||||
`, [docId]);
|
||||
|
||||
if (!docRes.rowCount || docRes.rowCount === 0) {
|
||||
return errorResponse(404, "Document not found", { headers: corsHeaders });
|
||||
}
|
||||
|
||||
const doc = docRes.rows[0];
|
||||
|
||||
if (account.role !== 'admin' && doc.kode_toko !== account.kodeToko) {
|
||||
return errorResponse(403, "Forbidden: You do not have permission to view this document", { headers: corsHeaders });
|
||||
}
|
||||
|
||||
const itemsRes = await query(`
|
||||
SELECT row_index, kode_barang, nama_barang, banyak, jumlah
|
||||
FROM ocr_items
|
||||
WHERE document_id = $1
|
||||
ORDER BY row_index
|
||||
`, [docId]);
|
||||
|
||||
return NextResponse.json({
|
||||
status: "success",
|
||||
data: mapDocumentRow(doc, itemsRes.rows)
|
||||
}, { headers: corsHeaders });
|
||||
|
||||
} catch (error: unknown) {
|
||||
console.error("Error in get document API v1 route:", error);
|
||||
const message = error instanceof Error ? error.message : "Internal server error";
|
||||
return errorResponse(500, message, { headers: corsHeaders });
|
||||
}
|
||||
}
|
||||
|
||||
export async function PUT(
|
||||
req: NextRequest,
|
||||
context: { params: Promise<{ id: string }> }
|
||||
) {
|
||||
try {
|
||||
const account = getAccountFromAuthHeader(req.headers.get("authorization"));
|
||||
if (!account) {
|
||||
return errorResponse(401, "Unauthorized", { headers: corsHeaders });
|
||||
}
|
||||
|
||||
const params = await context.params;
|
||||
const { id } = params;
|
||||
const docId = parseInt(id);
|
||||
|
||||
if (isNaN(docId)) {
|
||||
return errorResponse(400, "Invalid document ID", { headers: corsHeaders });
|
||||
}
|
||||
|
||||
// Check if document exists
|
||||
const checkRes = await query("SELECT id, filename, upload_time, kode_toko FROM documents WHERE id = $1", [docId]);
|
||||
if (!checkRes.rowCount || checkRes.rowCount === 0) {
|
||||
return errorResponse(404, "Document not found", { headers: corsHeaders });
|
||||
}
|
||||
|
||||
const doc = checkRes.rows[0];
|
||||
|
||||
if (account.role !== 'admin' && doc.kode_toko !== account.kodeToko) {
|
||||
return errorResponse(403, "Forbidden: You do not have permission to modify this document", { headers: corsHeaders });
|
||||
}
|
||||
|
||||
const body = await req.json();
|
||||
const {
|
||||
tanggal,
|
||||
noPo,
|
||||
noSo,
|
||||
noDo,
|
||||
kepadaYth,
|
||||
orderUntuk,
|
||||
alamat,
|
||||
platTruk,
|
||||
namaDriver,
|
||||
namaPenerima,
|
||||
latitude,
|
||||
longitude,
|
||||
items = []
|
||||
} = body;
|
||||
|
||||
// Structuring metadata JSONB to store both formats for full compatibility
|
||||
const metadata = {
|
||||
// Legacy Next.js web parser format
|
||||
tanggal: tanggal || "",
|
||||
noPO: noPo || "",
|
||||
noSO: noSo || "",
|
||||
noDO: noDo || doc.filename || "",
|
||||
customerInfo: kepadaYth || "",
|
||||
headerRemark: namaPenerima || "",
|
||||
|
||||
// Mobile native app format
|
||||
header: {
|
||||
tanggal: tanggal || "",
|
||||
no_po: noPo || "",
|
||||
no_so: noSo || "",
|
||||
no_do: noDo || ""
|
||||
},
|
||||
shipment: {
|
||||
kepada_yth: kepadaYth || "",
|
||||
order_untuk: orderUntuk || "",
|
||||
alamat: alamat || "",
|
||||
plat_truk: platTruk || "",
|
||||
nama_driver: namaDriver || "",
|
||||
nama_penerima: namaPenerima || ""
|
||||
}
|
||||
};
|
||||
|
||||
const latFloat = latitude ? parseFloat(latitude.toString()) : null;
|
||||
const lngFloat = longitude ? parseFloat(longitude.toString()) : null;
|
||||
|
||||
// Update document record. `confirmed = true` is the one and only place
|
||||
// this flips - this PUT is literally "the user tapped Simpan & Konfirmasi"
|
||||
// (see docs/api-contract-map.md G11).
|
||||
await query(`
|
||||
UPDATE documents
|
||||
SET parsed = true,
|
||||
confirmed = true,
|
||||
latitude = $2,
|
||||
longitude = $3,
|
||||
metadata = $4
|
||||
WHERE id = $1
|
||||
`, [docId, latFloat, lngFloat, JSON.stringify(metadata)]);
|
||||
|
||||
// Delete-then-reinsert must be atomic: without a transaction, a failure partway
|
||||
// through the insert loop leaves the document with its header already updated
|
||||
// above but only some (or none) of its items, since the delete has already
|
||||
// committed independently.
|
||||
await withTransaction(async (client) => {
|
||||
await client.query("DELETE FROM ocr_items WHERE document_id = $1", [docId]);
|
||||
|
||||
for (let i = 0; i < items.length; i++) {
|
||||
const item = items[i];
|
||||
const nomorSku = item.nomor_sku || item.nomorSku || "";
|
||||
const namaBarang = item.nama_barang || item.namaBarang || "";
|
||||
const banyak = item.banyak || "";
|
||||
const jumlah = item.jumlah || "";
|
||||
|
||||
await client.query(`
|
||||
INSERT INTO ocr_items (
|
||||
document_id, row_index,
|
||||
kode_barang_original, kode_barang,
|
||||
nama_barang,
|
||||
banyak_original, banyak,
|
||||
jumlah_original, jumlah,
|
||||
is_flagged, remark
|
||||
) VALUES ($1, $2, $3, $3, $4, $5, $5, $6, $6, false, '')
|
||||
`, [docId, i, nomorSku, namaBarang, banyak, jumlah]);
|
||||
}
|
||||
});
|
||||
|
||||
// Return the updated document mapping
|
||||
const mappedData = {
|
||||
id: docId.toString(),
|
||||
filePath: doc.filename,
|
||||
createdAt: doc.upload_time.toISOString(),
|
||||
header: {
|
||||
tanggal: tanggal || "",
|
||||
no_po: noPo || "",
|
||||
no_so: noSo || "",
|
||||
no_do: noDo || ""
|
||||
},
|
||||
shipment: {
|
||||
kepada_yth: kepadaYth || "",
|
||||
order_untuk: orderUntuk || "",
|
||||
alamat: alamat || "",
|
||||
plat_truk: platTruk || "",
|
||||
nama_driver: namaDriver || "",
|
||||
nama_penerima: namaPenerima || ""
|
||||
},
|
||||
items: items.map((item: any) => ({
|
||||
nomor_sku: item.nomor_sku || item.nomorSku || "",
|
||||
nama_barang: item.nama_barang || item.namaBarang || "",
|
||||
banyak: item.banyak || "",
|
||||
jumlah: item.jumlah || ""
|
||||
})),
|
||||
latitude: latFloat,
|
||||
longitude: lngFloat
|
||||
};
|
||||
|
||||
return NextResponse.json({
|
||||
status: "success",
|
||||
message: "Document updated successfully",
|
||||
data: mappedData
|
||||
}, { headers: corsHeaders });
|
||||
|
||||
} catch (error: unknown) {
|
||||
console.error("Error in update document API v1 route:", error);
|
||||
const message = error instanceof Error ? error.message : "Internal server error";
|
||||
return errorResponse(500, message, { headers: corsHeaders });
|
||||
}
|
||||
}
|
||||
import { NextRequest, NextResponse } from "next/server";
|
||||
import { query, withTransaction } from "../../../../../db";
|
||||
import { errorResponse } from "@/utils/api-error";
|
||||
import { getAccountFromAuthHeader } from "@/utils/auth";
|
||||
import { mapDocumentRow } from "@/utils/document-mapper";
|
||||
|
||||
const corsHeaders = {
|
||||
"Access-Control-Allow-Origin": "*",
|
||||
"Access-Control-Allow-Methods": "GET, POST, PUT, DELETE, OPTIONS",
|
||||
"Access-Control-Allow-Headers": "Content-Type, Authorization"
|
||||
};
|
||||
|
||||
export async function OPTIONS() {
|
||||
return new NextResponse(null, { status: 204, headers: corsHeaders });
|
||||
}
|
||||
|
||||
export async function GET(
|
||||
req: NextRequest,
|
||||
context: { params: Promise<{ id: string }> }
|
||||
) {
|
||||
try {
|
||||
const account = getAccountFromAuthHeader(req.headers.get("authorization"));
|
||||
if (!account) {
|
||||
return errorResponse(401, "Unauthorized", { headers: corsHeaders });
|
||||
}
|
||||
|
||||
const params = await context.params;
|
||||
const docId = parseInt(params.id);
|
||||
if (isNaN(docId)) {
|
||||
return errorResponse(400, "Invalid document ID", { headers: corsHeaders });
|
||||
}
|
||||
|
||||
// Deliberately not filtering on `parsed = true` here (unlike the list route) -
|
||||
// the whole point of this endpoint is to let the poller see pending/failed
|
||||
// documents, not just done ones.
|
||||
const docRes = await query(`
|
||||
SELECT id, filename, upload_time, parsed, is_sample, metadata, latitude, longitude, kode_toko, scan_mode, parse_error, confirmed
|
||||
FROM documents
|
||||
WHERE id = $1
|
||||
`, [docId]);
|
||||
|
||||
if (!docRes.rowCount || docRes.rowCount === 0) {
|
||||
return errorResponse(404, "Document not found", { headers: corsHeaders });
|
||||
}
|
||||
|
||||
const doc = docRes.rows[0];
|
||||
|
||||
if (account.role !== 'admin' && doc.kode_toko !== account.kodeToko) {
|
||||
return errorResponse(403, "Forbidden: You do not have permission to view this document", { headers: corsHeaders });
|
||||
}
|
||||
|
||||
const itemsRes = await query(`
|
||||
SELECT row_index, kode_barang, nama_barang, banyak, jumlah
|
||||
FROM ocr_items
|
||||
WHERE document_id = $1
|
||||
ORDER BY row_index
|
||||
`, [docId]);
|
||||
|
||||
return NextResponse.json({
|
||||
status: "success",
|
||||
data: mapDocumentRow(doc, itemsRes.rows)
|
||||
}, { headers: corsHeaders });
|
||||
|
||||
} catch (error: unknown) {
|
||||
console.error("Error in get document API v1 route:", error);
|
||||
const message = error instanceof Error ? error.message : "Internal server error";
|
||||
return errorResponse(500, message, { headers: corsHeaders });
|
||||
}
|
||||
}
|
||||
|
||||
export async function PUT(
|
||||
req: NextRequest,
|
||||
context: { params: Promise<{ id: string }> }
|
||||
) {
|
||||
try {
|
||||
const account = getAccountFromAuthHeader(req.headers.get("authorization"));
|
||||
if (!account) {
|
||||
return errorResponse(401, "Unauthorized", { headers: corsHeaders });
|
||||
}
|
||||
|
||||
const params = await context.params;
|
||||
const { id } = params;
|
||||
const docId = parseInt(id);
|
||||
|
||||
if (isNaN(docId)) {
|
||||
return errorResponse(400, "Invalid document ID", { headers: corsHeaders });
|
||||
}
|
||||
|
||||
// Check if document exists
|
||||
const checkRes = await query("SELECT id, filename, upload_time, kode_toko FROM documents WHERE id = $1", [docId]);
|
||||
if (!checkRes.rowCount || checkRes.rowCount === 0) {
|
||||
return errorResponse(404, "Document not found", { headers: corsHeaders });
|
||||
}
|
||||
|
||||
const doc = checkRes.rows[0];
|
||||
|
||||
if (account.role !== 'admin' && doc.kode_toko !== account.kodeToko) {
|
||||
return errorResponse(403, "Forbidden: You do not have permission to modify this document", { headers: corsHeaders });
|
||||
}
|
||||
|
||||
const body = await req.json();
|
||||
const {
|
||||
tanggal,
|
||||
noPo,
|
||||
noSo,
|
||||
noDo,
|
||||
kepadaYth,
|
||||
orderUntuk,
|
||||
alamat,
|
||||
platTruk,
|
||||
namaDriver,
|
||||
namaPenerima,
|
||||
latitude,
|
||||
longitude,
|
||||
items = []
|
||||
} = body;
|
||||
|
||||
// Structuring metadata JSONB to store both formats for full compatibility
|
||||
const metadata = {
|
||||
// Legacy Next.js web parser format
|
||||
tanggal: tanggal || "",
|
||||
noPO: noPo || "",
|
||||
noSO: noSo || "",
|
||||
noDO: noDo || doc.filename || "",
|
||||
customerInfo: kepadaYth || "",
|
||||
headerRemark: namaPenerima || "",
|
||||
|
||||
// Mobile native app format
|
||||
header: {
|
||||
tanggal: tanggal || "",
|
||||
no_po: noPo || "",
|
||||
no_so: noSo || "",
|
||||
no_do: noDo || ""
|
||||
},
|
||||
shipment: {
|
||||
kepada_yth: kepadaYth || "",
|
||||
order_untuk: orderUntuk || "",
|
||||
alamat: alamat || "",
|
||||
plat_truk: platTruk || "",
|
||||
nama_driver: namaDriver || "",
|
||||
nama_penerima: namaPenerima || ""
|
||||
}
|
||||
};
|
||||
|
||||
const latFloat = latitude ? parseFloat(latitude.toString()) : null;
|
||||
const lngFloat = longitude ? parseFloat(longitude.toString()) : null;
|
||||
|
||||
// Update document record. `confirmed = true` is the one and only place
|
||||
// this flips - this PUT is literally "the user tapped Simpan & Konfirmasi"
|
||||
// (see docs/api-contract-map.md G11).
|
||||
await query(`
|
||||
UPDATE documents
|
||||
SET parsed = true,
|
||||
confirmed = true,
|
||||
latitude = $2,
|
||||
longitude = $3,
|
||||
metadata = $4
|
||||
WHERE id = $1
|
||||
`, [docId, latFloat, lngFloat, JSON.stringify(metadata)]);
|
||||
|
||||
// Delete-then-reinsert must be atomic: without a transaction, a failure partway
|
||||
// through the insert loop leaves the document with its header already updated
|
||||
// above but only some (or none) of its items, since the delete has already
|
||||
// committed independently.
|
||||
await withTransaction(async (client) => {
|
||||
await client.query("DELETE FROM ocr_items WHERE document_id = $1", [docId]);
|
||||
|
||||
for (let i = 0; i < items.length; i++) {
|
||||
const item = items[i];
|
||||
const nomorSku = item.nomor_sku || item.nomorSku || "";
|
||||
const namaBarang = item.nama_barang || item.namaBarang || "";
|
||||
const banyak = item.banyak || "";
|
||||
const jumlah = item.jumlah || "";
|
||||
|
||||
await client.query(`
|
||||
INSERT INTO ocr_items (
|
||||
document_id, row_index,
|
||||
kode_barang_original, kode_barang,
|
||||
nama_barang,
|
||||
banyak_original, banyak,
|
||||
jumlah_original, jumlah,
|
||||
is_flagged, remark
|
||||
) VALUES ($1, $2, $3, $3, $4, $5, $5, $6, $6, false, '')
|
||||
`, [docId, i, nomorSku, namaBarang, banyak, jumlah]);
|
||||
}
|
||||
});
|
||||
|
||||
// Return the updated document mapping
|
||||
const mappedData = {
|
||||
id: docId.toString(),
|
||||
filePath: doc.filename,
|
||||
createdAt: doc.upload_time.toISOString(),
|
||||
header: {
|
||||
tanggal: tanggal || "",
|
||||
no_po: noPo || "",
|
||||
no_so: noSo || "",
|
||||
no_do: noDo || ""
|
||||
},
|
||||
shipment: {
|
||||
kepada_yth: kepadaYth || "",
|
||||
order_untuk: orderUntuk || "",
|
||||
alamat: alamat || "",
|
||||
plat_truk: platTruk || "",
|
||||
nama_driver: namaDriver || "",
|
||||
nama_penerima: namaPenerima || ""
|
||||
},
|
||||
items: items.map((item: any) => ({
|
||||
nomor_sku: item.nomor_sku || item.nomorSku || "",
|
||||
nama_barang: item.nama_barang || item.namaBarang || "",
|
||||
banyak: item.banyak || "",
|
||||
jumlah: item.jumlah || ""
|
||||
})),
|
||||
latitude: latFloat,
|
||||
longitude: lngFloat
|
||||
};
|
||||
|
||||
return NextResponse.json({
|
||||
status: "success",
|
||||
message: "Document updated successfully",
|
||||
data: mappedData
|
||||
}, { headers: corsHeaders });
|
||||
|
||||
} catch (error: unknown) {
|
||||
console.error("Error in update document API v1 route:", error);
|
||||
const message = error instanceof Error ? error.message : "Internal server error";
|
||||
return errorResponse(500, message, { headers: corsHeaders });
|
||||
}
|
||||
}
|
||||
@@ -1,66 +1,66 @@
|
||||
import { NextRequest, NextResponse } from "next/server";
|
||||
import { query } from "../../../../db";
|
||||
import { errorResponse } from "@/utils/api-error";
|
||||
import { getAccountFromAuthHeader } from "@/utils/auth";
|
||||
import { mapDocumentRow } from "@/utils/document-mapper";
|
||||
|
||||
const corsHeaders = {
|
||||
"Access-Control-Allow-Origin": "*",
|
||||
"Access-Control-Allow-Methods": "GET, POST, PUT, DELETE, OPTIONS",
|
||||
"Access-Control-Allow-Headers": "Content-Type, Authorization"
|
||||
};
|
||||
|
||||
export async function OPTIONS() {
|
||||
return new NextResponse(null, { status: 204, headers: corsHeaders });
|
||||
}
|
||||
|
||||
export async function GET(req: NextRequest) {
|
||||
try {
|
||||
const account = getAccountFromAuthHeader(req.headers.get("authorization"));
|
||||
if (!account) {
|
||||
return errorResponse(401, "Unauthorized", { headers: corsHeaders });
|
||||
}
|
||||
|
||||
// Retrieve all custom-uploaded documents
|
||||
let docsQuery = `
|
||||
SELECT id, filename, upload_time, size, parsed, is_sample, metadata, latitude, longitude, scan_mode, parse_error, confirmed
|
||||
FROM documents
|
||||
WHERE is_sample = false AND parsed = true AND confirmed = true
|
||||
`;
|
||||
const queryParams: any[] = [];
|
||||
|
||||
if (account.role !== 'admin') {
|
||||
docsQuery += ` AND kode_toko = $1`;
|
||||
queryParams.push(account.kodeToko);
|
||||
}
|
||||
|
||||
docsQuery += ` ORDER BY upload_time DESC`;
|
||||
|
||||
const docRes = await query(docsQuery, queryParams);
|
||||
|
||||
const documents = docRes.rows;
|
||||
const mappedList = [];
|
||||
|
||||
for (const doc of documents) {
|
||||
// Retrieve items from ocr_items
|
||||
const itemsRes = await query(`
|
||||
SELECT row_index, kode_barang, nama_barang, banyak, jumlah
|
||||
FROM ocr_items
|
||||
WHERE document_id = $1
|
||||
ORDER BY row_index
|
||||
`, [doc.id]);
|
||||
|
||||
mappedList.push(mapDocumentRow(doc, itemsRes.rows));
|
||||
}
|
||||
|
||||
return NextResponse.json({
|
||||
status: "success",
|
||||
data: mappedList
|
||||
}, { headers: corsHeaders });
|
||||
|
||||
} catch (error: unknown) {
|
||||
console.error("Error in list documents API v1 route:", error);
|
||||
const message = error instanceof Error ? error.message : "Internal server error";
|
||||
return errorResponse(500, message, { headers: corsHeaders });
|
||||
}
|
||||
}
|
||||
import { NextRequest, NextResponse } from "next/server";
|
||||
import { query } from "../../../../db";
|
||||
import { errorResponse } from "@/utils/api-error";
|
||||
import { getAccountFromAuthHeader } from "@/utils/auth";
|
||||
import { mapDocumentRow } from "@/utils/document-mapper";
|
||||
|
||||
const corsHeaders = {
|
||||
"Access-Control-Allow-Origin": "*",
|
||||
"Access-Control-Allow-Methods": "GET, POST, PUT, DELETE, OPTIONS",
|
||||
"Access-Control-Allow-Headers": "Content-Type, Authorization"
|
||||
};
|
||||
|
||||
export async function OPTIONS() {
|
||||
return new NextResponse(null, { status: 204, headers: corsHeaders });
|
||||
}
|
||||
|
||||
export async function GET(req: NextRequest) {
|
||||
try {
|
||||
const account = getAccountFromAuthHeader(req.headers.get("authorization"));
|
||||
if (!account) {
|
||||
return errorResponse(401, "Unauthorized", { headers: corsHeaders });
|
||||
}
|
||||
|
||||
// Retrieve all custom-uploaded documents
|
||||
let docsQuery = `
|
||||
SELECT id, filename, upload_time, size, parsed, is_sample, metadata, latitude, longitude, scan_mode, parse_error, confirmed
|
||||
FROM documents
|
||||
WHERE is_sample = false AND parsed = true AND confirmed = true
|
||||
`;
|
||||
const queryParams: any[] = [];
|
||||
|
||||
if (account.role !== 'admin') {
|
||||
docsQuery += ` AND kode_toko = $1`;
|
||||
queryParams.push(account.kodeToko);
|
||||
}
|
||||
|
||||
docsQuery += ` ORDER BY upload_time DESC`;
|
||||
|
||||
const docRes = await query(docsQuery, queryParams);
|
||||
|
||||
const documents = docRes.rows;
|
||||
const mappedList = [];
|
||||
|
||||
for (const doc of documents) {
|
||||
// Retrieve items from ocr_items
|
||||
const itemsRes = await query(`
|
||||
SELECT row_index, kode_barang, nama_barang, banyak, jumlah
|
||||
FROM ocr_items
|
||||
WHERE document_id = $1
|
||||
ORDER BY row_index
|
||||
`, [doc.id]);
|
||||
|
||||
mappedList.push(mapDocumentRow(doc, itemsRes.rows));
|
||||
}
|
||||
|
||||
return NextResponse.json({
|
||||
status: "success",
|
||||
data: mappedList
|
||||
}, { headers: corsHeaders });
|
||||
|
||||
} catch (error: unknown) {
|
||||
console.error("Error in list documents API v1 route:", error);
|
||||
const message = error instanceof Error ? error.message : "Internal server error";
|
||||
return errorResponse(500, message, { headers: corsHeaders });
|
||||
}
|
||||
}
|
||||
@@ -1,187 +1,187 @@
|
||||
import { NextRequest, NextResponse } from "next/server";
|
||||
import fs from "fs";
|
||||
import path from "path";
|
||||
import crypto from "crypto";
|
||||
import { query } from "../../../../../db";
|
||||
import { errorResponse } from "@/utils/api-error";
|
||||
import { getAccountFromAuthHeader } from "@/utils/auth";
|
||||
import { mapDocumentRow } from "@/utils/document-mapper";
|
||||
|
||||
const UPLOADS_DIR = "/uploads";
|
||||
|
||||
const corsHeaders = {
|
||||
"Access-Control-Allow-Origin": "*",
|
||||
"Access-Control-Allow-Methods": "GET, POST, PUT, DELETE, OPTIONS",
|
||||
"Access-Control-Allow-Headers": "Content-Type, Authorization"
|
||||
};
|
||||
|
||||
export async function OPTIONS() {
|
||||
return new NextResponse(null, { status: 204, headers: corsHeaders });
|
||||
}
|
||||
|
||||
export async function POST(req: NextRequest) {
|
||||
try {
|
||||
// Ensure uploads directory exists
|
||||
if (!fs.existsSync(UPLOADS_DIR)) {
|
||||
fs.mkdirSync(UPLOADS_DIR, { recursive: true });
|
||||
}
|
||||
|
||||
// The account uploading is assigned exactly one store (kode_toko) - pass
|
||||
// it through to /api/parse so store name/address are set directly from
|
||||
// that assignment instead of being OCR-detected from the document photo.
|
||||
const account = getAccountFromAuthHeader(req.headers.get("authorization"));
|
||||
if (!account) {
|
||||
return errorResponse(401, "Unauthorized", { headers: corsHeaders });
|
||||
}
|
||||
|
||||
const formData = await req.formData();
|
||||
const file = (formData.get("image") || formData.get("file")) as Blob | null;
|
||||
const scanMode = formData.get("scan_mode")?.toString() || "DO";
|
||||
console.log(`[Upload] Received scan_mode: "${scanMode}"`);
|
||||
|
||||
if (!file) {
|
||||
return errorResponse(400, "No file uploaded", { headers: corsHeaders });
|
||||
}
|
||||
|
||||
const originalName = file instanceof File ? file.name : "document.jpg";
|
||||
const safeName = path.basename(originalName).replace(/\s+/g, "_");
|
||||
const filename = `${Date.now()}-${safeName}`;
|
||||
const filePath = path.join(UPLOADS_DIR, filename);
|
||||
|
||||
// Compute hash before writing/inserting anything, so we can detect a duplicate
|
||||
// upload (e.g. the client retrying after a perceived timeout on a slow OCR pass)
|
||||
// without creating a second document row or re-running the pipeline on it.
|
||||
const arrayBuffer = await file.arrayBuffer();
|
||||
const buffer = Buffer.from(arrayBuffer);
|
||||
const fileHash = crypto.createHash("sha256").update(buffer).digest("hex");
|
||||
|
||||
// Geolocation tags
|
||||
const latVal = formData.get("latitude");
|
||||
const lngVal = formData.get("longitude");
|
||||
const latitude = latVal ? parseFloat(latVal.toString()) : null;
|
||||
const longitude = lngVal ? parseFloat(lngVal.toString()) : null;
|
||||
|
||||
// Basic dedup
|
||||
const dedupQuery = account?.kodeToko
|
||||
? "SELECT id, filename, upload_time, parsed, metadata, latitude, longitude, scan_mode, parse_error, confirmed FROM documents WHERE file_hash = $1 AND kode_toko = $2 ORDER BY upload_time ASC LIMIT 1"
|
||||
: "SELECT id, filename, upload_time, parsed, metadata, latitude, longitude, scan_mode, parse_error, confirmed FROM documents WHERE file_hash = $1 AND kode_toko IS NULL ORDER BY upload_time ASC LIMIT 1";
|
||||
const dedupParams = account?.kodeToko ? [fileHash, account.kodeToko] : [fileHash];
|
||||
|
||||
const existing = await query(dedupQuery, dedupParams);
|
||||
|
||||
if (existing.rows.length > 0) {
|
||||
const existingDoc = existing.rows[0];
|
||||
console.log(`[Dedup] Identical content already uploaded as document ${existingDoc.id}. Skipping duplicate insert and re-parse.`);
|
||||
|
||||
// Return the original document's actual current parse state instead of an
|
||||
// always-empty stub, so a retried upload doesn't look permanently "fresh."
|
||||
const itemsRes = await query(`
|
||||
SELECT row_index, kode_barang, nama_barang, banyak, jumlah
|
||||
FROM ocr_items
|
||||
WHERE document_id = $1
|
||||
ORDER BY row_index
|
||||
`, [existingDoc.id]);
|
||||
|
||||
const mappedData = mapDocumentRow(existingDoc, itemsRes.rows);
|
||||
// Fall back to this retry's own GPS tag if the original document never got one.
|
||||
if (mappedData.latitude === null) mappedData.latitude = latitude;
|
||||
if (mappedData.longitude === null) mappedData.longitude = longitude;
|
||||
|
||||
return NextResponse.json({
|
||||
status: "success",
|
||||
message: "Document already uploaded",
|
||||
data: mappedData
|
||||
}, { status: 201, headers: corsHeaders });
|
||||
}
|
||||
|
||||
// Save file
|
||||
fs.writeFileSync(filePath, buffer);
|
||||
|
||||
let docId: number;
|
||||
let finalFilename = filename;
|
||||
|
||||
// `confirmed = false`: this row isn't visible via GET /api/v1/documents
|
||||
// until the user's editor PUT confirms it (see docs/api-contract-map.md G11).
|
||||
const insertRes = await query(`
|
||||
INSERT INTO documents (filename, upload_time, size, parsed, is_sample, file_hash, latitude, longitude, kode_toko, scan_mode, confirmed)
|
||||
VALUES ($1, $2, $3, $4, $5, $6, $7, $8, $9, $10, $11)
|
||||
RETURNING id
|
||||
`, [
|
||||
filename,
|
||||
new Date(),
|
||||
buffer.length,
|
||||
false,
|
||||
false,
|
||||
fileHash,
|
||||
latitude,
|
||||
longitude,
|
||||
account?.kodeToko || null,
|
||||
scanMode,
|
||||
false
|
||||
]);
|
||||
docId = insertRes.rows[0].id;
|
||||
|
||||
// Trigger parsing synchronously to ensure it is processed immediately on receiving the image.
|
||||
// Bounded well above /api/parse's own per-pass pipeline timeout (2 passes worst case) so a
|
||||
// wedged GPU container doesn't hang this request forever - it still won't fit under the
|
||||
// mobile client's 2-minute receive timeout in the worst case, but bounds the hang to a fixed,
|
||||
// known ceiling instead of an indefinite one.
|
||||
//
|
||||
// /api/parse has its own error handlers that mark the document parsed=true with
|
||||
// "Not Found" placeholder metadata on a pipeline failure - so those cases already
|
||||
// resolve out of "pending". The one gap is this call itself never completing
|
||||
// (network error / the 210s abort firing): /api/parse's handlers never even run,
|
||||
// so the document is otherwise silently stuck at parsed=false forever. Record
|
||||
// that case explicitly so GET /api/v1/documents/:id can report parseStatus "failed"
|
||||
// instead of the client burning its own full timeout waiting on "pending".
|
||||
try {
|
||||
const parseRes = await fetch("http://127.0.0.1:3000/api/parse", {
|
||||
method: "POST",
|
||||
headers: { "Content-Type": "application/json" },
|
||||
body: JSON.stringify({ filename: finalFilename, kodeToko: account?.kodeToko, scanMode }),
|
||||
signal: AbortSignal.timeout(210_000)
|
||||
});
|
||||
if (!parseRes.ok) {
|
||||
await query("UPDATE documents SET parse_error = $1 WHERE id = $2", [`Pipeline error: HTTP ${parseRes.status}`, docId]);
|
||||
}
|
||||
} catch (err) {
|
||||
console.error("Error triggering parse synchronously:", err);
|
||||
const message = err instanceof Error ? err.message : "Parse request failed";
|
||||
await query("UPDATE documents SET parse_error = $1 WHERE id = $2", [message, docId]);
|
||||
}
|
||||
|
||||
// Return the response structured as DocumentModel.fromJson format
|
||||
const mappedData = {
|
||||
id: docId.toString(),
|
||||
header: {
|
||||
tanggal: "",
|
||||
no_po: "",
|
||||
no_so: "",
|
||||
no_do: ""
|
||||
},
|
||||
shipment: {
|
||||
kepada_yth: "PT.PRIMAFOOD INTERNATIONAL",
|
||||
order_untuk: "",
|
||||
alamat: "",
|
||||
plat_truk: "",
|
||||
nama_driver: "",
|
||||
nama_penerima: ""
|
||||
},
|
||||
items: [] as any[],
|
||||
latitude: latitude,
|
||||
longitude: longitude,
|
||||
createdAt: new Date().toISOString()
|
||||
};
|
||||
|
||||
return NextResponse.json({
|
||||
status: "success",
|
||||
message: "Document uploaded successfully",
|
||||
data: mappedData
|
||||
}, { status: 201, headers: corsHeaders });
|
||||
|
||||
} catch (error: unknown) {
|
||||
console.error("Error in upload API v1 route:", error);
|
||||
const message = error instanceof Error ? error.message : "Internal server error";
|
||||
return errorResponse(500, message, { headers: corsHeaders });
|
||||
}
|
||||
}
|
||||
import { NextRequest, NextResponse } from "next/server";
|
||||
import fs from "fs";
|
||||
import path from "path";
|
||||
import crypto from "crypto";
|
||||
import { query } from "../../../../../db";
|
||||
import { errorResponse } from "@/utils/api-error";
|
||||
import { getAccountFromAuthHeader } from "@/utils/auth";
|
||||
import { mapDocumentRow } from "@/utils/document-mapper";
|
||||
|
||||
const UPLOADS_DIR = "/uploads";
|
||||
|
||||
const corsHeaders = {
|
||||
"Access-Control-Allow-Origin": "*",
|
||||
"Access-Control-Allow-Methods": "GET, POST, PUT, DELETE, OPTIONS",
|
||||
"Access-Control-Allow-Headers": "Content-Type, Authorization"
|
||||
};
|
||||
|
||||
export async function OPTIONS() {
|
||||
return new NextResponse(null, { status: 204, headers: corsHeaders });
|
||||
}
|
||||
|
||||
export async function POST(req: NextRequest) {
|
||||
try {
|
||||
// Ensure uploads directory exists
|
||||
if (!fs.existsSync(UPLOADS_DIR)) {
|
||||
fs.mkdirSync(UPLOADS_DIR, { recursive: true });
|
||||
}
|
||||
|
||||
// The account uploading is assigned exactly one store (kode_toko) - pass
|
||||
// it through to /api/parse so store name/address are set directly from
|
||||
// that assignment instead of being OCR-detected from the document photo.
|
||||
const account = getAccountFromAuthHeader(req.headers.get("authorization"));
|
||||
if (!account) {
|
||||
return errorResponse(401, "Unauthorized", { headers: corsHeaders });
|
||||
}
|
||||
|
||||
const formData = await req.formData();
|
||||
const file = (formData.get("image") || formData.get("file")) as Blob | null;
|
||||
const scanMode = formData.get("scan_mode")?.toString() || "DO";
|
||||
console.log(`[Upload] Received scan_mode: "${scanMode}"`);
|
||||
|
||||
if (!file) {
|
||||
return errorResponse(400, "No file uploaded", { headers: corsHeaders });
|
||||
}
|
||||
|
||||
const originalName = file instanceof File ? file.name : "document.jpg";
|
||||
const safeName = path.basename(originalName).replace(/\s+/g, "_");
|
||||
const filename = `${Date.now()}-${safeName}`;
|
||||
const filePath = path.join(UPLOADS_DIR, filename);
|
||||
|
||||
// Compute hash before writing/inserting anything, so we can detect a duplicate
|
||||
// upload (e.g. the client retrying after a perceived timeout on a slow OCR pass)
|
||||
// without creating a second document row or re-running the pipeline on it.
|
||||
const arrayBuffer = await file.arrayBuffer();
|
||||
const buffer = Buffer.from(arrayBuffer);
|
||||
const fileHash = crypto.createHash("sha256").update(buffer).digest("hex");
|
||||
|
||||
// Geolocation tags
|
||||
const latVal = formData.get("latitude");
|
||||
const lngVal = formData.get("longitude");
|
||||
const latitude = latVal ? parseFloat(latVal.toString()) : null;
|
||||
const longitude = lngVal ? parseFloat(lngVal.toString()) : null;
|
||||
|
||||
// Basic dedup
|
||||
const dedupQuery = account?.kodeToko
|
||||
? "SELECT id, filename, upload_time, parsed, metadata, latitude, longitude, scan_mode, parse_error, confirmed FROM documents WHERE file_hash = $1 AND kode_toko = $2 ORDER BY upload_time ASC LIMIT 1"
|
||||
: "SELECT id, filename, upload_time, parsed, metadata, latitude, longitude, scan_mode, parse_error, confirmed FROM documents WHERE file_hash = $1 AND kode_toko IS NULL ORDER BY upload_time ASC LIMIT 1";
|
||||
const dedupParams = account?.kodeToko ? [fileHash, account.kodeToko] : [fileHash];
|
||||
|
||||
const existing = await query(dedupQuery, dedupParams);
|
||||
|
||||
if (existing.rows.length > 0) {
|
||||
const existingDoc = existing.rows[0];
|
||||
console.log(`[Dedup] Identical content already uploaded as document ${existingDoc.id}. Skipping duplicate insert and re-parse.`);
|
||||
|
||||
// Return the original document's actual current parse state instead of an
|
||||
// always-empty stub, so a retried upload doesn't look permanently "fresh."
|
||||
const itemsRes = await query(`
|
||||
SELECT row_index, kode_barang, nama_barang, banyak, jumlah
|
||||
FROM ocr_items
|
||||
WHERE document_id = $1
|
||||
ORDER BY row_index
|
||||
`, [existingDoc.id]);
|
||||
|
||||
const mappedData = mapDocumentRow(existingDoc, itemsRes.rows);
|
||||
// Fall back to this retry's own GPS tag if the original document never got one.
|
||||
if (mappedData.latitude === null) mappedData.latitude = latitude;
|
||||
if (mappedData.longitude === null) mappedData.longitude = longitude;
|
||||
|
||||
return NextResponse.json({
|
||||
status: "success",
|
||||
message: "Document already uploaded",
|
||||
data: mappedData
|
||||
}, { status: 201, headers: corsHeaders });
|
||||
}
|
||||
|
||||
// Save file
|
||||
fs.writeFileSync(filePath, buffer);
|
||||
|
||||
let docId: number;
|
||||
let finalFilename = filename;
|
||||
|
||||
// `confirmed = false`: this row isn't visible via GET /api/v1/documents
|
||||
// until the user's editor PUT confirms it (see docs/api-contract-map.md G11).
|
||||
const insertRes = await query(`
|
||||
INSERT INTO documents (filename, upload_time, size, parsed, is_sample, file_hash, latitude, longitude, kode_toko, scan_mode, confirmed)
|
||||
VALUES ($1, $2, $3, $4, $5, $6, $7, $8, $9, $10, $11)
|
||||
RETURNING id
|
||||
`, [
|
||||
filename,
|
||||
new Date(),
|
||||
buffer.length,
|
||||
false,
|
||||
false,
|
||||
fileHash,
|
||||
latitude,
|
||||
longitude,
|
||||
account?.kodeToko || null,
|
||||
scanMode,
|
||||
false
|
||||
]);
|
||||
docId = insertRes.rows[0].id;
|
||||
|
||||
// Trigger parsing synchronously to ensure it is processed immediately on receiving the image.
|
||||
// Bounded well above /api/parse's own per-pass pipeline timeout (2 passes worst case) so a
|
||||
// wedged GPU container doesn't hang this request forever - it still won't fit under the
|
||||
// mobile client's 2-minute receive timeout in the worst case, but bounds the hang to a fixed,
|
||||
// known ceiling instead of an indefinite one.
|
||||
//
|
||||
// /api/parse has its own error handlers that mark the document parsed=true with
|
||||
// "Not Found" placeholder metadata on a pipeline failure - so those cases already
|
||||
// resolve out of "pending". The one gap is this call itself never completing
|
||||
// (network error / the 210s abort firing): /api/parse's handlers never even run,
|
||||
// so the document is otherwise silently stuck at parsed=false forever. Record
|
||||
// that case explicitly so GET /api/v1/documents/:id can report parseStatus "failed"
|
||||
// instead of the client burning its own full timeout waiting on "pending".
|
||||
try {
|
||||
const parseRes = await fetch("http://127.0.0.1:3000/api/parse", {
|
||||
method: "POST",
|
||||
headers: { "Content-Type": "application/json" },
|
||||
body: JSON.stringify({ filename: finalFilename, kodeToko: account?.kodeToko, scanMode }),
|
||||
signal: AbortSignal.timeout(210_000)
|
||||
});
|
||||
if (!parseRes.ok) {
|
||||
await query("UPDATE documents SET parse_error = $1 WHERE id = $2", [`Pipeline error: HTTP ${parseRes.status}`, docId]);
|
||||
}
|
||||
} catch (err) {
|
||||
console.error("Error triggering parse synchronously:", err);
|
||||
const message = err instanceof Error ? err.message : "Parse request failed";
|
||||
await query("UPDATE documents SET parse_error = $1 WHERE id = $2", [message, docId]);
|
||||
}
|
||||
|
||||
// Return the response structured as DocumentModel.fromJson format
|
||||
const mappedData = {
|
||||
id: docId.toString(),
|
||||
header: {
|
||||
tanggal: "",
|
||||
no_po: "",
|
||||
no_so: "",
|
||||
no_do: ""
|
||||
},
|
||||
shipment: {
|
||||
kepada_yth: "PT.PRIMAFOOD INTERNATIONAL",
|
||||
order_untuk: "",
|
||||
alamat: "",
|
||||
plat_truk: "",
|
||||
nama_driver: "",
|
||||
nama_penerima: ""
|
||||
},
|
||||
items: [] as any[],
|
||||
latitude: latitude,
|
||||
longitude: longitude,
|
||||
createdAt: new Date().toISOString()
|
||||
};
|
||||
|
||||
return NextResponse.json({
|
||||
status: "success",
|
||||
message: "Document uploaded successfully",
|
||||
data: mappedData
|
||||
}, { status: 201, headers: corsHeaders });
|
||||
|
||||
} catch (error: unknown) {
|
||||
console.error("Error in upload API v1 route:", error);
|
||||
const message = error instanceof Error ? error.message : "Internal server error";
|
||||
return errorResponse(500, message, { headers: corsHeaders });
|
||||
}
|
||||
}
|
||||
@@ -1,56 +1,56 @@
|
||||
import { NextResponse } from "next/server";
|
||||
import { query } from "../../../../db";
|
||||
|
||||
const corsHeaders = {
|
||||
"Access-Control-Allow-Origin": "*",
|
||||
"Access-Control-Allow-Methods": "GET, OPTIONS",
|
||||
"Access-Control-Allow-Headers": "Content-Type, Authorization"
|
||||
};
|
||||
|
||||
export async function OPTIONS() {
|
||||
return new NextResponse(null, { status: 204, headers: corsHeaders });
|
||||
}
|
||||
|
||||
export async function GET() {
|
||||
let dbHealthy = false;
|
||||
let pipelineHealthy = false;
|
||||
|
||||
// Check Database
|
||||
try {
|
||||
const res = await query("SELECT 1 as healthy");
|
||||
if (res.rowCount && res.rows[0].healthy === 1) {
|
||||
dbHealthy = true;
|
||||
}
|
||||
} catch (err) {
|
||||
console.error("Health check - DB ping failed:", err);
|
||||
}
|
||||
|
||||
// Check Pipeline API
|
||||
try {
|
||||
const pipelineUrl = process.env.PIPELINE_URL;
|
||||
// e.g. http://paddleocr-pipeline-api:8090/layout-parsing
|
||||
if (pipelineUrl) {
|
||||
const healthUrl = new URL("/", pipelineUrl).toString();
|
||||
const response = await fetch(healthUrl, { method: "GET", signal: AbortSignal.timeout(3000) });
|
||||
// As long as the server responds (even with 404 or 405), it is running.
|
||||
if (response.status) {
|
||||
pipelineHealthy = true;
|
||||
}
|
||||
} else {
|
||||
console.warn("Health check - PIPELINE_URL not configured in environment");
|
||||
}
|
||||
} catch (err) {
|
||||
console.error("Health check - Pipeline ping failed:", err);
|
||||
}
|
||||
|
||||
const isHealthy = dbHealthy && pipelineHealthy;
|
||||
|
||||
return NextResponse.json({
|
||||
status: isHealthy ? "ok" : "error",
|
||||
db: dbHealthy,
|
||||
pipeline: pipelineHealthy,
|
||||
}, {
|
||||
status: isHealthy ? 200 : 503,
|
||||
headers: corsHeaders
|
||||
});
|
||||
}
|
||||
import { NextResponse } from "next/server";
|
||||
import { query } from "../../../../db";
|
||||
|
||||
const corsHeaders = {
|
||||
"Access-Control-Allow-Origin": "*",
|
||||
"Access-Control-Allow-Methods": "GET, OPTIONS",
|
||||
"Access-Control-Allow-Headers": "Content-Type, Authorization"
|
||||
};
|
||||
|
||||
export async function OPTIONS() {
|
||||
return new NextResponse(null, { status: 204, headers: corsHeaders });
|
||||
}
|
||||
|
||||
export async function GET() {
|
||||
let dbHealthy = false;
|
||||
let pipelineHealthy = false;
|
||||
|
||||
// Check Database
|
||||
try {
|
||||
const res = await query("SELECT 1 as healthy");
|
||||
if (res.rowCount && res.rows[0].healthy === 1) {
|
||||
dbHealthy = true;
|
||||
}
|
||||
} catch (err) {
|
||||
console.error("Health check - DB ping failed:", err);
|
||||
}
|
||||
|
||||
// Check Pipeline API
|
||||
try {
|
||||
const pipelineUrl = process.env.PIPELINE_URL;
|
||||
// e.g. http://paddleocr-pipeline-api:8090/layout-parsing
|
||||
if (pipelineUrl) {
|
||||
const healthUrl = new URL("/", pipelineUrl).toString();
|
||||
const response = await fetch(healthUrl, { method: "GET", signal: AbortSignal.timeout(3000) });
|
||||
// As long as the server responds (even with 404 or 405), it is running.
|
||||
if (response.status) {
|
||||
pipelineHealthy = true;
|
||||
}
|
||||
} else {
|
||||
console.warn("Health check - PIPELINE_URL not configured in environment");
|
||||
}
|
||||
} catch (err) {
|
||||
console.error("Health check - Pipeline ping failed:", err);
|
||||
}
|
||||
|
||||
const isHealthy = dbHealthy && pipelineHealthy;
|
||||
|
||||
return NextResponse.json({
|
||||
status: isHealthy ? "ok" : "error",
|
||||
db: dbHealthy,
|
||||
pipeline: pipelineHealthy,
|
||||
}, {
|
||||
status: isHealthy ? 200 : 503,
|
||||
headers: corsHeaders
|
||||
});
|
||||
}
|
||||
@@ -1,71 +1,71 @@
|
||||
import { NextRequest, NextResponse } from "next/server";
|
||||
import { query } from "@/db";
|
||||
import { errorResponse } from "@/utils/api-error";
|
||||
import { getAccountFromAuthHeader } from "@/utils/auth";
|
||||
|
||||
export async function PUT(
|
||||
req: NextRequest,
|
||||
context: { params: Promise<{ kode: string }> }
|
||||
) {
|
||||
try {
|
||||
const account = getAccountFromAuthHeader(req.headers.get("authorization"));
|
||||
if (!account || account.role !== 'admin') {
|
||||
return errorResponse(403, "Forbidden: Admin access required");
|
||||
}
|
||||
|
||||
const { kode } = await context.params;
|
||||
const body = await req.json();
|
||||
const {
|
||||
nama_item,
|
||||
jenis_outer,
|
||||
standar_jumlah
|
||||
} = body;
|
||||
|
||||
const res = await query(
|
||||
`UPDATE sku_master
|
||||
SET nama_item = $1, jenis_outer = $2, standar_jumlah = $3
|
||||
WHERE no_sku = $4 RETURNING *`,
|
||||
[
|
||||
nama_item,
|
||||
jenis_outer || '',
|
||||
String(standar_jumlah || '1'),
|
||||
kode
|
||||
]
|
||||
);
|
||||
|
||||
if (res.rowCount === 0) {
|
||||
return errorResponse(404, "SKU not found");
|
||||
}
|
||||
|
||||
return NextResponse.json({ status: "success", data: res.rows[0] });
|
||||
} catch (err: any) {
|
||||
return errorResponse(500, err.message);
|
||||
}
|
||||
}
|
||||
|
||||
export async function DELETE(
|
||||
req: NextRequest,
|
||||
context: { params: Promise<{ kode: string }> }
|
||||
) {
|
||||
try {
|
||||
const account = getAccountFromAuthHeader(req.headers.get("authorization"));
|
||||
if (!account || account.role !== 'admin') {
|
||||
return errorResponse(403, "Forbidden: Admin access required");
|
||||
}
|
||||
|
||||
const { kode } = await context.params;
|
||||
|
||||
const res = await query("DELETE FROM sku_master WHERE no_sku = $1 RETURNING *", [kode]);
|
||||
|
||||
if (res.rowCount === 0) {
|
||||
return errorResponse(404, "SKU not found");
|
||||
}
|
||||
|
||||
return NextResponse.json({ status: "success", message: "SKU deleted successfully" });
|
||||
} catch (err: any) {
|
||||
if (err.code === '23503') { // foreign key violation
|
||||
return errorResponse(409, "Cannot delete SKU because it is referenced in documents");
|
||||
}
|
||||
return errorResponse(500, err.message);
|
||||
}
|
||||
}
|
||||
import { NextRequest, NextResponse } from "next/server";
|
||||
import { query } from "@/db";
|
||||
import { errorResponse } from "@/utils/api-error";
|
||||
import { getAccountFromAuthHeader } from "@/utils/auth";
|
||||
|
||||
export async function PUT(
|
||||
req: NextRequest,
|
||||
context: { params: Promise<{ kode: string }> }
|
||||
) {
|
||||
try {
|
||||
const account = getAccountFromAuthHeader(req.headers.get("authorization"));
|
||||
if (!account || account.role !== 'admin') {
|
||||
return errorResponse(403, "Forbidden: Admin access required");
|
||||
}
|
||||
|
||||
const { kode } = await context.params;
|
||||
const body = await req.json();
|
||||
const {
|
||||
nama_item,
|
||||
jenis_outer,
|
||||
standar_jumlah
|
||||
} = body;
|
||||
|
||||
const res = await query(
|
||||
`UPDATE sku_master
|
||||
SET nama_item = $1, jenis_outer = $2, standar_jumlah = $3
|
||||
WHERE no_sku = $4 RETURNING *`,
|
||||
[
|
||||
nama_item,
|
||||
jenis_outer || '',
|
||||
String(standar_jumlah || '1'),
|
||||
kode
|
||||
]
|
||||
);
|
||||
|
||||
if (res.rowCount === 0) {
|
||||
return errorResponse(404, "SKU not found");
|
||||
}
|
||||
|
||||
return NextResponse.json({ status: "success", data: res.rows[0] });
|
||||
} catch (err: any) {
|
||||
return errorResponse(500, err.message);
|
||||
}
|
||||
}
|
||||
|
||||
export async function DELETE(
|
||||
req: NextRequest,
|
||||
context: { params: Promise<{ kode: string }> }
|
||||
) {
|
||||
try {
|
||||
const account = getAccountFromAuthHeader(req.headers.get("authorization"));
|
||||
if (!account || account.role !== 'admin') {
|
||||
return errorResponse(403, "Forbidden: Admin access required");
|
||||
}
|
||||
|
||||
const { kode } = await context.params;
|
||||
|
||||
const res = await query("DELETE FROM sku_master WHERE no_sku = $1 RETURNING *", [kode]);
|
||||
|
||||
if (res.rowCount === 0) {
|
||||
return errorResponse(404, "SKU not found");
|
||||
}
|
||||
|
||||
return NextResponse.json({ status: "success", message: "SKU deleted successfully" });
|
||||
} catch (err: any) {
|
||||
if (err.code === '23503') { // foreign key violation
|
||||
return errorResponse(409, "Cannot delete SKU because it is referenced in documents");
|
||||
}
|
||||
return errorResponse(500, err.message);
|
||||
}
|
||||
}
|
||||
@@ -1,68 +1,68 @@
|
||||
import { NextRequest, NextResponse } from "next/server";
|
||||
import { query } from "@/db";
|
||||
import { errorResponse } from "@/utils/api-error";
|
||||
import { getAccountFromAuthHeader } from "@/utils/auth";
|
||||
|
||||
export async function GET(req: NextRequest) {
|
||||
try {
|
||||
// Read access is open to any authenticated account (task 9.2) - the
|
||||
// Flutter product editor needs this to populate its SKU dropdown, and
|
||||
// has no admin role of its own. Writes below stay admin-gated.
|
||||
const account = getAccountFromAuthHeader(req.headers.get("authorization"));
|
||||
if (!account) {
|
||||
return errorResponse(401, "Unauthorized");
|
||||
}
|
||||
|
||||
const res = await query(`
|
||||
SELECT no_sku, nama_item, standar_jumlah, berat_kemasan, isi_outer_kg, isi_outer_pac, jenis_outer
|
||||
FROM sku_master
|
||||
ORDER BY no_sku ASC
|
||||
`);
|
||||
return NextResponse.json({ status: "success", data: res.rows });
|
||||
} catch (err: any) {
|
||||
return errorResponse(500, err.message);
|
||||
}
|
||||
}
|
||||
|
||||
export async function POST(req: NextRequest) {
|
||||
try {
|
||||
const account = getAccountFromAuthHeader(req.headers.get("authorization"));
|
||||
if (!account || account.role !== 'admin') {
|
||||
return errorResponse(403, "Forbidden: Admin access required");
|
||||
}
|
||||
|
||||
const body = await req.json();
|
||||
const {
|
||||
kode_item,
|
||||
no_sku,
|
||||
nama_item,
|
||||
jenis_outer,
|
||||
standar_jumlah
|
||||
} = body;
|
||||
|
||||
const skuCode = no_sku || kode_item;
|
||||
|
||||
if (!skuCode || !nama_item) {
|
||||
return errorResponse(400, "no_sku and nama_item are required");
|
||||
}
|
||||
|
||||
await query(
|
||||
`INSERT INTO sku_master
|
||||
(no_sku, nama_item, jenis_outer, standar_jumlah)
|
||||
VALUES ($1, $2, $3, $4)`,
|
||||
[
|
||||
skuCode,
|
||||
nama_item,
|
||||
jenis_outer || '',
|
||||
String(standar_jumlah || '1')
|
||||
]
|
||||
);
|
||||
|
||||
return NextResponse.json({ status: "success", message: "SKU created successfully" });
|
||||
} catch (err: any) {
|
||||
if (err.code === '23505') { // unique violation
|
||||
return errorResponse(409, "SKU with this kode_item already exists");
|
||||
}
|
||||
return errorResponse(500, err.message);
|
||||
}
|
||||
}
|
||||
import { NextRequest, NextResponse } from "next/server";
|
||||
import { query } from "@/db";
|
||||
import { errorResponse } from "@/utils/api-error";
|
||||
import { getAccountFromAuthHeader } from "@/utils/auth";
|
||||
|
||||
export async function GET(req: NextRequest) {
|
||||
try {
|
||||
// Read access is open to any authenticated account (task 9.2) - the
|
||||
// Flutter product editor needs this to populate its SKU dropdown, and
|
||||
// has no admin role of its own. Writes below stay admin-gated.
|
||||
const account = getAccountFromAuthHeader(req.headers.get("authorization"));
|
||||
if (!account) {
|
||||
return errorResponse(401, "Unauthorized");
|
||||
}
|
||||
|
||||
const res = await query(`
|
||||
SELECT no_sku, nama_item, standar_jumlah, berat_kemasan, isi_outer_kg, isi_outer_pac, jenis_outer
|
||||
FROM sku_master
|
||||
ORDER BY no_sku ASC
|
||||
`);
|
||||
return NextResponse.json({ status: "success", data: res.rows });
|
||||
} catch (err: any) {
|
||||
return errorResponse(500, err.message);
|
||||
}
|
||||
}
|
||||
|
||||
export async function POST(req: NextRequest) {
|
||||
try {
|
||||
const account = getAccountFromAuthHeader(req.headers.get("authorization"));
|
||||
if (!account || account.role !== 'admin') {
|
||||
return errorResponse(403, "Forbidden: Admin access required");
|
||||
}
|
||||
|
||||
const body = await req.json();
|
||||
const {
|
||||
kode_item,
|
||||
no_sku,
|
||||
nama_item,
|
||||
jenis_outer,
|
||||
standar_jumlah
|
||||
} = body;
|
||||
|
||||
const skuCode = no_sku || kode_item;
|
||||
|
||||
if (!skuCode || !nama_item) {
|
||||
return errorResponse(400, "no_sku and nama_item are required");
|
||||
}
|
||||
|
||||
await query(
|
||||
`INSERT INTO sku_master
|
||||
(no_sku, nama_item, jenis_outer, standar_jumlah)
|
||||
VALUES ($1, $2, $3, $4)`,
|
||||
[
|
||||
skuCode,
|
||||
nama_item,
|
||||
jenis_outer || '',
|
||||
String(standar_jumlah || '1')
|
||||
]
|
||||
);
|
||||
|
||||
return NextResponse.json({ status: "success", message: "SKU created successfully" });
|
||||
} catch (err: any) {
|
||||
if (err.code === '23505') { // unique violation
|
||||
return errorResponse(409, "SKU with this kode_item already exists");
|
||||
}
|
||||
return errorResponse(500, err.message);
|
||||
}
|
||||
}
|
||||
@@ -1,68 +1,68 @@
|
||||
import { NextRequest, NextResponse } from "next/server";
|
||||
import { query, withTransaction } from "@/db";
|
||||
import { errorResponse } from "@/utils/api-error";
|
||||
import { getAccountFromAuthHeader } from "@/utils/auth";
|
||||
|
||||
export async function PUT(
|
||||
req: NextRequest,
|
||||
context: { params: Promise<{ kode: string }> }
|
||||
) {
|
||||
try {
|
||||
const account = getAccountFromAuthHeader(req.headers.get("authorization"));
|
||||
if (!account || account.role !== 'admin') {
|
||||
return errorResponse(403, "Forbidden: Admin access required");
|
||||
}
|
||||
|
||||
const { kode } = await context.params;
|
||||
const body = await req.json();
|
||||
const { nama_toko, alamat } = body;
|
||||
|
||||
const res = await query(
|
||||
"UPDATE store_master SET nama_toko = $1, alamat = $2 WHERE kode_toko = $3 RETURNING *",
|
||||
[nama_toko, alamat || '', kode]
|
||||
);
|
||||
|
||||
if (res.rowCount === 0) {
|
||||
return errorResponse(404, "Store not found");
|
||||
}
|
||||
|
||||
return NextResponse.json({ status: "success", data: res.rows[0] });
|
||||
} catch (err: any) {
|
||||
return errorResponse(500, err.message);
|
||||
}
|
||||
}
|
||||
|
||||
export async function DELETE(
|
||||
req: NextRequest,
|
||||
context: { params: Promise<{ kode: string }> }
|
||||
) {
|
||||
try {
|
||||
const account = getAccountFromAuthHeader(req.headers.get("authorization"));
|
||||
if (!account || account.role !== 'admin') {
|
||||
return errorResponse(403, "Forbidden: Admin access required");
|
||||
}
|
||||
|
||||
const { kode } = await context.params;
|
||||
|
||||
await withTransaction(async (client) => {
|
||||
// Delete associated account first due to FK account -> store_master
|
||||
await client.query("DELETE FROM accounts WHERE kode_toko = $1", [kode]);
|
||||
|
||||
const res = await client.query("DELETE FROM store_master WHERE kode_toko = $1 RETURNING *", [kode]);
|
||||
|
||||
if (res.rowCount === 0) {
|
||||
throw new Error("Store not found");
|
||||
}
|
||||
});
|
||||
|
||||
return NextResponse.json({ status: "success", message: "Store and associated account deleted successfully" });
|
||||
} catch (err: any) {
|
||||
if (err.code === '23503') { // foreign key violation (e.g. documents exist)
|
||||
return errorResponse(409, "Cannot delete store because it has associated documents");
|
||||
}
|
||||
if (err.message === "Store not found") {
|
||||
return errorResponse(404, err.message);
|
||||
}
|
||||
return errorResponse(500, err.message);
|
||||
}
|
||||
}
|
||||
import { NextRequest, NextResponse } from "next/server";
|
||||
import { query, withTransaction } from "@/db";
|
||||
import { errorResponse } from "@/utils/api-error";
|
||||
import { getAccountFromAuthHeader } from "@/utils/auth";
|
||||
|
||||
export async function PUT(
|
||||
req: NextRequest,
|
||||
context: { params: Promise<{ kode: string }> }
|
||||
) {
|
||||
try {
|
||||
const account = getAccountFromAuthHeader(req.headers.get("authorization"));
|
||||
if (!account || account.role !== 'admin') {
|
||||
return errorResponse(403, "Forbidden: Admin access required");
|
||||
}
|
||||
|
||||
const { kode } = await context.params;
|
||||
const body = await req.json();
|
||||
const { nama_toko, alamat } = body;
|
||||
|
||||
const res = await query(
|
||||
"UPDATE store_master SET nama_toko = $1, alamat = $2 WHERE kode_toko = $3 RETURNING *",
|
||||
[nama_toko, alamat || '', kode]
|
||||
);
|
||||
|
||||
if (res.rowCount === 0) {
|
||||
return errorResponse(404, "Store not found");
|
||||
}
|
||||
|
||||
return NextResponse.json({ status: "success", data: res.rows[0] });
|
||||
} catch (err: any) {
|
||||
return errorResponse(500, err.message);
|
||||
}
|
||||
}
|
||||
|
||||
export async function DELETE(
|
||||
req: NextRequest,
|
||||
context: { params: Promise<{ kode: string }> }
|
||||
) {
|
||||
try {
|
||||
const account = getAccountFromAuthHeader(req.headers.get("authorization"));
|
||||
if (!account || account.role !== 'admin') {
|
||||
return errorResponse(403, "Forbidden: Admin access required");
|
||||
}
|
||||
|
||||
const { kode } = await context.params;
|
||||
|
||||
await withTransaction(async (client) => {
|
||||
// Delete associated account first due to FK account -> store_master
|
||||
await client.query("DELETE FROM accounts WHERE kode_toko = $1", [kode]);
|
||||
|
||||
const res = await client.query("DELETE FROM store_master WHERE kode_toko = $1 RETURNING *", [kode]);
|
||||
|
||||
if (res.rowCount === 0) {
|
||||
throw new Error("Store not found");
|
||||
}
|
||||
});
|
||||
|
||||
return NextResponse.json({ status: "success", message: "Store and associated account deleted successfully" });
|
||||
} catch (err: any) {
|
||||
if (err.code === '23503') { // foreign key violation (e.g. documents exist)
|
||||
return errorResponse(409, "Cannot delete store because it has associated documents");
|
||||
}
|
||||
if (err.message === "Store not found") {
|
||||
return errorResponse(404, err.message);
|
||||
}
|
||||
return errorResponse(500, err.message);
|
||||
}
|
||||
}
|
||||
@@ -1,63 +1,63 @@
|
||||
import { NextRequest, NextResponse } from "next/server";
|
||||
import { query, withTransaction } from "@/db";
|
||||
import { errorResponse } from "@/utils/api-error";
|
||||
import { getAccountFromAuthHeader } from "@/utils/auth";
|
||||
import bcrypt from "bcryptjs";
|
||||
|
||||
export async function GET(req: NextRequest) {
|
||||
try {
|
||||
const authHeader = req.headers.get("authorization");
|
||||
console.log("Auth Header in GET:", authHeader);
|
||||
const account = getAccountFromAuthHeader(authHeader);
|
||||
console.log("Decoded Account:", account);
|
||||
if (!account || account.role !== 'admin') {
|
||||
return errorResponse(403, "Forbidden: Admin access required");
|
||||
}
|
||||
|
||||
const res = await query("SELECT kode_toko, nama_toko, alamat FROM store_master ORDER BY kode_toko ASC");
|
||||
return NextResponse.json({ status: "success", data: res.rows });
|
||||
} catch (err: any) {
|
||||
return errorResponse(500, err.message);
|
||||
}
|
||||
}
|
||||
|
||||
export async function POST(req: NextRequest) {
|
||||
try {
|
||||
const account = getAccountFromAuthHeader(req.headers.get("authorization"));
|
||||
if (!account || account.role !== 'admin') {
|
||||
return errorResponse(403, "Forbidden: Admin access required");
|
||||
}
|
||||
|
||||
const body = await req.json();
|
||||
const { kode_toko, nama_toko, alamat } = body;
|
||||
|
||||
if (!kode_toko || !nama_toko) {
|
||||
return errorResponse(400, "kode_toko and nama_toko are required");
|
||||
}
|
||||
|
||||
await withTransaction(async (client) => {
|
||||
// 1. Insert store
|
||||
await client.query(
|
||||
"INSERT INTO store_master (kode_toko, nama_toko, alamat) VALUES ($1, $2, $3)",
|
||||
[kode_toko, nama_toko, alamat || '']
|
||||
);
|
||||
|
||||
// 2. Hash default password
|
||||
const hashedPassword = await bcrypt.hash('123', 10);
|
||||
|
||||
// 3. Create default account
|
||||
await client.query(
|
||||
`INSERT INTO accounts (username, password, role, is_active, kode_toko)
|
||||
VALUES ($1, $2, 'store', true, $3)`,
|
||||
[kode_toko, hashedPassword, kode_toko]
|
||||
);
|
||||
});
|
||||
|
||||
return NextResponse.json({ status: "success", message: "Store and account created successfully" });
|
||||
} catch (err: any) {
|
||||
if (err.code === '23505') { // unique violation
|
||||
return errorResponse(409, "Store with this kode_toko already exists");
|
||||
}
|
||||
return errorResponse(500, err.message);
|
||||
}
|
||||
}
|
||||
import { NextRequest, NextResponse } from "next/server";
|
||||
import { query, withTransaction } from "@/db";
|
||||
import { errorResponse } from "@/utils/api-error";
|
||||
import { getAccountFromAuthHeader } from "@/utils/auth";
|
||||
import bcrypt from "bcryptjs";
|
||||
|
||||
export async function GET(req: NextRequest) {
|
||||
try {
|
||||
const authHeader = req.headers.get("authorization");
|
||||
console.log("Auth Header in GET:", authHeader);
|
||||
const account = getAccountFromAuthHeader(authHeader);
|
||||
console.log("Decoded Account:", account);
|
||||
if (!account || account.role !== 'admin') {
|
||||
return errorResponse(403, "Forbidden: Admin access required");
|
||||
}
|
||||
|
||||
const res = await query("SELECT kode_toko, nama_toko, alamat FROM store_master ORDER BY kode_toko ASC");
|
||||
return NextResponse.json({ status: "success", data: res.rows });
|
||||
} catch (err: any) {
|
||||
return errorResponse(500, err.message);
|
||||
}
|
||||
}
|
||||
|
||||
export async function POST(req: NextRequest) {
|
||||
try {
|
||||
const account = getAccountFromAuthHeader(req.headers.get("authorization"));
|
||||
if (!account || account.role !== 'admin') {
|
||||
return errorResponse(403, "Forbidden: Admin access required");
|
||||
}
|
||||
|
||||
const body = await req.json();
|
||||
const { kode_toko, nama_toko, alamat } = body;
|
||||
|
||||
if (!kode_toko || !nama_toko) {
|
||||
return errorResponse(400, "kode_toko and nama_toko are required");
|
||||
}
|
||||
|
||||
await withTransaction(async (client) => {
|
||||
// 1. Insert store
|
||||
await client.query(
|
||||
"INSERT INTO store_master (kode_toko, nama_toko, alamat) VALUES ($1, $2, $3)",
|
||||
[kode_toko, nama_toko, alamat || '']
|
||||
);
|
||||
|
||||
// 2. Hash default password
|
||||
const hashedPassword = await bcrypt.hash('123', 10);
|
||||
|
||||
// 3. Create default account
|
||||
await client.query(
|
||||
`INSERT INTO accounts (username, password, role, is_active, kode_toko)
|
||||
VALUES ($1, $2, 'store', true, $3)`,
|
||||
[kode_toko, hashedPassword, kode_toko]
|
||||
);
|
||||
});
|
||||
|
||||
return NextResponse.json({ status: "success", message: "Store and account created successfully" });
|
||||
} catch (err: any) {
|
||||
if (err.code === '23505') { // unique violation
|
||||
return errorResponse(409, "Store with this kode_toko already exists");
|
||||
}
|
||||
return errorResponse(500, err.message);
|
||||
}
|
||||
}
|
||||
@@ -1,60 +1,60 @@
|
||||
import { NextRequest, NextResponse } from "next/server";
|
||||
import { errorResponse } from "@/utils/api-error";
|
||||
import { getAccountFromAuthHeader } from "@/utils/auth";
|
||||
import { classifyAndMatchProduct, ClassifierError } from "@/utils/product-scan";
|
||||
|
||||
const corsHeaders = {
|
||||
"Access-Control-Allow-Origin": "*",
|
||||
"Access-Control-Allow-Methods": "GET, POST, PUT, DELETE, OPTIONS",
|
||||
"Access-Control-Allow-Headers": "Content-Type, Authorization"
|
||||
};
|
||||
|
||||
export async function OPTIONS() {
|
||||
return new NextResponse(null, { status: 204, headers: corsHeaders });
|
||||
}
|
||||
|
||||
export async function POST(req: NextRequest) {
|
||||
try {
|
||||
// Any authenticated account may scan - unlike sku_master writes, this is the
|
||||
// route the mobile app itself calls to do a product scan, not an admin tool.
|
||||
const account = getAccountFromAuthHeader(req.headers.get("authorization"));
|
||||
if (!account) {
|
||||
return errorResponse(401, "Unauthorized", { headers: corsHeaders });
|
||||
}
|
||||
|
||||
let imageBase64: string | null = null;
|
||||
const contentType = req.headers.get("content-type") || "";
|
||||
|
||||
if (contentType.includes("multipart/form-data")) {
|
||||
const formData = await req.formData();
|
||||
const file = (formData.get("image") || formData.get("file")) as Blob | null;
|
||||
if (!file) {
|
||||
return errorResponse(400, "Image is required", { headers: corsHeaders });
|
||||
}
|
||||
const buffer = Buffer.from(await file.arrayBuffer());
|
||||
imageBase64 = buffer.toString("base64");
|
||||
} else {
|
||||
const body = await req.json();
|
||||
imageBase64 = body.image_base64 || body.image || null;
|
||||
}
|
||||
|
||||
if (!imageBase64) {
|
||||
return errorResponse(400, "Image is required", { headers: corsHeaders });
|
||||
}
|
||||
|
||||
const result = await classifyAndMatchProduct(imageBase64);
|
||||
|
||||
return NextResponse.json({
|
||||
status: "success",
|
||||
data: result
|
||||
}, { headers: corsHeaders });
|
||||
|
||||
} catch (error: unknown) {
|
||||
console.error("Error in v1 scan-product API route:", error);
|
||||
if (error instanceof ClassifierError) {
|
||||
return errorResponse(error.status, error.message, { headers: corsHeaders });
|
||||
}
|
||||
const message = error instanceof Error ? error.message : "Internal server error";
|
||||
return errorResponse(500, message, { headers: corsHeaders });
|
||||
}
|
||||
}
|
||||
import { NextRequest, NextResponse } from "next/server";
|
||||
import { errorResponse } from "@/utils/api-error";
|
||||
import { getAccountFromAuthHeader } from "@/utils/auth";
|
||||
import { classifyAndMatchProduct, ClassifierError } from "@/utils/product-scan";
|
||||
|
||||
const corsHeaders = {
|
||||
"Access-Control-Allow-Origin": "*",
|
||||
"Access-Control-Allow-Methods": "GET, POST, PUT, DELETE, OPTIONS",
|
||||
"Access-Control-Allow-Headers": "Content-Type, Authorization"
|
||||
};
|
||||
|
||||
export async function OPTIONS() {
|
||||
return new NextResponse(null, { status: 204, headers: corsHeaders });
|
||||
}
|
||||
|
||||
export async function POST(req: NextRequest) {
|
||||
try {
|
||||
// Any authenticated account may scan - unlike sku_master writes, this is the
|
||||
// route the mobile app itself calls to do a product scan, not an admin tool.
|
||||
const account = getAccountFromAuthHeader(req.headers.get("authorization"));
|
||||
if (!account) {
|
||||
return errorResponse(401, "Unauthorized", { headers: corsHeaders });
|
||||
}
|
||||
|
||||
let imageBase64: string | null = null;
|
||||
const contentType = req.headers.get("content-type") || "";
|
||||
|
||||
if (contentType.includes("multipart/form-data")) {
|
||||
const formData = await req.formData();
|
||||
const file = (formData.get("image") || formData.get("file")) as Blob | null;
|
||||
if (!file) {
|
||||
return errorResponse(400, "Image is required", { headers: corsHeaders });
|
||||
}
|
||||
const buffer = Buffer.from(await file.arrayBuffer());
|
||||
imageBase64 = buffer.toString("base64");
|
||||
} else {
|
||||
const body = await req.json();
|
||||
imageBase64 = body.image_base64 || body.image || null;
|
||||
}
|
||||
|
||||
if (!imageBase64) {
|
||||
return errorResponse(400, "Image is required", { headers: corsHeaders });
|
||||
}
|
||||
|
||||
const result = await classifyAndMatchProduct(imageBase64);
|
||||
|
||||
return NextResponse.json({
|
||||
status: "success",
|
||||
data: result
|
||||
}, { headers: corsHeaders });
|
||||
|
||||
} catch (error: unknown) {
|
||||
console.error("Error in v1 scan-product API route:", error);
|
||||
if (error instanceof ClassifierError) {
|
||||
return errorResponse(error.status, error.message, { headers: corsHeaders });
|
||||
}
|
||||
const message = error instanceof Error ? error.message : "Internal server error";
|
||||
return errorResponse(500, message, { headers: corsHeaders });
|
||||
}
|
||||
}
|
||||
@@ -1,135 +1,135 @@
|
||||
import { NextRequest, NextResponse } from "next/server";
|
||||
import { logVllmCallToAll } from "../../../../utils/active-log";
|
||||
import { errorResponse } from "@/utils/api-error";
|
||||
|
||||
export const dynamic = "force-dynamic";
|
||||
|
||||
export async function POST(req: NextRequest) {
|
||||
return handleProxy(req);
|
||||
}
|
||||
|
||||
export async function GET(req: NextRequest) {
|
||||
return handleProxy(req);
|
||||
}
|
||||
|
||||
export async function PUT(req: NextRequest) {
|
||||
return handleProxy(req);
|
||||
}
|
||||
|
||||
export async function DELETE(req: NextRequest) {
|
||||
return handleProxy(req);
|
||||
}
|
||||
|
||||
async function handleProxy(req: NextRequest) {
|
||||
try {
|
||||
const pathname = req.nextUrl.pathname;
|
||||
// Extract everything after /api/vllm-proxy
|
||||
const relPath = pathname.replace(/^\/api\/vllm-proxy/, "");
|
||||
|
||||
// The real vLLM server is at paddleocr-vllm-server:8118 inside docker compose
|
||||
const realBaseUrl = process.env.VLLM_SERVER_REAL_URL || "http://paddleocr-vllm-server:8118";
|
||||
|
||||
// Construct the destination URL
|
||||
const destUrl = `${realBaseUrl}${relPath}${req.nextUrl.search}`;
|
||||
|
||||
console.log(`[vllm-proxy] Routing request from ${pathname} to ${destUrl}`);
|
||||
|
||||
// Read the request body if present
|
||||
let reqBody: any = null;
|
||||
let reqBodyBuffer: Buffer | null = null;
|
||||
|
||||
if (req.body) {
|
||||
const arrayBuffer = await req.arrayBuffer();
|
||||
reqBodyBuffer = Buffer.from(arrayBuffer);
|
||||
|
||||
const contentType = req.headers.get("content-type") || "";
|
||||
if (contentType.includes("application/json")) {
|
||||
try {
|
||||
reqBody = JSON.parse(reqBodyBuffer.toString("utf-8"));
|
||||
} catch (e) {
|
||||
console.warn("[vllm-proxy] Failed to parse request body as JSON:", e);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Reconstruct headers, filtering out headers that might cause issues (e.g. Host)
|
||||
const headers = new Headers();
|
||||
req.headers.forEach((value, key) => {
|
||||
if (key.toLowerCase() !== "host" && key.toLowerCase() !== "content-length") {
|
||||
headers.set(key, value);
|
||||
}
|
||||
});
|
||||
|
||||
// Make the actual call to the real vLLM server
|
||||
const forwardResponse = await fetch(destUrl, {
|
||||
method: req.method,
|
||||
headers: headers,
|
||||
body: reqBodyBuffer ? new Uint8Array(reqBodyBuffer) : null,
|
||||
// @ts-ignore
|
||||
duplex: "half"
|
||||
});
|
||||
|
||||
// Read the response content
|
||||
const resBodyBuffer = Buffer.from(await forwardResponse.arrayBuffer());
|
||||
let resBody: any = null;
|
||||
|
||||
const resContentType = forwardResponse.headers.get("content-type") || "";
|
||||
if (resContentType.includes("application/json")) {
|
||||
try {
|
||||
resBody = JSON.parse(resBodyBuffer.toString("utf-8"));
|
||||
} catch (e) {
|
||||
console.warn("[vllm-proxy] Failed to parse response body as JSON:", e);
|
||||
}
|
||||
} else {
|
||||
resBody = resBodyBuffer.toString("utf-8");
|
||||
}
|
||||
|
||||
// Log the interaction if it looks like a completion call
|
||||
if (pathname.includes("/chat/completions") || pathname.includes("/completions")) {
|
||||
// Make a clean copy of the request to log (hiding huge base64 images if they clutter logs)
|
||||
const cleanReq = sanitizeLogPayload(reqBody);
|
||||
logVllmCallToAll(cleanReq, resBody);
|
||||
}
|
||||
|
||||
// Return the response back to pipeline-api
|
||||
const responseHeaders = new Headers();
|
||||
forwardResponse.headers.forEach((value, key) => {
|
||||
responseHeaders.set(key, value);
|
||||
});
|
||||
|
||||
return new Response(resBodyBuffer, {
|
||||
status: forwardResponse.status,
|
||||
statusText: forwardResponse.statusText,
|
||||
headers: responseHeaders
|
||||
});
|
||||
|
||||
} catch (error) {
|
||||
console.error("[vllm-proxy] Error forwarding request:", error);
|
||||
return errorResponse(500, "Failed to proxy request to vLLM server");
|
||||
}
|
||||
}
|
||||
|
||||
// Helper function to keep log sizes reasonable by truncating huge base64 image strings
|
||||
function sanitizeLogPayload(payload: any): any {
|
||||
if (!payload) return payload;
|
||||
try {
|
||||
const copy = JSON.parse(JSON.stringify(payload));
|
||||
if (copy.messages && Array.isArray(copy.messages)) {
|
||||
for (const msg of copy.messages) {
|
||||
if (msg.content && Array.isArray(msg.content)) {
|
||||
for (const part of msg.content) {
|
||||
if (part.type === "image_url" && part.image_url && part.image_url.url) {
|
||||
const url = part.image_url.url;
|
||||
if (url.startsWith("data:") && url.length > 200) {
|
||||
part.image_url.url = url.substring(0, 100) + "...[TRUNCATED BASE64]..." + url.substring(url.length - 50);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
return copy;
|
||||
} catch (e) {
|
||||
return payload;
|
||||
}
|
||||
}
|
||||
import { NextRequest, NextResponse } from "next/server";
|
||||
import { logVllmCallToAll } from "../../../../utils/active-log";
|
||||
import { errorResponse } from "@/utils/api-error";
|
||||
|
||||
export const dynamic = "force-dynamic";
|
||||
|
||||
export async function POST(req: NextRequest) {
|
||||
return handleProxy(req);
|
||||
}
|
||||
|
||||
export async function GET(req: NextRequest) {
|
||||
return handleProxy(req);
|
||||
}
|
||||
|
||||
export async function PUT(req: NextRequest) {
|
||||
return handleProxy(req);
|
||||
}
|
||||
|
||||
export async function DELETE(req: NextRequest) {
|
||||
return handleProxy(req);
|
||||
}
|
||||
|
||||
async function handleProxy(req: NextRequest) {
|
||||
try {
|
||||
const pathname = req.nextUrl.pathname;
|
||||
// Extract everything after /api/vllm-proxy
|
||||
const relPath = pathname.replace(/^\/api\/vllm-proxy/, "");
|
||||
|
||||
// The real vLLM server is at paddleocr-vllm-server:8118 inside docker compose
|
||||
const realBaseUrl = process.env.VLLM_SERVER_REAL_URL || "http://paddleocr-vllm-server:8118";
|
||||
|
||||
// Construct the destination URL
|
||||
const destUrl = `${realBaseUrl}${relPath}${req.nextUrl.search}`;
|
||||
|
||||
console.log(`[vllm-proxy] Routing request from ${pathname} to ${destUrl}`);
|
||||
|
||||
// Read the request body if present
|
||||
let reqBody: any = null;
|
||||
let reqBodyBuffer: Buffer | null = null;
|
||||
|
||||
if (req.body) {
|
||||
const arrayBuffer = await req.arrayBuffer();
|
||||
reqBodyBuffer = Buffer.from(arrayBuffer);
|
||||
|
||||
const contentType = req.headers.get("content-type") || "";
|
||||
if (contentType.includes("application/json")) {
|
||||
try {
|
||||
reqBody = JSON.parse(reqBodyBuffer.toString("utf-8"));
|
||||
} catch (e) {
|
||||
console.warn("[vllm-proxy] Failed to parse request body as JSON:", e);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Reconstruct headers, filtering out headers that might cause issues (e.g. Host)
|
||||
const headers = new Headers();
|
||||
req.headers.forEach((value, key) => {
|
||||
if (key.toLowerCase() !== "host" && key.toLowerCase() !== "content-length") {
|
||||
headers.set(key, value);
|
||||
}
|
||||
});
|
||||
|
||||
// Make the actual call to the real vLLM server
|
||||
const forwardResponse = await fetch(destUrl, {
|
||||
method: req.method,
|
||||
headers: headers,
|
||||
body: reqBodyBuffer ? new Uint8Array(reqBodyBuffer) : null,
|
||||
// @ts-ignore
|
||||
duplex: "half"
|
||||
});
|
||||
|
||||
// Read the response content
|
||||
const resBodyBuffer = Buffer.from(await forwardResponse.arrayBuffer());
|
||||
let resBody: any = null;
|
||||
|
||||
const resContentType = forwardResponse.headers.get("content-type") || "";
|
||||
if (resContentType.includes("application/json")) {
|
||||
try {
|
||||
resBody = JSON.parse(resBodyBuffer.toString("utf-8"));
|
||||
} catch (e) {
|
||||
console.warn("[vllm-proxy] Failed to parse response body as JSON:", e);
|
||||
}
|
||||
} else {
|
||||
resBody = resBodyBuffer.toString("utf-8");
|
||||
}
|
||||
|
||||
// Log the interaction if it looks like a completion call
|
||||
if (pathname.includes("/chat/completions") || pathname.includes("/completions")) {
|
||||
// Make a clean copy of the request to log (hiding huge base64 images if they clutter logs)
|
||||
const cleanReq = sanitizeLogPayload(reqBody);
|
||||
logVllmCallToAll(cleanReq, resBody);
|
||||
}
|
||||
|
||||
// Return the response back to pipeline-api
|
||||
const responseHeaders = new Headers();
|
||||
forwardResponse.headers.forEach((value, key) => {
|
||||
responseHeaders.set(key, value);
|
||||
});
|
||||
|
||||
return new Response(resBodyBuffer, {
|
||||
status: forwardResponse.status,
|
||||
statusText: forwardResponse.statusText,
|
||||
headers: responseHeaders
|
||||
});
|
||||
|
||||
} catch (error) {
|
||||
console.error("[vllm-proxy] Error forwarding request:", error);
|
||||
return errorResponse(500, "Failed to proxy request to vLLM server");
|
||||
}
|
||||
}
|
||||
|
||||
// Helper function to keep log sizes reasonable by truncating huge base64 image strings
|
||||
function sanitizeLogPayload(payload: any): any {
|
||||
if (!payload) return payload;
|
||||
try {
|
||||
const copy = JSON.parse(JSON.stringify(payload));
|
||||
if (copy.messages && Array.isArray(copy.messages)) {
|
||||
for (const msg of copy.messages) {
|
||||
if (msg.content && Array.isArray(msg.content)) {
|
||||
for (const part of msg.content) {
|
||||
if (part.type === "image_url" && part.image_url && part.image_url.url) {
|
||||
const url = part.image_url.url;
|
||||
if (url.startsWith("data:") && url.length > 200) {
|
||||
part.image_url.url = url.substring(0, 100) + "...[TRUNCATED BASE64]..." + url.substring(url.length - 50);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
return copy;
|
||||
} catch (e) {
|
||||
return payload;
|
||||
}
|
||||
}
|
||||
@@ -1,20 +1,20 @@
|
||||
@import "tailwindcss";
|
||||
|
||||
:root {
|
||||
--background: #0f172a;
|
||||
--foreground: #f8fafc;
|
||||
}
|
||||
|
||||
@media (prefers-color-scheme: dark) {
|
||||
:root {
|
||||
--background: #0a0a0a;
|
||||
--foreground: #ededed;
|
||||
}
|
||||
}
|
||||
|
||||
body {
|
||||
background: var(--background);
|
||||
color: var(--foreground);
|
||||
font-family: system-ui, -apple-system, sans-serif;
|
||||
margin: 0;
|
||||
}
|
||||
@import "tailwindcss";
|
||||
|
||||
:root {
|
||||
--background: #0f172a;
|
||||
--foreground: #f8fafc;
|
||||
}
|
||||
|
||||
@media (prefers-color-scheme: dark) {
|
||||
:root {
|
||||
--background: #0a0a0a;
|
||||
--foreground: #ededed;
|
||||
}
|
||||
}
|
||||
|
||||
body {
|
||||
background: var(--background);
|
||||
color: var(--foreground);
|
||||
font-family: system-ui, -apple-system, sans-serif;
|
||||
margin: 0;
|
||||
}
|
||||
@@ -1,23 +1,23 @@
|
||||
import type { Metadata } from "next";
|
||||
import "./globals.css";
|
||||
|
||||
export const metadata: Metadata = {
|
||||
title: "AI OCR Delivery Order",
|
||||
description: "Generated by create next app",
|
||||
};
|
||||
|
||||
export default function RootLayout({
|
||||
children,
|
||||
}: Readonly<{
|
||||
children: React.ReactNode;
|
||||
}>) {
|
||||
return (
|
||||
<html
|
||||
lang="en"
|
||||
className="h-full antialiased text-slate-100 bg-slate-950"
|
||||
suppressHydrationWarning
|
||||
>
|
||||
<body className="min-h-full flex flex-col font-sans">{children}</body>
|
||||
</html>
|
||||
);
|
||||
}
|
||||
import type { Metadata } from "next";
|
||||
import "./globals.css";
|
||||
|
||||
export const metadata: Metadata = {
|
||||
title: "AI OCR Delivery Order",
|
||||
description: "Generated by create next app",
|
||||
};
|
||||
|
||||
export default function RootLayout({
|
||||
children,
|
||||
}: Readonly<{
|
||||
children: React.ReactNode;
|
||||
}>) {
|
||||
return (
|
||||
<html
|
||||
lang="en"
|
||||
className="h-full antialiased text-slate-100 bg-slate-950"
|
||||
suppressHydrationWarning
|
||||
>
|
||||
<body className="min-h-full flex flex-col font-sans">{children}</body>
|
||||
</html>
|
||||
);
|
||||
}
|
||||
@@ -1,253 +1,253 @@
|
||||
"use client";
|
||||
|
||||
import React, { useState, useEffect } from "react";
|
||||
import { Sidebar } from "@/components/manual-label-scan/Sidebar";
|
||||
import { Editor, ScanLabelFormData, AiPredictedData } from "@/components/manual-label-scan/Editor";
|
||||
import { ImageViewer } from "@/components/manual-label-scan/ImageViewer";
|
||||
import { getErrorMessage } from "@/utils/client-error";
|
||||
|
||||
export default function ManualLabelScanPage() {
|
||||
const [files, setFiles] = useState<{ url: string; filename: string }[]>([]);
|
||||
const [currentIndex, setCurrentIndex] = useState(-1);
|
||||
|
||||
const [formData, setFormData] = useState<ScanLabelFormData>({
|
||||
filename: "",
|
||||
no_sku: "",
|
||||
nama_item: "",
|
||||
expiry_date: "",
|
||||
notes: ""
|
||||
});
|
||||
const [aiPredicted, setAiPredicted] = useState<AiPredictedData | null>(null);
|
||||
const [aiSource, setAiSource] = useState<{ type: "batch" | "live"; timestamp: string; method?: string; confidence?: number } | null>(null);
|
||||
|
||||
const [skuList, setSkuList] = useState<Array<{ no_sku: string; nama_item: string }>>([]);
|
||||
const [isScanning, setIsScanning] = useState(false);
|
||||
const [savingGT, setSavingGT] = useState(false);
|
||||
|
||||
const [toast, setToast] = useState({ message: "", show: false, isError: false });
|
||||
const showToast = (message: string, isError = false) => {
|
||||
setToast({ message, show: true, isError });
|
||||
setTimeout(() => setToast(p => ({ ...p, show: false })), 2500);
|
||||
};
|
||||
|
||||
useEffect(() => {
|
||||
const fetchAllData = async () => {
|
||||
try {
|
||||
// Fetch Skus
|
||||
const skuRes = await fetch("/api/skus");
|
||||
if (skuRes.ok) {
|
||||
const skuData = await skuRes.json();
|
||||
setSkuList(skuData.skus || []);
|
||||
}
|
||||
|
||||
// Fetch Test Images — the frozen 79-image Validation Set
|
||||
// (product-test-images-fixed/), the only set the accuracy harness
|
||||
// scores. Gallery/training photos (foto-kemasan-v2/) are not shown
|
||||
// here: they don't need per-photo ground truth, only correct
|
||||
// SKU-folder placement for classifier training.
|
||||
const testRes = await fetch("/api/product-images");
|
||||
let testFiles: { url: string; filename: string }[] = [];
|
||||
if (testRes.ok) {
|
||||
const testData = await testRes.json();
|
||||
testFiles = (testData.files || []).map((f: string) => ({
|
||||
url: `/api/product-images?filename=${encodeURIComponent(f)}`,
|
||||
filename: f
|
||||
}));
|
||||
}
|
||||
|
||||
setFiles(testFiles);
|
||||
if (testFiles.length > 0) setCurrentIndex(0);
|
||||
|
||||
} catch (err) {
|
||||
console.error("Error initializing page", err);
|
||||
showToast("Error loading dataset files", true);
|
||||
}
|
||||
};
|
||||
fetchAllData();
|
||||
}, []);
|
||||
|
||||
useEffect(() => {
|
||||
if (currentIndex < 0 || currentIndex >= files.length) return;
|
||||
const file = files[currentIndex];
|
||||
|
||||
const loadLabel = async () => {
|
||||
try {
|
||||
const res = await fetch(`/api/manual-label-scan?filename=${encodeURIComponent(file.filename)}`);
|
||||
if (res.ok) {
|
||||
const data = await res.json();
|
||||
setFormData({
|
||||
filename: data.filename || file.filename,
|
||||
no_sku: data.no_sku || "",
|
||||
nama_item: data.nama_item || "",
|
||||
expiry_date: data.expiry_date || "",
|
||||
notes: data.notes || ""
|
||||
});
|
||||
}
|
||||
} catch (err) {
|
||||
console.error("Error fetching label", err);
|
||||
}
|
||||
|
||||
// Default-load the AI prediction from the last batch accuracy run
|
||||
// (not a live re-scan) so failures are visible immediately while
|
||||
// browsing - "Scan with AI" below can still be used to get a fresh
|
||||
// live result for this exact image.
|
||||
setAiPredicted(null);
|
||||
setAiSource(null);
|
||||
try {
|
||||
const aiRes = await fetch(`/api/product-scan-results?filename=${encodeURIComponent(file.filename)}`);
|
||||
if (aiRes.ok) {
|
||||
const aiData = await aiRes.json();
|
||||
if (aiData.found) {
|
||||
setAiPredicted({
|
||||
no_sku: aiData.no_sku,
|
||||
nama_item: aiData.nama_item,
|
||||
expiry_date: aiData.expiry_date
|
||||
});
|
||||
setAiSource({ type: "batch", timestamp: aiData.timestamp, method: aiData.method, confidence: aiData.confidence });
|
||||
}
|
||||
}
|
||||
} catch (err) {
|
||||
console.error("Error fetching batch AI result", err);
|
||||
}
|
||||
};
|
||||
loadLabel();
|
||||
}, [currentIndex, files]);
|
||||
|
||||
// Handle Ctrl+S keyboard shortcut
|
||||
useEffect(() => {
|
||||
const handleKeyDown = (e: KeyboardEvent) => {
|
||||
if ((e.ctrlKey || e.metaKey) && e.key === "s") {
|
||||
e.preventDefault();
|
||||
handleSave();
|
||||
}
|
||||
};
|
||||
window.addEventListener("keydown", handleKeyDown);
|
||||
return () => window.removeEventListener("keydown", handleKeyDown);
|
||||
}, [formData]);
|
||||
|
||||
const handleChange = (field: keyof ScanLabelFormData, value: string) => {
|
||||
setFormData(prev => ({ ...prev, [field]: value }));
|
||||
};
|
||||
|
||||
const handleScanWithAi = async () => {
|
||||
if (currentIndex < 0) return;
|
||||
const currentFile = files[currentIndex];
|
||||
setIsScanning(true);
|
||||
try {
|
||||
// Fetch image as base64
|
||||
const imgRes = await fetch(currentFile.url);
|
||||
const blob = await imgRes.blob();
|
||||
const base64 = await new Promise<string>((resolve) => {
|
||||
const reader = new FileReader();
|
||||
reader.onloadend = () => resolve(reader.result as string);
|
||||
reader.readAsDataURL(blob);
|
||||
});
|
||||
|
||||
const scanRes = await fetch("/api/scan-pfm", {
|
||||
method: "POST",
|
||||
headers: { "Content-Type": "application/json" },
|
||||
body: JSON.stringify({ image: base64 })
|
||||
});
|
||||
|
||||
if (!scanRes.ok) throw new Error("Pipeline API error");
|
||||
const scanData = await scanRes.json();
|
||||
|
||||
// Compare against the sku_master-resolved best match (what the app
|
||||
// actually shows/saves as nama_item, and what the accuracy harness
|
||||
// scores), not classification.top1_name - that's the classifier's raw
|
||||
// internal class label (e.g. the foto-kemasan-v2 folder name), which
|
||||
// structurally never matches a sku_master-style ground truth string
|
||||
// even when the classification itself is correct.
|
||||
const bestMatch = (scanData.possibleMatches || []).find((m: { isBestMatch?: boolean }) => m.isBestMatch);
|
||||
|
||||
setAiPredicted({
|
||||
no_sku: bestMatch?.no_sku,
|
||||
nama_item: bestMatch?.nama_item,
|
||||
expiry_date: scanData.ocr?.extracted_expired_date
|
||||
});
|
||||
setAiSource({
|
||||
type: "live",
|
||||
timestamp: new Date().toISOString(),
|
||||
method: scanData.classification?.method,
|
||||
confidence: scanData.classification?.top1_confidence
|
||||
});
|
||||
|
||||
showToast("AI Scan complete!");
|
||||
} catch (err) {
|
||||
showToast(getErrorMessage(err, undefined, "AI Scan failed"), true);
|
||||
} finally {
|
||||
setIsScanning(false);
|
||||
}
|
||||
};
|
||||
|
||||
const handleSave = async () => {
|
||||
setSavingGT(true);
|
||||
try {
|
||||
const res = await fetch("/api/manual-label-scan", {
|
||||
method: "POST",
|
||||
headers: { "Content-Type": "application/json" },
|
||||
body: JSON.stringify(formData)
|
||||
});
|
||||
|
||||
if (res.ok) {
|
||||
showToast("Ground truth saved successfully!");
|
||||
if (currentIndex < files.length - 1) {
|
||||
setCurrentIndex(prev => prev + 1);
|
||||
} else {
|
||||
showToast("All images completed!");
|
||||
}
|
||||
} else {
|
||||
const errData = await res.json().catch(() => ({}));
|
||||
showToast(getErrorMessage(null, errData, "Save failed"), true);
|
||||
}
|
||||
} catch (err) {
|
||||
showToast(getErrorMessage(err, undefined, "Save failed"), true);
|
||||
} finally {
|
||||
setSavingGT(false);
|
||||
}
|
||||
};
|
||||
|
||||
return (
|
||||
<div className="h-screen w-screen flex flex-col bg-slate-950 font-sans overflow-hidden">
|
||||
{/* Header */}
|
||||
<header className="h-14 border-b border-slate-800 bg-slate-900/80 backdrop-blur-md flex items-center justify-between px-6 shrink-0 z-10">
|
||||
<div className="flex items-center gap-3">
|
||||
<div className="w-8 h-8 rounded-lg bg-gradient-to-tr from-teal-500 to-cyan-500 flex items-center justify-center font-bold text-white text-xs shadow-md">
|
||||
SP
|
||||
</div>
|
||||
<span className="text-sm font-bold text-slate-100">Product Scan Annotation</span>
|
||||
</div>
|
||||
<div className="text-xs">
|
||||
<a href="/scan-pfm" className="text-slate-400 hover:text-slate-100 transition">← Back to Scanner</a>
|
||||
</div>
|
||||
</header>
|
||||
|
||||
{/* Main Body */}
|
||||
<div className="flex-1 flex overflow-hidden">
|
||||
<Sidebar
|
||||
files={files.map(f => f.filename)}
|
||||
currentIndex={currentIndex}
|
||||
onSelect={setCurrentIndex}
|
||||
/>
|
||||
<ImageViewer src={currentIndex >= 0 ? files[currentIndex].url : null} />
|
||||
<Editor
|
||||
formData={formData}
|
||||
aiPredicted={aiPredicted}
|
||||
aiSource={aiSource}
|
||||
skuList={skuList}
|
||||
isScanning={isScanning}
|
||||
onScanWithAi={handleScanWithAi}
|
||||
onChange={handleChange}
|
||||
onSave={handleSave}
|
||||
savingGT={savingGT}
|
||||
/>
|
||||
</div>
|
||||
|
||||
{/* Toast */}
|
||||
<div className={`fixed bottom-6 left-1/2 -translate-x-1/2 px-5 py-3 rounded-lg flex items-center gap-2.5 shadow-2xl font-medium z-[100] transition duration-300 ${toast.show ? "translate-y-0 opacity-100 scale-100" : "translate-y-12 opacity-0 scale-95 pointer-events-none"} ${toast.isError ? "bg-rose-600 text-white" : "bg-emerald-600 text-white"}`}>
|
||||
<span>{toast.isError ? "❌" : "✅"}</span>
|
||||
<span className="text-sm">{toast.message}</span>
|
||||
</div>
|
||||
</div>
|
||||
);
|
||||
}
|
||||
"use client";
|
||||
|
||||
import React, { useState, useEffect } from "react";
|
||||
import { Sidebar } from "@/components/manual-label-scan/Sidebar";
|
||||
import { Editor, ScanLabelFormData, AiPredictedData } from "@/components/manual-label-scan/Editor";
|
||||
import { ImageViewer } from "@/components/manual-label-scan/ImageViewer";
|
||||
import { getErrorMessage } from "@/utils/client-error";
|
||||
|
||||
export default function ManualLabelScanPage() {
|
||||
const [files, setFiles] = useState<{ url: string; filename: string }[]>([]);
|
||||
const [currentIndex, setCurrentIndex] = useState(-1);
|
||||
|
||||
const [formData, setFormData] = useState<ScanLabelFormData>({
|
||||
filename: "",
|
||||
no_sku: "",
|
||||
nama_item: "",
|
||||
expiry_date: "",
|
||||
notes: ""
|
||||
});
|
||||
const [aiPredicted, setAiPredicted] = useState<AiPredictedData | null>(null);
|
||||
const [aiSource, setAiSource] = useState<{ type: "batch" | "live"; timestamp: string; method?: string; confidence?: number } | null>(null);
|
||||
|
||||
const [skuList, setSkuList] = useState<Array<{ no_sku: string; nama_item: string }>>([]);
|
||||
const [isScanning, setIsScanning] = useState(false);
|
||||
const [savingGT, setSavingGT] = useState(false);
|
||||
|
||||
const [toast, setToast] = useState({ message: "", show: false, isError: false });
|
||||
const showToast = (message: string, isError = false) => {
|
||||
setToast({ message, show: true, isError });
|
||||
setTimeout(() => setToast(p => ({ ...p, show: false })), 2500);
|
||||
};
|
||||
|
||||
useEffect(() => {
|
||||
const fetchAllData = async () => {
|
||||
try {
|
||||
// Fetch Skus
|
||||
const skuRes = await fetch("/api/skus");
|
||||
if (skuRes.ok) {
|
||||
const skuData = await skuRes.json();
|
||||
setSkuList(skuData.skus || []);
|
||||
}
|
||||
|
||||
// Fetch Test Images — the frozen 79-image Validation Set
|
||||
// (product-test-images-fixed/), the only set the accuracy harness
|
||||
// scores. Gallery/training photos (foto-kemasan-v2/) are not shown
|
||||
// here: they don't need per-photo ground truth, only correct
|
||||
// SKU-folder placement for classifier training.
|
||||
const testRes = await fetch("/api/product-images");
|
||||
let testFiles: { url: string; filename: string }[] = [];
|
||||
if (testRes.ok) {
|
||||
const testData = await testRes.json();
|
||||
testFiles = (testData.files || []).map((f: string) => ({
|
||||
url: `/api/product-images?filename=${encodeURIComponent(f)}`,
|
||||
filename: f
|
||||
}));
|
||||
}
|
||||
|
||||
setFiles(testFiles);
|
||||
if (testFiles.length > 0) setCurrentIndex(0);
|
||||
|
||||
} catch (err) {
|
||||
console.error("Error initializing page", err);
|
||||
showToast("Error loading dataset files", true);
|
||||
}
|
||||
};
|
||||
fetchAllData();
|
||||
}, []);
|
||||
|
||||
useEffect(() => {
|
||||
if (currentIndex < 0 || currentIndex >= files.length) return;
|
||||
const file = files[currentIndex];
|
||||
|
||||
const loadLabel = async () => {
|
||||
try {
|
||||
const res = await fetch(`/api/manual-label-scan?filename=${encodeURIComponent(file.filename)}`);
|
||||
if (res.ok) {
|
||||
const data = await res.json();
|
||||
setFormData({
|
||||
filename: data.filename || file.filename,
|
||||
no_sku: data.no_sku || "",
|
||||
nama_item: data.nama_item || "",
|
||||
expiry_date: data.expiry_date || "",
|
||||
notes: data.notes || ""
|
||||
});
|
||||
}
|
||||
} catch (err) {
|
||||
console.error("Error fetching label", err);
|
||||
}
|
||||
|
||||
// Default-load the AI prediction from the last batch accuracy run
|
||||
// (not a live re-scan) so failures are visible immediately while
|
||||
// browsing - "Scan with AI" below can still be used to get a fresh
|
||||
// live result for this exact image.
|
||||
setAiPredicted(null);
|
||||
setAiSource(null);
|
||||
try {
|
||||
const aiRes = await fetch(`/api/product-scan-results?filename=${encodeURIComponent(file.filename)}`);
|
||||
if (aiRes.ok) {
|
||||
const aiData = await aiRes.json();
|
||||
if (aiData.found) {
|
||||
setAiPredicted({
|
||||
no_sku: aiData.no_sku,
|
||||
nama_item: aiData.nama_item,
|
||||
expiry_date: aiData.expiry_date
|
||||
});
|
||||
setAiSource({ type: "batch", timestamp: aiData.timestamp, method: aiData.method, confidence: aiData.confidence });
|
||||
}
|
||||
}
|
||||
} catch (err) {
|
||||
console.error("Error fetching batch AI result", err);
|
||||
}
|
||||
};
|
||||
loadLabel();
|
||||
}, [currentIndex, files]);
|
||||
|
||||
// Handle Ctrl+S keyboard shortcut
|
||||
useEffect(() => {
|
||||
const handleKeyDown = (e: KeyboardEvent) => {
|
||||
if ((e.ctrlKey || e.metaKey) && e.key === "s") {
|
||||
e.preventDefault();
|
||||
handleSave();
|
||||
}
|
||||
};
|
||||
window.addEventListener("keydown", handleKeyDown);
|
||||
return () => window.removeEventListener("keydown", handleKeyDown);
|
||||
}, [formData]);
|
||||
|
||||
const handleChange = (field: keyof ScanLabelFormData, value: string) => {
|
||||
setFormData(prev => ({ ...prev, [field]: value }));
|
||||
};
|
||||
|
||||
const handleScanWithAi = async () => {
|
||||
if (currentIndex < 0) return;
|
||||
const currentFile = files[currentIndex];
|
||||
setIsScanning(true);
|
||||
try {
|
||||
// Fetch image as base64
|
||||
const imgRes = await fetch(currentFile.url);
|
||||
const blob = await imgRes.blob();
|
||||
const base64 = await new Promise<string>((resolve) => {
|
||||
const reader = new FileReader();
|
||||
reader.onloadend = () => resolve(reader.result as string);
|
||||
reader.readAsDataURL(blob);
|
||||
});
|
||||
|
||||
const scanRes = await fetch("/api/scan-pfm", {
|
||||
method: "POST",
|
||||
headers: { "Content-Type": "application/json" },
|
||||
body: JSON.stringify({ image: base64 })
|
||||
});
|
||||
|
||||
if (!scanRes.ok) throw new Error("Pipeline API error");
|
||||
const scanData = await scanRes.json();
|
||||
|
||||
// Compare against the sku_master-resolved best match (what the app
|
||||
// actually shows/saves as nama_item, and what the accuracy harness
|
||||
// scores), not classification.top1_name - that's the classifier's raw
|
||||
// internal class label (e.g. the foto-kemasan-v2 folder name), which
|
||||
// structurally never matches a sku_master-style ground truth string
|
||||
// even when the classification itself is correct.
|
||||
const bestMatch = (scanData.possibleMatches || []).find((m: { isBestMatch?: boolean }) => m.isBestMatch);
|
||||
|
||||
setAiPredicted({
|
||||
no_sku: bestMatch?.no_sku,
|
||||
nama_item: bestMatch?.nama_item,
|
||||
expiry_date: scanData.ocr?.extracted_expired_date
|
||||
});
|
||||
setAiSource({
|
||||
type: "live",
|
||||
timestamp: new Date().toISOString(),
|
||||
method: scanData.classification?.method,
|
||||
confidence: scanData.classification?.top1_confidence
|
||||
});
|
||||
|
||||
showToast("AI Scan complete!");
|
||||
} catch (err) {
|
||||
showToast(getErrorMessage(err, undefined, "AI Scan failed"), true);
|
||||
} finally {
|
||||
setIsScanning(false);
|
||||
}
|
||||
};
|
||||
|
||||
const handleSave = async () => {
|
||||
setSavingGT(true);
|
||||
try {
|
||||
const res = await fetch("/api/manual-label-scan", {
|
||||
method: "POST",
|
||||
headers: { "Content-Type": "application/json" },
|
||||
body: JSON.stringify(formData)
|
||||
});
|
||||
|
||||
if (res.ok) {
|
||||
showToast("Ground truth saved successfully!");
|
||||
if (currentIndex < files.length - 1) {
|
||||
setCurrentIndex(prev => prev + 1);
|
||||
} else {
|
||||
showToast("All images completed!");
|
||||
}
|
||||
} else {
|
||||
const errData = await res.json().catch(() => ({}));
|
||||
showToast(getErrorMessage(null, errData, "Save failed"), true);
|
||||
}
|
||||
} catch (err) {
|
||||
showToast(getErrorMessage(err, undefined, "Save failed"), true);
|
||||
} finally {
|
||||
setSavingGT(false);
|
||||
}
|
||||
};
|
||||
|
||||
return (
|
||||
<div className="h-screen w-screen flex flex-col bg-slate-950 font-sans overflow-hidden">
|
||||
{/* Header */}
|
||||
<header className="h-14 border-b border-slate-800 bg-slate-900/80 backdrop-blur-md flex items-center justify-between px-6 shrink-0 z-10">
|
||||
<div className="flex items-center gap-3">
|
||||
<div className="w-8 h-8 rounded-lg bg-gradient-to-tr from-teal-500 to-cyan-500 flex items-center justify-center font-bold text-white text-xs shadow-md">
|
||||
SP
|
||||
</div>
|
||||
<span className="text-sm font-bold text-slate-100">Product Scan Annotation</span>
|
||||
</div>
|
||||
<div className="text-xs">
|
||||
<a href="/scan-pfm" className="text-slate-400 hover:text-slate-100 transition">← Back to Scanner</a>
|
||||
</div>
|
||||
</header>
|
||||
|
||||
{/* Main Body */}
|
||||
<div className="flex-1 flex overflow-hidden">
|
||||
<Sidebar
|
||||
files={files.map(f => f.filename)}
|
||||
currentIndex={currentIndex}
|
||||
onSelect={setCurrentIndex}
|
||||
/>
|
||||
<ImageViewer src={currentIndex >= 0 ? files[currentIndex].url : null} />
|
||||
<Editor
|
||||
formData={formData}
|
||||
aiPredicted={aiPredicted}
|
||||
aiSource={aiSource}
|
||||
skuList={skuList}
|
||||
isScanning={isScanning}
|
||||
onScanWithAi={handleScanWithAi}
|
||||
onChange={handleChange}
|
||||
onSave={handleSave}
|
||||
savingGT={savingGT}
|
||||
/>
|
||||
</div>
|
||||
|
||||
{/* Toast */}
|
||||
<div className={`fixed bottom-6 left-1/2 -translate-x-1/2 px-5 py-3 rounded-lg flex items-center gap-2.5 shadow-2xl font-medium z-[100] transition duration-300 ${toast.show ? "translate-y-0 opacity-100 scale-100" : "translate-y-12 opacity-0 scale-95 pointer-events-none"} ${toast.isError ? "bg-rose-600 text-white" : "bg-emerald-600 text-white"}`}>
|
||||
<span>{toast.isError ? "❌" : "✅"}</span>
|
||||
<span className="text-sm">{toast.message}</span>
|
||||
</div>
|
||||
</div>
|
||||
);
|
||||
}
|
||||
File diff suppressed because it is too large.
Load diff
File diff suppressed because it is too large.
Load diff
File diff suppressed because it is too large.
Load diff
@@ -1,169 +1,169 @@
|
||||
import React from "react";
|
||||
|
||||
export interface ScanLabelFormData {
|
||||
filename: string;
|
||||
no_sku: string;
|
||||
nama_item: string;
|
||||
expiry_date: string;
|
||||
notes: string;
|
||||
}
|
||||
|
||||
export interface AiPredictedData {
|
||||
no_sku?: string;
|
||||
nama_item?: string;
|
||||
expiry_date?: string;
|
||||
}
|
||||
|
||||
export interface AiSourceInfo {
|
||||
type: "batch" | "live";
|
||||
timestamp: string;
|
||||
method?: string;
|
||||
confidence?: number;
|
||||
}
|
||||
|
||||
interface EditorProps {
|
||||
formData: ScanLabelFormData;
|
||||
aiPredicted: AiPredictedData | null;
|
||||
aiSource: AiSourceInfo | null;
|
||||
skuList: Array<{ no_sku: string; nama_item: string }>;
|
||||
isScanning: boolean;
|
||||
onScanWithAi: () => void;
|
||||
onChange: (field: keyof ScanLabelFormData, value: string) => void;
|
||||
onSave: () => void;
|
||||
savingGT: boolean;
|
||||
}
|
||||
|
||||
export function Editor({
|
||||
formData,
|
||||
aiPredicted,
|
||||
aiSource,
|
||||
skuList,
|
||||
isScanning,
|
||||
onScanWithAi,
|
||||
onChange,
|
||||
onSave,
|
||||
savingGT
|
||||
}: EditorProps) {
|
||||
// Autofill item name based on SKU if available
|
||||
const handleSkuChange = (value: string) => {
|
||||
onChange("no_sku", value);
|
||||
const matched = skuList.find(s => s.no_sku === value);
|
||||
if (matched && !formData.nama_item) {
|
||||
onChange("nama_item", matched.nama_item);
|
||||
}
|
||||
};
|
||||
|
||||
const AiNote = ({ current, aiValue }: { current: string; aiValue: string | undefined }) => {
|
||||
if (aiValue === undefined) return null;
|
||||
const differs = (current || "").trim() !== (aiValue || "").trim();
|
||||
return (
|
||||
<div className={`text-[10px] leading-tight mt-1 ${differs ? "text-amber-500" : "text-slate-500"}`}>
|
||||
AI: {aiValue || "(not detected)"}
|
||||
</div>
|
||||
);
|
||||
};
|
||||
|
||||
return (
|
||||
<div className="w-[400px] border-l border-slate-800 bg-slate-900/50 flex flex-col h-full shrink-0 overflow-y-auto">
|
||||
<div className="p-5 space-y-6">
|
||||
|
||||
{/* Header & AI Action */}
|
||||
<div className="flex flex-col gap-3 pb-4 border-b border-slate-800">
|
||||
<div>
|
||||
<h3 className="text-sm font-bold text-slate-100">Ground Truth Editor</h3>
|
||||
<p className="text-[11px] text-slate-500 truncate mt-0.5">{formData.filename || "No file selected"}</p>
|
||||
</div>
|
||||
<button
|
||||
onClick={onScanWithAi}
|
||||
disabled={isScanning || !formData.filename}
|
||||
className="w-full bg-teal-600/20 text-teal-400 hover:bg-teal-600/30 disabled:opacity-50 border border-teal-500/30 rounded-lg py-2 text-xs font-semibold transition flex items-center justify-center gap-2"
|
||||
>
|
||||
{isScanning ? "Scanning with Pipeline..." : "Scan with AI 🤖 (re-run live)"}
|
||||
</button>
|
||||
{aiSource ? (
|
||||
<p className="text-[10px] text-slate-500 leading-snug">
|
||||
{aiSource.type === "batch" ? (
|
||||
<>Showing result from last batch test ({new Date(aiSource.timestamp).toLocaleString()})</>
|
||||
) : (
|
||||
<>Live scan result ({new Date(aiSource.timestamp).toLocaleTimeString()})</>
|
||||
)}
|
||||
{aiSource.method && <> · {aiSource.method}</>}
|
||||
{typeof aiSource.confidence === "number" && <> · conf {aiSource.confidence.toFixed(3)}</>}
|
||||
</p>
|
||||
) : (
|
||||
<p className="text-[10px] text-slate-600 italic">No AI result yet for this image — click "Scan with AI" or run the accuracy batch test.</p>
|
||||
)}
|
||||
</div>
|
||||
|
||||
{/* Form Fields */}
|
||||
<div className="space-y-4">
|
||||
<div className="flex flex-col gap-1.5">
|
||||
<label className="text-xs font-semibold text-slate-400 uppercase tracking-wider">SKU</label>
|
||||
<input
|
||||
type="text"
|
||||
list="skuOptions"
|
||||
value={formData.no_sku}
|
||||
onChange={(e) => handleSkuChange(e.target.value)}
|
||||
placeholder="e.g. 12010119"
|
||||
className="bg-slate-950 border border-slate-800 text-slate-100 rounded-lg px-3 py-2 text-sm focus:outline-none focus:border-emerald-500 transition-colors"
|
||||
/>
|
||||
<datalist id="skuOptions">
|
||||
{skuList.map((s) => (
|
||||
<option key={s.no_sku} value={s.no_sku}>
|
||||
{s.nama_item}
|
||||
</option>
|
||||
))}
|
||||
</datalist>
|
||||
<AiNote current={formData.no_sku} aiValue={aiPredicted?.no_sku} />
|
||||
</div>
|
||||
|
||||
<div className="flex flex-col gap-1.5">
|
||||
<label className="text-xs font-semibold text-slate-400 uppercase tracking-wider">Product Name</label>
|
||||
<input
|
||||
type="text"
|
||||
value={formData.nama_item}
|
||||
onChange={(e) => onChange("nama_item", e.target.value)}
|
||||
placeholder="e.g. FIESTA NUGGET 400 GR"
|
||||
className="bg-slate-950 border border-slate-800 text-slate-100 rounded-lg px-3 py-2 text-sm focus:outline-none focus:border-emerald-500 transition-colors"
|
||||
/>
|
||||
<AiNote current={formData.nama_item} aiValue={aiPredicted?.nama_item} />
|
||||
</div>
|
||||
|
||||
<div className="flex flex-col gap-1.5">
|
||||
<label className="text-xs font-semibold text-slate-400 uppercase tracking-wider">Expiry Date</label>
|
||||
<input
|
||||
type="text"
|
||||
value={formData.expiry_date}
|
||||
onChange={(e) => onChange("expiry_date", e.target.value)}
|
||||
placeholder="e.g. 05/11/2026"
|
||||
className="bg-slate-950 border border-slate-800 text-slate-100 rounded-lg px-3 py-2 text-sm focus:outline-none focus:border-emerald-500 transition-colors"
|
||||
/>
|
||||
<AiNote current={formData.expiry_date} aiValue={aiPredicted?.expiry_date} />
|
||||
</div>
|
||||
|
||||
<div className="flex flex-col gap-1.5">
|
||||
<label className="text-xs font-semibold text-slate-400 uppercase tracking-wider">Notes</label>
|
||||
<textarea
|
||||
value={formData.notes}
|
||||
onChange={(e) => onChange("notes", e.target.value)}
|
||||
placeholder="Optional notes..."
|
||||
rows={3}
|
||||
className="bg-slate-950 border border-slate-800 text-slate-100 rounded-lg px-3 py-2 text-sm focus:outline-none focus:border-emerald-500 transition-colors resize-none"
|
||||
/>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
</div>
|
||||
|
||||
<div className="mt-auto p-5 border-t border-slate-800 bg-slate-900">
|
||||
<button
|
||||
onClick={onSave}
|
||||
disabled={savingGT || !formData.filename}
|
||||
className="w-full bg-emerald-600 hover:bg-emerald-500 text-white rounded-xl py-3 text-sm font-bold shadow-lg shadow-emerald-500/20 disabled:opacity-50 transition-all flex items-center justify-center"
|
||||
>
|
||||
{savingGT ? "Saving..." : "Save Ground Truth"}
|
||||
</button>
|
||||
</div>
|
||||
</div>
|
||||
);
|
||||
}
|
||||
import React from "react";
|
||||
|
||||
export interface ScanLabelFormData {
|
||||
filename: string;
|
||||
no_sku: string;
|
||||
nama_item: string;
|
||||
expiry_date: string;
|
||||
notes: string;
|
||||
}
|
||||
|
||||
export interface AiPredictedData {
|
||||
no_sku?: string;
|
||||
nama_item?: string;
|
||||
expiry_date?: string;
|
||||
}
|
||||
|
||||
export interface AiSourceInfo {
|
||||
type: "batch" | "live";
|
||||
timestamp: string;
|
||||
method?: string;
|
||||
confidence?: number;
|
||||
}
|
||||
|
||||
interface EditorProps {
|
||||
formData: ScanLabelFormData;
|
||||
aiPredicted: AiPredictedData | null;
|
||||
aiSource: AiSourceInfo | null;
|
||||
skuList: Array<{ no_sku: string; nama_item: string }>;
|
||||
isScanning: boolean;
|
||||
onScanWithAi: () => void;
|
||||
onChange: (field: keyof ScanLabelFormData, value: string) => void;
|
||||
onSave: () => void;
|
||||
savingGT: boolean;
|
||||
}
|
||||
|
||||
export function Editor({
|
||||
formData,
|
||||
aiPredicted,
|
||||
aiSource,
|
||||
skuList,
|
||||
isScanning,
|
||||
onScanWithAi,
|
||||
onChange,
|
||||
onSave,
|
||||
savingGT
|
||||
}: EditorProps) {
|
||||
// Autofill item name based on SKU if available
|
||||
const handleSkuChange = (value: string) => {
|
||||
onChange("no_sku", value);
|
||||
const matched = skuList.find(s => s.no_sku === value);
|
||||
if (matched && !formData.nama_item) {
|
||||
onChange("nama_item", matched.nama_item);
|
||||
}
|
||||
};
|
||||
|
||||
const AiNote = ({ current, aiValue }: { current: string; aiValue: string | undefined }) => {
|
||||
if (aiValue === undefined) return null;
|
||||
const differs = (current || "").trim() !== (aiValue || "").trim();
|
||||
return (
|
||||
<div className={`text-[10px] leading-tight mt-1 ${differs ? "text-amber-500" : "text-slate-500"}`}>
|
||||
AI: {aiValue || "(not detected)"}
|
||||
</div>
|
||||
);
|
||||
};
|
||||
|
||||
return (
|
||||
<div className="w-[400px] border-l border-slate-800 bg-slate-900/50 flex flex-col h-full shrink-0 overflow-y-auto">
|
||||
<div className="p-5 space-y-6">
|
||||
|
||||
{/* Header & AI Action */}
|
||||
<div className="flex flex-col gap-3 pb-4 border-b border-slate-800">
|
||||
<div>
|
||||
<h3 className="text-sm font-bold text-slate-100">Ground Truth Editor</h3>
|
||||
<p className="text-[11px] text-slate-500 truncate mt-0.5">{formData.filename || "No file selected"}</p>
|
||||
</div>
|
||||
<button
|
||||
onClick={onScanWithAi}
|
||||
disabled={isScanning || !formData.filename}
|
||||
className="w-full bg-teal-600/20 text-teal-400 hover:bg-teal-600/30 disabled:opacity-50 border border-teal-500/30 rounded-lg py-2 text-xs font-semibold transition flex items-center justify-center gap-2"
|
||||
>
|
||||
{isScanning ? "Scanning with Pipeline..." : "Scan with AI 🤖 (re-run live)"}
|
||||
</button>
|
||||
{aiSource ? (
|
||||
<p className="text-[10px] text-slate-500 leading-snug">
|
||||
{aiSource.type === "batch" ? (
|
||||
<>Showing result from last batch test ({new Date(aiSource.timestamp).toLocaleString()})</>
|
||||
) : (
|
||||
<>Live scan result ({new Date(aiSource.timestamp).toLocaleTimeString()})</>
|
||||
)}
|
||||
{aiSource.method && <> · {aiSource.method}</>}
|
||||
{typeof aiSource.confidence === "number" && <> · conf {aiSource.confidence.toFixed(3)}</>}
|
||||
</p>
|
||||
) : (
|
||||
<p className="text-[10px] text-slate-600 italic">No AI result yet for this image — click "Scan with AI" or run the accuracy batch test.</p>
|
||||
)}
|
||||
</div>
|
||||
|
||||
{/* Form Fields */}
|
||||
<div className="space-y-4">
|
||||
<div className="flex flex-col gap-1.5">
|
||||
<label className="text-xs font-semibold text-slate-400 uppercase tracking-wider">SKU</label>
|
||||
<input
|
||||
type="text"
|
||||
list="skuOptions"
|
||||
value={formData.no_sku}
|
||||
onChange={(e) => handleSkuChange(e.target.value)}
|
||||
placeholder="e.g. 12010119"
|
||||
className="bg-slate-950 border border-slate-800 text-slate-100 rounded-lg px-3 py-2 text-sm focus:outline-none focus:border-emerald-500 transition-colors"
|
||||
/>
|
||||
<datalist id="skuOptions">
|
||||
{skuList.map((s) => (
|
||||
<option key={s.no_sku} value={s.no_sku}>
|
||||
{s.nama_item}
|
||||
</option>
|
||||
))}
|
||||
</datalist>
|
||||
<AiNote current={formData.no_sku} aiValue={aiPredicted?.no_sku} />
|
||||
</div>
|
||||
|
||||
<div className="flex flex-col gap-1.5">
|
||||
<label className="text-xs font-semibold text-slate-400 uppercase tracking-wider">Product Name</label>
|
||||
<input
|
||||
type="text"
|
||||
value={formData.nama_item}
|
||||
onChange={(e) => onChange("nama_item", e.target.value)}
|
||||
placeholder="e.g. FIESTA NUGGET 400 GR"
|
||||
className="bg-slate-950 border border-slate-800 text-slate-100 rounded-lg px-3 py-2 text-sm focus:outline-none focus:border-emerald-500 transition-colors"
|
||||
/>
|
||||
<AiNote current={formData.nama_item} aiValue={aiPredicted?.nama_item} />
|
||||
</div>
|
||||
|
||||
<div className="flex flex-col gap-1.5">
|
||||
<label className="text-xs font-semibold text-slate-400 uppercase tracking-wider">Expiry Date</label>
|
||||
<input
|
||||
type="text"
|
||||
value={formData.expiry_date}
|
||||
onChange={(e) => onChange("expiry_date", e.target.value)}
|
||||
placeholder="e.g. 05/11/2026"
|
||||
className="bg-slate-950 border border-slate-800 text-slate-100 rounded-lg px-3 py-2 text-sm focus:outline-none focus:border-emerald-500 transition-colors"
|
||||
/>
|
||||
<AiNote current={formData.expiry_date} aiValue={aiPredicted?.expiry_date} />
|
||||
</div>
|
||||
|
||||
<div className="flex flex-col gap-1.5">
|
||||
<label className="text-xs font-semibold text-slate-400 uppercase tracking-wider">Notes</label>
|
||||
<textarea
|
||||
value={formData.notes}
|
||||
onChange={(e) => onChange("notes", e.target.value)}
|
||||
placeholder="Optional notes..."
|
||||
rows={3}
|
||||
className="bg-slate-950 border border-slate-800 text-slate-100 rounded-lg px-3 py-2 text-sm focus:outline-none focus:border-emerald-500 transition-colors resize-none"
|
||||
/>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
</div>
|
||||
|
||||
<div className="mt-auto p-5 border-t border-slate-800 bg-slate-900">
|
||||
<button
|
||||
onClick={onSave}
|
||||
disabled={savingGT || !formData.filename}
|
||||
className="w-full bg-emerald-600 hover:bg-emerald-500 text-white rounded-xl py-3 text-sm font-bold shadow-lg shadow-emerald-500/20 disabled:opacity-50 transition-all flex items-center justify-center"
|
||||
>
|
||||
{savingGT ? "Saving..." : "Save Ground Truth"}
|
||||
</button>
|
||||
</div>
|
||||
</div>
|
||||
);
|
||||
}
|
||||
@@ -1,79 +1,79 @@
|
||||
import React, { useState } from "react";
|
||||
|
||||
interface ImageViewerProps {
|
||||
src: string | null;
|
||||
}
|
||||
|
||||
export function ImageViewer({ src }: ImageViewerProps) {
|
||||
const [scale, setScale] = useState(1);
|
||||
const [rotation, setRotation] = useState(0);
|
||||
|
||||
if (!src) {
|
||||
return (
|
||||
<div className="flex-1 flex items-center justify-center bg-slate-950">
|
||||
<span className="text-slate-600 text-sm">No image selected</span>
|
||||
</div>
|
||||
);
|
||||
}
|
||||
|
||||
return (
|
||||
<div className="flex-1 relative flex flex-col bg-slate-950 overflow-hidden">
|
||||
{/* Controls */}
|
||||
<div className="absolute top-4 left-1/2 -translate-x-1/2 z-10 flex items-center gap-2 bg-slate-900/80 backdrop-blur border border-slate-700 p-1.5 rounded-xl shadow-xl">
|
||||
<button
|
||||
onClick={() => setScale((s) => Math.max(0.5, s - 0.25))}
|
||||
className="w-8 h-8 rounded-lg hover:bg-slate-700 text-slate-300 flex items-center justify-center text-lg font-mono"
|
||||
>
|
||||
-
|
||||
</button>
|
||||
<span className="text-xs font-medium text-slate-400 w-12 text-center">
|
||||
{Math.round(scale * 100)}%
|
||||
</span>
|
||||
<button
|
||||
onClick={() => setScale((s) => Math.min(3, s + 0.25))}
|
||||
className="w-8 h-8 rounded-lg hover:bg-slate-700 text-slate-300 flex items-center justify-center text-lg font-mono"
|
||||
>
|
||||
+
|
||||
</button>
|
||||
<div className="w-px h-5 bg-slate-700 mx-1" />
|
||||
<button
|
||||
onClick={() => setRotation((r) => r - 90)}
|
||||
className="w-8 h-8 rounded-lg hover:bg-slate-700 text-slate-300 flex items-center justify-center text-sm"
|
||||
title="Rotate Left"
|
||||
>
|
||||
↺
|
||||
</button>
|
||||
<button
|
||||
onClick={() => setRotation((r) => r + 90)}
|
||||
className="w-8 h-8 rounded-lg hover:bg-slate-700 text-slate-300 flex items-center justify-center text-sm"
|
||||
title="Rotate Right"
|
||||
>
|
||||
↻
|
||||
</button>
|
||||
<div className="w-px h-5 bg-slate-700 mx-1" />
|
||||
<button
|
||||
onClick={() => { setScale(1); setRotation(0); }}
|
||||
className="px-3 h-8 rounded-lg hover:bg-slate-700 text-slate-300 flex items-center justify-center text-xs font-medium"
|
||||
>
|
||||
Reset
|
||||
</button>
|
||||
</div>
|
||||
|
||||
{/* Viewport */}
|
||||
<div className="flex-1 overflow-auto flex items-center justify-center p-4">
|
||||
{/* Using standard img for easy rotation/scaling without Next.js Image component strictness */}
|
||||
<img
|
||||
src={src}
|
||||
alt="Product Scan"
|
||||
style={{
|
||||
transform: `scale(${scale}) rotate(${rotation}deg)`,
|
||||
transition: "transform 0.2s ease-out",
|
||||
maxHeight: "80vh"
|
||||
}}
|
||||
className="shadow-2xl rounded-sm object-contain"
|
||||
crossOrigin="anonymous"
|
||||
/>
|
||||
</div>
|
||||
</div>
|
||||
);
|
||||
}
|
||||
import React, { useState } from "react";
|
||||
|
||||
interface ImageViewerProps {
|
||||
src: string | null;
|
||||
}
|
||||
|
||||
export function ImageViewer({ src }: ImageViewerProps) {
|
||||
const [scale, setScale] = useState(1);
|
||||
const [rotation, setRotation] = useState(0);
|
||||
|
||||
if (!src) {
|
||||
return (
|
||||
<div className="flex-1 flex items-center justify-center bg-slate-950">
|
||||
<span className="text-slate-600 text-sm">No image selected</span>
|
||||
</div>
|
||||
);
|
||||
}
|
||||
|
||||
return (
|
||||
<div className="flex-1 relative flex flex-col bg-slate-950 overflow-hidden">
|
||||
{/* Controls */}
|
||||
<div className="absolute top-4 left-1/2 -translate-x-1/2 z-10 flex items-center gap-2 bg-slate-900/80 backdrop-blur border border-slate-700 p-1.5 rounded-xl shadow-xl">
|
||||
<button
|
||||
onClick={() => setScale((s) => Math.max(0.5, s - 0.25))}
|
||||
className="w-8 h-8 rounded-lg hover:bg-slate-700 text-slate-300 flex items-center justify-center text-lg font-mono"
|
||||
>
|
||||
-
|
||||
</button>
|
||||
<span className="text-xs font-medium text-slate-400 w-12 text-center">
|
||||
{Math.round(scale * 100)}%
|
||||
</span>
|
||||
<button
|
||||
onClick={() => setScale((s) => Math.min(3, s + 0.25))}
|
||||
className="w-8 h-8 rounded-lg hover:bg-slate-700 text-slate-300 flex items-center justify-center text-lg font-mono"
|
||||
>
|
||||
+
|
||||
</button>
|
||||
<div className="w-px h-5 bg-slate-700 mx-1" />
|
||||
<button
|
||||
onClick={() => setRotation((r) => r - 90)}
|
||||
className="w-8 h-8 rounded-lg hover:bg-slate-700 text-slate-300 flex items-center justify-center text-sm"
|
||||
title="Rotate Left"
|
||||
>
|
||||
↺
|
||||
</button>
|
||||
<button
|
||||
onClick={() => setRotation((r) => r + 90)}
|
||||
className="w-8 h-8 rounded-lg hover:bg-slate-700 text-slate-300 flex items-center justify-center text-sm"
|
||||
title="Rotate Right"
|
||||
>
|
||||
↻
|
||||
</button>
|
||||
<div className="w-px h-5 bg-slate-700 mx-1" />
|
||||
<button
|
||||
onClick={() => { setScale(1); setRotation(0); }}
|
||||
className="px-3 h-8 rounded-lg hover:bg-slate-700 text-slate-300 flex items-center justify-center text-xs font-medium"
|
||||
>
|
||||
Reset
|
||||
</button>
|
||||
</div>
|
||||
|
||||
{/* Viewport */}
|
||||
<div className="flex-1 overflow-auto flex items-center justify-center p-4">
|
||||
{/* Using standard img for easy rotation/scaling without Next.js Image component strictness */}
|
||||
<img
|
||||
src={src}
|
||||
alt="Product Scan"
|
||||
style={{
|
||||
transform: `scale(${scale}) rotate(${rotation}deg)`,
|
||||
transition: "transform 0.2s ease-out",
|
||||
maxHeight: "80vh"
|
||||
}}
|
||||
className="shadow-2xl rounded-sm object-contain"
|
||||
crossOrigin="anonymous"
|
||||
/>
|
||||
</div>
|
||||
</div>
|
||||
);
|
||||
}
|
||||
@@ -1,44 +1,44 @@
|
||||
import React from "react";
|
||||
|
||||
interface SidebarProps {
|
||||
files: string[];
|
||||
currentIndex: number;
|
||||
onSelect: (index: number) => void;
|
||||
}
|
||||
|
||||
export function Sidebar({ files, currentIndex, onSelect }: SidebarProps) {
|
||||
return (
|
||||
<div className="w-64 border-r border-slate-800 bg-slate-900/50 flex flex-col h-full shrink-0">
|
||||
<div className="p-4 border-b border-slate-800">
|
||||
<h2 className="text-sm font-bold text-slate-100">Dataset Images</h2>
|
||||
<p className="text-xs text-slate-500 mt-1">{files.length} files found</p>
|
||||
</div>
|
||||
<div className="flex-1 overflow-y-auto p-2 space-y-1">
|
||||
{files.map((file, idx) => {
|
||||
const isSelected = idx === currentIndex;
|
||||
// Extract just the filename for display
|
||||
const display = file.split("/").pop() || file;
|
||||
return (
|
||||
<button
|
||||
key={file}
|
||||
onClick={() => onSelect(idx)}
|
||||
className={`w-full text-left px-3 py-2 rounded-lg text-xs truncate transition-colors ${
|
||||
isSelected
|
||||
? "bg-emerald-500/20 text-emerald-400 font-medium"
|
||||
: "text-slate-400 hover:bg-slate-800/50 hover:text-slate-200"
|
||||
}`}
|
||||
title={file}
|
||||
>
|
||||
{idx + 1}. {display}
|
||||
</button>
|
||||
);
|
||||
})}
|
||||
{files.length === 0 && (
|
||||
<div className="text-center text-xs text-slate-500 mt-4">
|
||||
No images found.
|
||||
</div>
|
||||
)}
|
||||
</div>
|
||||
</div>
|
||||
);
|
||||
}
|
||||
import React from "react";
|
||||
|
||||
interface SidebarProps {
|
||||
files: string[];
|
||||
currentIndex: number;
|
||||
onSelect: (index: number) => void;
|
||||
}
|
||||
|
||||
export function Sidebar({ files, currentIndex, onSelect }: SidebarProps) {
|
||||
return (
|
||||
<div className="w-64 border-r border-slate-800 bg-slate-900/50 flex flex-col h-full shrink-0">
|
||||
<div className="p-4 border-b border-slate-800">
|
||||
<h2 className="text-sm font-bold text-slate-100">Dataset Images</h2>
|
||||
<p className="text-xs text-slate-500 mt-1">{files.length} files found</p>
|
||||
</div>
|
||||
<div className="flex-1 overflow-y-auto p-2 space-y-1">
|
||||
{files.map((file, idx) => {
|
||||
const isSelected = idx === currentIndex;
|
||||
// Extract just the filename for display
|
||||
const display = file.split("/").pop() || file;
|
||||
return (
|
||||
<button
|
||||
key={file}
|
||||
onClick={() => onSelect(idx)}
|
||||
className={`w-full text-left px-3 py-2 rounded-lg text-xs truncate transition-colors ${
|
||||
isSelected
|
||||
? "bg-emerald-500/20 text-emerald-400 font-medium"
|
||||
: "text-slate-400 hover:bg-slate-800/50 hover:text-slate-200"
|
||||
}`}
|
||||
title={file}
|
||||
>
|
||||
{idx + 1}. {display}
|
||||
</button>
|
||||
);
|
||||
})}
|
||||
{files.length === 0 && (
|
||||
<div className="text-center text-xs text-slate-500 mt-4">
|
||||
No images found.
|
||||
</div>
|
||||
)}
|
||||
</div>
|
||||
</div>
|
||||
);
|
||||
}
|
||||
+255
-255
@@ -1,255 +1,255 @@
|
||||
import { Pool, PoolClient } from "pg";
|
||||
import { initDb } from "./init";
|
||||
|
||||
const pool = new Pool({
|
||||
host: process.env.PGHOST || "localhost",
|
||||
port: parseInt(process.env.PGPORT || "5432"),
|
||||
user: process.env.PGUSER || "postgres",
|
||||
password: process.env.PGPASSWORD || "postgres",
|
||||
database: process.env.PGDATABASE || "dopfm",
|
||||
});
|
||||
|
||||
let initialized = false;
|
||||
let initPromise: Promise<Pool> | null = null;
|
||||
|
||||
export async function getPool(): Promise<Pool> {
|
||||
if (initialized) {
|
||||
return pool;
|
||||
}
|
||||
if (!initPromise) {
|
||||
initPromise = (async () => {
|
||||
try {
|
||||
await initDb(pool);
|
||||
initialized = true;
|
||||
} catch (err) {
|
||||
console.error("Failed to initialize database:", err);
|
||||
}
|
||||
return pool;
|
||||
})();
|
||||
}
|
||||
return initPromise;
|
||||
}
|
||||
|
||||
export async function query(text: string, params?: unknown[]) {
|
||||
const p = await getPool();
|
||||
return p.query(text, params);
|
||||
}
|
||||
|
||||
/** Runs `fn` inside a BEGIN/COMMIT transaction on a single held connection, rolling back and rethrowing on any failure. */
|
||||
export async function withTransaction<T>(
|
||||
fn: (client: PoolClient) => Promise<T>
|
||||
): Promise<T> {
|
||||
const p = await getPool();
|
||||
const client = await p.connect();
|
||||
try {
|
||||
await client.query("BEGIN");
|
||||
const result = await fn(client);
|
||||
await client.query("COMMIT");
|
||||
return result;
|
||||
} catch (err) {
|
||||
await client.query("ROLLBACK");
|
||||
throw err;
|
||||
} finally {
|
||||
client.release();
|
||||
}
|
||||
}
|
||||
|
||||
export async function cleanupAndReindexItems(docId: number) {
|
||||
// 1. Delete rows where kode_barang is blank/null or doesn't match an 8-digit number
|
||||
await query(
|
||||
`DELETE FROM ocr_items
|
||||
WHERE document_id = $1
|
||||
AND (kode_barang IS NULL OR TRIM(kode_barang) = '' OR NOT (kode_barang ~ '^[0-9]{8}$'))`,
|
||||
[docId]
|
||||
);
|
||||
|
||||
// 2. Fetch remaining rows ordered by row_index
|
||||
const res = await query(
|
||||
`SELECT id, row_index
|
||||
FROM ocr_items
|
||||
WHERE document_id = $1
|
||||
ORDER BY row_index`,
|
||||
[docId]
|
||||
);
|
||||
|
||||
// 3. Update row_index to be sequential
|
||||
for (let i = 0; i < res.rows.length; i++) {
|
||||
const row = res.rows[i];
|
||||
if (row.row_index !== i) {
|
||||
await query(
|
||||
`UPDATE ocr_items
|
||||
SET row_index = $1
|
||||
WHERE id = $2`,
|
||||
[i, row.id]
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
const STORE_STOPWORDS = new Set([
|
||||
"dan", "dki", "area", "yth", "kepada", "order", "untuk", "alamat", "kel", "kec", "rt", "rw",
|
||||
"jalan", "raya", "blok", "nomor", "kelurahan", "kecamatan", "kota", "kabupaten", "provinsi"
|
||||
]);
|
||||
|
||||
function tokenize(text: string): string[] {
|
||||
return text.toLowerCase()
|
||||
.replace(/[^a-z0-9\s]/g, " ")
|
||||
.split(/\s+/)
|
||||
.filter(w => w.length > 2 && !STORE_STOPWORDS.has(w));
|
||||
}
|
||||
|
||||
// customers.name is stored as "CUSTOMER NAME, JL. street address..." - split on the first
|
||||
// street-address marker to get just the canonical address portion.
|
||||
function splitCustomerAddress(name: string): string {
|
||||
const m = name.match(/\b(?:JL\.?|JALAN)\b[\s\S]*/i);
|
||||
return (m ? m[0] : name).replace(/\s+/g, " ").trim();
|
||||
}
|
||||
|
||||
// A noisy OCR'd address (varying per document due to misread letters) that recognizably
|
||||
// belongs to a known customer should be reported as that customer's clean canonical address,
|
||||
// rather than whatever garbled text this particular scan happened to produce.
|
||||
async function canonicalizeCustomerAddress(extracted: string): Promise<string> {
|
||||
if (!extracted) return extracted;
|
||||
|
||||
const extractedTokens = new Set(tokenize(extracted));
|
||||
if (extractedTokens.size === 0) return extracted;
|
||||
|
||||
const customersRes = await query("SELECT name FROM customers");
|
||||
|
||||
let bestAddress: string | null = null;
|
||||
let bestMatchCount = 0;
|
||||
let bestScore = 0;
|
||||
|
||||
for (const row of customersRes.rows) {
|
||||
const canonicalAddress = splitCustomerAddress(row.name);
|
||||
const addressTokens = tokenize(canonicalAddress);
|
||||
if (addressTokens.length === 0) continue;
|
||||
|
||||
const uniqueAddressTokens = new Set(addressTokens);
|
||||
let matchCount = 0;
|
||||
for (const token of uniqueAddressTokens) {
|
||||
if (extractedTokens.has(token)) matchCount++;
|
||||
}
|
||||
const score = matchCount / uniqueAddressTokens.size;
|
||||
|
||||
if (matchCount >= 3 && score >= 0.45 && (matchCount > bestMatchCount || (matchCount === bestMatchCount && score > bestScore))) {
|
||||
bestMatchCount = matchCount;
|
||||
bestScore = score;
|
||||
bestAddress = canonicalAddress;
|
||||
}
|
||||
}
|
||||
|
||||
return bestAddress ?? extracted;
|
||||
}
|
||||
|
||||
// The delivery truck/signature line near the bottom of the table ("Truck No. B 9427 UXT
|
||||
// PX HEAD OFFICE ANCOL : JL. ANCOL BARAT VIII...") names the actual destination store, when
|
||||
// present. Scoping the match to just this line (and just nama_toko, not nama_toko+alamat)
|
||||
// avoids the customer's own fixed head-office address elsewhere in the document being
|
||||
// mistaken for the destination - that address is present on every document regardless of
|
||||
// which store it's actually going to, so matching against it produces confident false
|
||||
// positives for documents that don't specify a destination store name at all.
|
||||
function extractTruckLineSnippet(fullText: string): string {
|
||||
const m = fullText.match(/Truck\s*No\.?[\s\S]{0,180}/i);
|
||||
return m ? m[0] : "";
|
||||
}
|
||||
|
||||
// True when the printed "Order Untuk" text is actually the customer's company name - a common
|
||||
// OCR layout jumble where the "Kepada Yth" and "Order Untuk" fields merge, meaning the real
|
||||
// destination value was lost and the truck line is the better signal.
|
||||
async function looksLikeCustomerName(text: string): Promise<boolean> {
|
||||
if (!text) return false;
|
||||
const textTokens = new Set(tokenize(text));
|
||||
if (textTokens.size === 0) return false;
|
||||
|
||||
const customersRes = await query("SELECT name FROM customers");
|
||||
for (const row of customersRes.rows) {
|
||||
const companyName = String(row.name).split(/\bJL\.?\b|\bJALAN\b/i)[0];
|
||||
const nameTokens = tokenize(companyName);
|
||||
if (nameTokens.length === 0) continue;
|
||||
let matchCount = 0;
|
||||
for (const token of new Set(nameTokens)) {
|
||||
if (textTokens.has(token)) matchCount++;
|
||||
}
|
||||
if (matchCount >= 1 && matchCount / new Set(nameTokens).size >= 0.5) return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
export async function resolveStoreFromText(fullMarkdown: string): Promise<{ orderUntuk: string; alamat: string }> {
|
||||
if (!fullMarkdown || fullMarkdown === "Not Found") {
|
||||
return { orderUntuk: "", alamat: "" };
|
||||
}
|
||||
|
||||
// The printed "Alamat" field is the customer's own (fixed) address, not the destination
|
||||
// store's registered address - it stays the same across documents regardless of which
|
||||
// store the truck line names. So alamat always comes from the literal printed text; only
|
||||
// the store name itself benefits from being resolved to its canonical store_master form.
|
||||
// The line right after "Alamat:" sometimes holds a region code ("DKI AREA") rather than the
|
||||
// street address, with the real address following on the next line(s) - capture the whole
|
||||
// block up to the item table and prefer the "JL./JALAN ..." street-address line within it.
|
||||
const alamatBlockMatch = fullMarkdown.match(/Alamat\s*[:\-]?\s*([\s\S]+?)(?=<table|$)/i);
|
||||
let literalAlamat = "";
|
||||
if (alamatBlockMatch) {
|
||||
const block = alamatBlockMatch[1];
|
||||
const streetMatch = block.match(/\b(?:JL\.?|JALAN)\b[\s\S]*/i);
|
||||
literalAlamat = (streetMatch ? streetMatch[0] : block).replace(/\s+/g, " ").trim();
|
||||
}
|
||||
literalAlamat = await canonicalizeCustomerAddress(literalAlamat);
|
||||
|
||||
// The printed "Order Untuk" value is the primary source for the store field: it usually
|
||||
// holds a region designator ("DKI AREA", "PFM-KU") or a store name, and that's what the
|
||||
// document actually says. Only when OCR jumbled it with the customer's company name (or
|
||||
// lost it entirely) do we fall back to matching the truck/signature line against
|
||||
// store_master to recover the destination store.
|
||||
const orderMatch = fullMarkdown.match(/Order\s+Untuk\s*[:\-]\s*([^\n]+)/i);
|
||||
const literalOrder = orderMatch ? orderMatch[1].trim() : "";
|
||||
const orderIsUsable = literalOrder !== "" && !(await looksLikeCustomerName(literalOrder));
|
||||
|
||||
if (orderIsUsable) {
|
||||
return { orderUntuk: literalOrder, alamat: literalAlamat };
|
||||
}
|
||||
|
||||
const storeRes = await query("SELECT nama_toko, kode_toko, alamat FROM store_master");
|
||||
const stores = storeRes.rows;
|
||||
|
||||
const truckSnippet = extractTruckLineSnippet(fullMarkdown);
|
||||
const snippetTokens = new Set(tokenize(truckSnippet));
|
||||
|
||||
let bestStore: any = null;
|
||||
let bestScore = 0;
|
||||
let bestMatchCount = 0;
|
||||
|
||||
if (snippetTokens.size > 0) {
|
||||
for (const store of stores) {
|
||||
const storeTokens = tokenize(store.nama_toko);
|
||||
if (storeTokens.length === 0) continue;
|
||||
|
||||
const uniqueStoreTokens = new Set(storeTokens);
|
||||
let matchCount = 0;
|
||||
for (const token of uniqueStoreTokens) {
|
||||
if (snippetTokens.has(token)) matchCount++;
|
||||
}
|
||||
|
||||
const score = matchCount / uniqueStoreTokens.size;
|
||||
// Two distinct matching tokens minimum: single-token overlaps (e.g. a store whose only
|
||||
// distinctive token is a common street/area word appearing in the snippet's address
|
||||
// text) produce far too many confident false positives.
|
||||
if (matchCount >= 2) {
|
||||
if (matchCount > bestMatchCount || (matchCount === bestMatchCount && score > bestScore)) {
|
||||
bestMatchCount = matchCount;
|
||||
bestScore = score;
|
||||
bestStore = store;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (bestStore) {
|
||||
return { orderUntuk: bestStore.nama_toko, alamat: literalAlamat };
|
||||
}
|
||||
|
||||
return { orderUntuk: literalOrder, alamat: literalAlamat };
|
||||
}
|
||||
|
||||
export { pool };
|
||||
import { Pool, PoolClient } from "pg";
|
||||
import { initDb } from "./init";
|
||||
|
||||
const pool = new Pool({
|
||||
host: process.env.PGHOST || "localhost",
|
||||
port: parseInt(process.env.PGPORT || "5432"),
|
||||
user: process.env.PGUSER || "postgres",
|
||||
password: process.env.PGPASSWORD || "postgres",
|
||||
database: process.env.PGDATABASE || "dopfm",
|
||||
});
|
||||
|
||||
let initialized = false;
|
||||
let initPromise: Promise<Pool> | null = null;
|
||||
|
||||
export async function getPool(): Promise<Pool> {
|
||||
if (initialized) {
|
||||
return pool;
|
||||
}
|
||||
if (!initPromise) {
|
||||
initPromise = (async () => {
|
||||
try {
|
||||
await initDb(pool);
|
||||
initialized = true;
|
||||
} catch (err) {
|
||||
console.error("Failed to initialize database:", err);
|
||||
}
|
||||
return pool;
|
||||
})();
|
||||
}
|
||||
return initPromise;
|
||||
}
|
||||
|
||||
export async function query(text: string, params?: unknown[]) {
|
||||
const p = await getPool();
|
||||
return p.query(text, params);
|
||||
}
|
||||
|
||||
/** Runs `fn` inside a BEGIN/COMMIT transaction on a single held connection, rolling back and rethrowing on any failure. */
|
||||
export async function withTransaction<T>(
|
||||
fn: (client: PoolClient) => Promise<T>
|
||||
): Promise<T> {
|
||||
const p = await getPool();
|
||||
const client = await p.connect();
|
||||
try {
|
||||
await client.query("BEGIN");
|
||||
const result = await fn(client);
|
||||
await client.query("COMMIT");
|
||||
return result;
|
||||
} catch (err) {
|
||||
await client.query("ROLLBACK");
|
||||
throw err;
|
||||
} finally {
|
||||
client.release();
|
||||
}
|
||||
}
|
||||
|
||||
export async function cleanupAndReindexItems(docId: number) {
|
||||
// 1. Delete rows where kode_barang is blank/null or doesn't match an 8-digit number
|
||||
await query(
|
||||
`DELETE FROM ocr_items
|
||||
WHERE document_id = $1
|
||||
AND (kode_barang IS NULL OR TRIM(kode_barang) = '' OR NOT (kode_barang ~ '^[0-9]{8}$'))`,
|
||||
[docId]
|
||||
);
|
||||
|
||||
// 2. Fetch remaining rows ordered by row_index
|
||||
const res = await query(
|
||||
`SELECT id, row_index
|
||||
FROM ocr_items
|
||||
WHERE document_id = $1
|
||||
ORDER BY row_index`,
|
||||
[docId]
|
||||
);
|
||||
|
||||
// 3. Update row_index to be sequential
|
||||
for (let i = 0; i < res.rows.length; i++) {
|
||||
const row = res.rows[i];
|
||||
if (row.row_index !== i) {
|
||||
await query(
|
||||
`UPDATE ocr_items
|
||||
SET row_index = $1
|
||||
WHERE id = $2`,
|
||||
[i, row.id]
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
const STORE_STOPWORDS = new Set([
|
||||
"dan", "dki", "area", "yth", "kepada", "order", "untuk", "alamat", "kel", "kec", "rt", "rw",
|
||||
"jalan", "raya", "blok", "nomor", "kelurahan", "kecamatan", "kota", "kabupaten", "provinsi"
|
||||
]);
|
||||
|
||||
function tokenize(text: string): string[] {
|
||||
return text.toLowerCase()
|
||||
.replace(/[^a-z0-9\s]/g, " ")
|
||||
.split(/\s+/)
|
||||
.filter(w => w.length > 2 && !STORE_STOPWORDS.has(w));
|
||||
}
|
||||
|
||||
// customers.name is stored as "CUSTOMER NAME, JL. street address..." - split on the first
|
||||
// street-address marker to get just the canonical address portion.
|
||||
function splitCustomerAddress(name: string): string {
|
||||
const m = name.match(/\b(?:JL\.?|JALAN)\b[\s\S]*/i);
|
||||
return (m ? m[0] : name).replace(/\s+/g, " ").trim();
|
||||
}
|
||||
|
||||
// A noisy OCR'd address (varying per document due to misread letters) that recognizably
|
||||
// belongs to a known customer should be reported as that customer's clean canonical address,
|
||||
// rather than whatever garbled text this particular scan happened to produce.
|
||||
async function canonicalizeCustomerAddress(extracted: string): Promise<string> {
|
||||
if (!extracted) return extracted;
|
||||
|
||||
const extractedTokens = new Set(tokenize(extracted));
|
||||
if (extractedTokens.size === 0) return extracted;
|
||||
|
||||
const customersRes = await query("SELECT name FROM customers");
|
||||
|
||||
let bestAddress: string | null = null;
|
||||
let bestMatchCount = 0;
|
||||
let bestScore = 0;
|
||||
|
||||
for (const row of customersRes.rows) {
|
||||
const canonicalAddress = splitCustomerAddress(row.name);
|
||||
const addressTokens = tokenize(canonicalAddress);
|
||||
if (addressTokens.length === 0) continue;
|
||||
|
||||
const uniqueAddressTokens = new Set(addressTokens);
|
||||
let matchCount = 0;
|
||||
for (const token of uniqueAddressTokens) {
|
||||
if (extractedTokens.has(token)) matchCount++;
|
||||
}
|
||||
const score = matchCount / uniqueAddressTokens.size;
|
||||
|
||||
if (matchCount >= 3 && score >= 0.45 && (matchCount > bestMatchCount || (matchCount === bestMatchCount && score > bestScore))) {
|
||||
bestMatchCount = matchCount;
|
||||
bestScore = score;
|
||||
bestAddress = canonicalAddress;
|
||||
}
|
||||
}
|
||||
|
||||
return bestAddress ?? extracted;
|
||||
}
|
||||
|
||||
// The delivery truck/signature line near the bottom of the table ("Truck No. B 9427 UXT
|
||||
// PX HEAD OFFICE ANCOL : JL. ANCOL BARAT VIII...") names the actual destination store, when
|
||||
// present. Scoping the match to just this line (and just nama_toko, not nama_toko+alamat)
|
||||
// avoids the customer's own fixed head-office address elsewhere in the document being
|
||||
// mistaken for the destination - that address is present on every document regardless of
|
||||
// which store it's actually going to, so matching against it produces confident false
|
||||
// positives for documents that don't specify a destination store name at all.
|
||||
function extractTruckLineSnippet(fullText: string): string {
|
||||
const m = fullText.match(/Truck\s*No\.?[\s\S]{0,180}/i);
|
||||
return m ? m[0] : "";
|
||||
}
|
||||
|
||||
// True when the printed "Order Untuk" text is actually the customer's company name - a common
|
||||
// OCR layout jumble where the "Kepada Yth" and "Order Untuk" fields merge, meaning the real
|
||||
// destination value was lost and the truck line is the better signal.
|
||||
async function looksLikeCustomerName(text: string): Promise<boolean> {
|
||||
if (!text) return false;
|
||||
const textTokens = new Set(tokenize(text));
|
||||
if (textTokens.size === 0) return false;
|
||||
|
||||
const customersRes = await query("SELECT name FROM customers");
|
||||
for (const row of customersRes.rows) {
|
||||
const companyName = String(row.name).split(/\bJL\.?\b|\bJALAN\b/i)[0];
|
||||
const nameTokens = tokenize(companyName);
|
||||
if (nameTokens.length === 0) continue;
|
||||
let matchCount = 0;
|
||||
for (const token of new Set(nameTokens)) {
|
||||
if (textTokens.has(token)) matchCount++;
|
||||
}
|
||||
if (matchCount >= 1 && matchCount / new Set(nameTokens).size >= 0.5) return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
export async function resolveStoreFromText(fullMarkdown: string): Promise<{ orderUntuk: string; alamat: string }> {
|
||||
if (!fullMarkdown || fullMarkdown === "Not Found") {
|
||||
return { orderUntuk: "", alamat: "" };
|
||||
}
|
||||
|
||||
// The printed "Alamat" field is the customer's own (fixed) address, not the destination
|
||||
// store's registered address - it stays the same across documents regardless of which
|
||||
// store the truck line names. So alamat always comes from the literal printed text; only
|
||||
// the store name itself benefits from being resolved to its canonical store_master form.
|
||||
// The line right after "Alamat:" sometimes holds a region code ("DKI AREA") rather than the
|
||||
// street address, with the real address following on the next line(s) - capture the whole
|
||||
// block up to the item table and prefer the "JL./JALAN ..." street-address line within it.
|
||||
const alamatBlockMatch = fullMarkdown.match(/Alamat\s*[:\-]?\s*([\s\S]+?)(?=<table|$)/i);
|
||||
let literalAlamat = "";
|
||||
if (alamatBlockMatch) {
|
||||
const block = alamatBlockMatch[1];
|
||||
const streetMatch = block.match(/\b(?:JL\.?|JALAN)\b[\s\S]*/i);
|
||||
literalAlamat = (streetMatch ? streetMatch[0] : block).replace(/\s+/g, " ").trim();
|
||||
}
|
||||
literalAlamat = await canonicalizeCustomerAddress(literalAlamat);
|
||||
|
||||
// The printed "Order Untuk" value is the primary source for the store field: it usually
|
||||
// holds a region designator ("DKI AREA", "PFM-KU") or a store name, and that's what the
|
||||
// document actually says. Only when OCR jumbled it with the customer's company name (or
|
||||
// lost it entirely) do we fall back to matching the truck/signature line against
|
||||
// store_master to recover the destination store.
|
||||
const orderMatch = fullMarkdown.match(/Order\s+Untuk\s*[:\-]\s*([^\n]+)/i);
|
||||
const literalOrder = orderMatch ? orderMatch[1].trim() : "";
|
||||
const orderIsUsable = literalOrder !== "" && !(await looksLikeCustomerName(literalOrder));
|
||||
|
||||
if (orderIsUsable) {
|
||||
return { orderUntuk: literalOrder, alamat: literalAlamat };
|
||||
}
|
||||
|
||||
const storeRes = await query("SELECT nama_toko, kode_toko, alamat FROM store_master");
|
||||
const stores = storeRes.rows;
|
||||
|
||||
const truckSnippet = extractTruckLineSnippet(fullMarkdown);
|
||||
const snippetTokens = new Set(tokenize(truckSnippet));
|
||||
|
||||
let bestStore: any = null;
|
||||
let bestScore = 0;
|
||||
let bestMatchCount = 0;
|
||||
|
||||
if (snippetTokens.size > 0) {
|
||||
for (const store of stores) {
|
||||
const storeTokens = tokenize(store.nama_toko);
|
||||
if (storeTokens.length === 0) continue;
|
||||
|
||||
const uniqueStoreTokens = new Set(storeTokens);
|
||||
let matchCount = 0;
|
||||
for (const token of uniqueStoreTokens) {
|
||||
if (snippetTokens.has(token)) matchCount++;
|
||||
}
|
||||
|
||||
const score = matchCount / uniqueStoreTokens.size;
|
||||
// Two distinct matching tokens minimum: single-token overlaps (e.g. a store whose only
|
||||
// distinctive token is a common street/area word appearing in the snippet's address
|
||||
// text) produce far too many confident false positives.
|
||||
if (matchCount >= 2) {
|
||||
if (matchCount > bestMatchCount || (matchCount === bestMatchCount && score > bestScore)) {
|
||||
bestMatchCount = matchCount;
|
||||
bestScore = score;
|
||||
bestStore = store;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (bestStore) {
|
||||
return { orderUntuk: bestStore.nama_toko, alamat: literalAlamat };
|
||||
}
|
||||
|
||||
return { orderUntuk: literalOrder, alamat: literalAlamat };
|
||||
}
|
||||
|
||||
export { pool };
|
||||
+562
-562
File diff suppressed because it is too large.
Load diff
@@ -1,47 +1,47 @@
|
||||
// Framework-agnostic HTTP status/error helpers shared by server routes and client components.
|
||||
|
||||
export const HTTP_STATUS_TEXT: Record<number, string> = {
|
||||
400: "Bad Request",
|
||||
401: "Unauthorized",
|
||||
403: "Forbidden",
|
||||
404: "Not Found",
|
||||
405: "Method Not Allowed",
|
||||
409: "Conflict",
|
||||
413: "Payload Too Large",
|
||||
422: "Unprocessable Entity",
|
||||
429: "Too Many Requests",
|
||||
500: "Internal Server Error",
|
||||
502: "Bad Gateway",
|
||||
503: "Service Unavailable",
|
||||
504: "Gateway Timeout",
|
||||
};
|
||||
|
||||
export function reasonPhraseForStatus(status: number): string {
|
||||
return HTTP_STATUS_TEXT[status] ?? "Error";
|
||||
}
|
||||
|
||||
export function codeForStatus(status: number): string {
|
||||
const phrase = HTTP_STATUS_TEXT[status];
|
||||
if (!phrase) return `HTTP_${status}`;
|
||||
return phrase.toUpperCase().replace(/[^A-Z0-9]+/g, "_");
|
||||
}
|
||||
|
||||
export interface ApiErrorBody {
|
||||
status: "error";
|
||||
error: {
|
||||
statusCode: number;
|
||||
code: string;
|
||||
message: string;
|
||||
};
|
||||
}
|
||||
|
||||
export function buildApiErrorBody(status: number, message: string, code?: string): ApiErrorBody {
|
||||
return {
|
||||
status: "error",
|
||||
error: {
|
||||
statusCode: status,
|
||||
code: code ?? codeForStatus(status),
|
||||
message,
|
||||
},
|
||||
};
|
||||
}
|
||||
// Framework-agnostic HTTP status/error helpers shared by server routes and client components.
|
||||
|
||||
export const HTTP_STATUS_TEXT: Record<number, string> = {
|
||||
400: "Bad Request",
|
||||
401: "Unauthorized",
|
||||
403: "Forbidden",
|
||||
404: "Not Found",
|
||||
405: "Method Not Allowed",
|
||||
409: "Conflict",
|
||||
413: "Payload Too Large",
|
||||
422: "Unprocessable Entity",
|
||||
429: "Too Many Requests",
|
||||
500: "Internal Server Error",
|
||||
502: "Bad Gateway",
|
||||
503: "Service Unavailable",
|
||||
504: "Gateway Timeout",
|
||||
};
|
||||
|
||||
export function reasonPhraseForStatus(status: number): string {
|
||||
return HTTP_STATUS_TEXT[status] ?? "Error";
|
||||
}
|
||||
|
||||
export function codeForStatus(status: number): string {
|
||||
const phrase = HTTP_STATUS_TEXT[status];
|
||||
if (!phrase) return `HTTP_${status}`;
|
||||
return phrase.toUpperCase().replace(/[^A-Z0-9]+/g, "_");
|
||||
}
|
||||
|
||||
export interface ApiErrorBody {
|
||||
status: "error";
|
||||
error: {
|
||||
statusCode: number;
|
||||
code: string;
|
||||
message: string;
|
||||
};
|
||||
}
|
||||
|
||||
export function buildApiErrorBody(status: number, message: string, code?: string): ApiErrorBody {
|
||||
return {
|
||||
status: "error",
|
||||
error: {
|
||||
statusCode: status,
|
||||
code: code ?? codeForStatus(status),
|
||||
message,
|
||||
},
|
||||
};
|
||||
}
|
||||
@@ -1,80 +1,80 @@
|
||||
// Active Log Tracker for OCR Processing Layers
|
||||
|
||||
export interface VllmCall {
|
||||
request: any;
|
||||
response: any;
|
||||
timestamp: string;
|
||||
}
|
||||
|
||||
export interface ActiveUploadLog {
|
||||
filename: string;
|
||||
vllm_calls: VllmCall[];
|
||||
ocr_raw?: any;
|
||||
stage_1_output?: any;
|
||||
stage_2_output?: any;
|
||||
frontend_response?: any;
|
||||
pipeline_info?: any;
|
||||
}
|
||||
|
||||
// Store active logs in global context as a Map keyed by filename
|
||||
// This supports concurrent uploads without race conditions
|
||||
const globalForActiveLog = global as unknown as {
|
||||
activeLogs: Map<string, ActiveUploadLog>;
|
||||
};
|
||||
|
||||
// Initialise the map once (survives Next.js hot-reloads on the same process)
|
||||
if (!globalForActiveLog.activeLogs) {
|
||||
globalForActiveLog.activeLogs = new Map();
|
||||
}
|
||||
|
||||
export function startActiveLog(filename: string) {
|
||||
globalForActiveLog.activeLogs.set(filename, {
|
||||
filename,
|
||||
vllm_calls: []
|
||||
});
|
||||
console.log(`[ActiveLog] Started tracking log for ${filename}`);
|
||||
}
|
||||
|
||||
export function logVllmCall(filename: string, request: any, response: any) {
|
||||
const log = globalForActiveLog.activeLogs.get(filename);
|
||||
if (log) {
|
||||
log.vllm_calls.push({
|
||||
request,
|
||||
response,
|
||||
timestamp: new Date().toISOString()
|
||||
});
|
||||
console.log(`[ActiveLog] Logged vLLM call for ${filename} (total calls: ${log.vllm_calls.length})`);
|
||||
} else {
|
||||
console.log(`[ActiveLog] Warning: Attempted to log vLLM call for "${filename}" but no active log session is running.`);
|
||||
}
|
||||
}
|
||||
|
||||
export function getActiveLog(filename: string): ActiveUploadLog | null {
|
||||
return globalForActiveLog.activeLogs.get(filename) || null;
|
||||
}
|
||||
|
||||
export function clearActiveLog(filename: string) {
|
||||
globalForActiveLog.activeLogs.delete(filename);
|
||||
console.log(`[ActiveLog] Cleared active log tracking context for ${filename}`);
|
||||
}
|
||||
|
||||
/**
|
||||
* Log a vLLM call to ALL currently active upload sessions.
|
||||
* Used by the vllm-proxy, which doesn't have per-upload filename context,
|
||||
* since the pipeline-api processes exactly one upload at a time.
|
||||
*/
|
||||
export function logVllmCallToAll(request: any, response: any) {
|
||||
const sessions = globalForActiveLog.activeLogs;
|
||||
if (sessions.size === 0) {
|
||||
console.log("[ActiveLog] Warning: Attempted to log vLLM call but no active log session is running.");
|
||||
return;
|
||||
}
|
||||
for (const [filename, log] of sessions) {
|
||||
log.vllm_calls.push({
|
||||
request,
|
||||
response,
|
||||
timestamp: new Date().toISOString()
|
||||
});
|
||||
console.log(`[ActiveLog] Logged vLLM call for ${filename} (total calls: ${log.vllm_calls.length})`);
|
||||
}
|
||||
}
|
||||
// Active Log Tracker for OCR Processing Layers
|
||||
|
||||
export interface VllmCall {
|
||||
request: any;
|
||||
response: any;
|
||||
timestamp: string;
|
||||
}
|
||||
|
||||
export interface ActiveUploadLog {
|
||||
filename: string;
|
||||
vllm_calls: VllmCall[];
|
||||
ocr_raw?: any;
|
||||
stage_1_output?: any;
|
||||
stage_2_output?: any;
|
||||
frontend_response?: any;
|
||||
pipeline_info?: any;
|
||||
}
|
||||
|
||||
// Store active logs in global context as a Map keyed by filename
|
||||
// This supports concurrent uploads without race conditions
|
||||
const globalForActiveLog = global as unknown as {
|
||||
activeLogs: Map<string, ActiveUploadLog>;
|
||||
};
|
||||
|
||||
// Initialise the map once (survives Next.js hot-reloads on the same process)
|
||||
if (!globalForActiveLog.activeLogs) {
|
||||
globalForActiveLog.activeLogs = new Map();
|
||||
}
|
||||
|
||||
export function startActiveLog(filename: string) {
|
||||
globalForActiveLog.activeLogs.set(filename, {
|
||||
filename,
|
||||
vllm_calls: []
|
||||
});
|
||||
console.log(`[ActiveLog] Started tracking log for ${filename}`);
|
||||
}
|
||||
|
||||
export function logVllmCall(filename: string, request: any, response: any) {
|
||||
const log = globalForActiveLog.activeLogs.get(filename);
|
||||
if (log) {
|
||||
log.vllm_calls.push({
|
||||
request,
|
||||
response,
|
||||
timestamp: new Date().toISOString()
|
||||
});
|
||||
console.log(`[ActiveLog] Logged vLLM call for ${filename} (total calls: ${log.vllm_calls.length})`);
|
||||
} else {
|
||||
console.log(`[ActiveLog] Warning: Attempted to log vLLM call for "${filename}" but no active log session is running.`);
|
||||
}
|
||||
}
|
||||
|
||||
export function getActiveLog(filename: string): ActiveUploadLog | null {
|
||||
return globalForActiveLog.activeLogs.get(filename) || null;
|
||||
}
|
||||
|
||||
export function clearActiveLog(filename: string) {
|
||||
globalForActiveLog.activeLogs.delete(filename);
|
||||
console.log(`[ActiveLog] Cleared active log tracking context for ${filename}`);
|
||||
}
|
||||
|
||||
/**
|
||||
* Log a vLLM call to ALL currently active upload sessions.
|
||||
* Used by the vllm-proxy, which doesn't have per-upload filename context,
|
||||
* since the pipeline-api processes exactly one upload at a time.
|
||||
*/
|
||||
export function logVllmCallToAll(request: any, response: any) {
|
||||
const sessions = globalForActiveLog.activeLogs;
|
||||
if (sessions.size === 0) {
|
||||
console.log("[ActiveLog] Warning: Attempted to log vLLM call but no active log session is running.");
|
||||
return;
|
||||
}
|
||||
for (const [filename, log] of sessions) {
|
||||
log.vllm_calls.push({
|
||||
request,
|
||||
response,
|
||||
timestamp: new Date().toISOString()
|
||||
});
|
||||
console.log(`[ActiveLog] Logged vLLM call for ${filename} (total calls: ${log.vllm_calls.length})`);
|
||||
}
|
||||
}
|
||||
@@ -1,13 +1,13 @@
|
||||
import { NextResponse } from "next/server";
|
||||
import { buildApiErrorBody } from "@/lib/http-status";
|
||||
|
||||
export function errorResponse(
|
||||
status: number,
|
||||
message: string,
|
||||
opts?: { code?: string; headers?: HeadersInit }
|
||||
): NextResponse {
|
||||
return NextResponse.json(buildApiErrorBody(status, message, opts?.code), {
|
||||
status,
|
||||
headers: opts?.headers,
|
||||
});
|
||||
}
|
||||
import { NextResponse } from "next/server";
|
||||
import { buildApiErrorBody } from "@/lib/http-status";
|
||||
|
||||
export function errorResponse(
|
||||
status: number,
|
||||
message: string,
|
||||
opts?: { code?: string; headers?: HeadersInit }
|
||||
): NextResponse {
|
||||
return NextResponse.json(buildApiErrorBody(status, message, opts?.code), {
|
||||
status,
|
||||
headers: opts?.headers,
|
||||
});
|
||||
}
|
||||
@@ -1,31 +1,31 @@
|
||||
import jwt from "jsonwebtoken";
|
||||
|
||||
const JWT_SECRET = process.env.JWT_SECRET || "dev-only-insecure-secret-change-me";
|
||||
|
||||
export interface AccountTokenPayload {
|
||||
accountId: number;
|
||||
username: string;
|
||||
kodeToko: string | null;
|
||||
role?: string;
|
||||
}
|
||||
|
||||
export function signAccountToken(payload: AccountTokenPayload): string {
|
||||
return jwt.sign(payload, JWT_SECRET, { expiresIn: "30d" });
|
||||
}
|
||||
|
||||
/** Returns the decoded payload, or null if the token is missing/invalid/expired. */
|
||||
export function verifyAccountToken(token: string | null | undefined): AccountTokenPayload | null {
|
||||
if (!token) return null;
|
||||
try {
|
||||
return jwt.verify(token, JWT_SECRET) as AccountTokenPayload;
|
||||
} catch {
|
||||
return null;
|
||||
}
|
||||
}
|
||||
|
||||
/** Extracts and verifies the Bearer token from a request's Authorization header. */
|
||||
export function getAccountFromAuthHeader(authHeader: string | null): AccountTokenPayload | null {
|
||||
if (!authHeader?.startsWith("Bearer ")) return null;
|
||||
const token = authHeader.slice("Bearer ".length).trim();
|
||||
return verifyAccountToken(token);
|
||||
}
|
||||
import jwt from "jsonwebtoken";
|
||||
|
||||
const JWT_SECRET = process.env.JWT_SECRET || "dev-only-insecure-secret-change-me";
|
||||
|
||||
export interface AccountTokenPayload {
|
||||
accountId: number;
|
||||
username: string;
|
||||
kodeToko: string | null;
|
||||
role?: string;
|
||||
}
|
||||
|
||||
export function signAccountToken(payload: AccountTokenPayload): string {
|
||||
return jwt.sign(payload, JWT_SECRET, { expiresIn: "30d" });
|
||||
}
|
||||
|
||||
/** Returns the decoded payload, or null if the token is missing/invalid/expired. */
|
||||
export function verifyAccountToken(token: string | null | undefined): AccountTokenPayload | null {
|
||||
if (!token) return null;
|
||||
try {
|
||||
return jwt.verify(token, JWT_SECRET) as AccountTokenPayload;
|
||||
} catch {
|
||||
return null;
|
||||
}
|
||||
}
|
||||
|
||||
/** Extracts and verifies the Bearer token from a request's Authorization header. */
|
||||
export function getAccountFromAuthHeader(authHeader: string | null): AccountTokenPayload | null {
|
||||
if (!authHeader?.startsWith("Bearer ")) return null;
|
||||
const token = authHeader.slice("Bearer ".length).trim();
|
||||
return verifyAccountToken(token);
|
||||
}
|
||||
@@ -1,20 +1,20 @@
|
||||
import { HTTP_STATUS_TEXT } from "@/lib/http-status";
|
||||
|
||||
/**
|
||||
* Formats an error the same way across every client component: given the
|
||||
* parsed JSON body of a failed fetch (if any) and/or the caught exception,
|
||||
* produce a single "404 Not Found: message" style string mirroring the
|
||||
* backend's { status: "error", error: { statusCode, code, message } } envelope.
|
||||
*/
|
||||
export function getErrorMessage(err: unknown, data?: unknown, fallback = "Unexpected error"): string {
|
||||
const errorBody = data && typeof data === "object" ? (data as Record<string, unknown>).error : undefined;
|
||||
if (errorBody && typeof errorBody === "object" && typeof (errorBody as Record<string, unknown>).message === "string") {
|
||||
const body = errorBody as Record<string, unknown>;
|
||||
const statusCode = body.statusCode;
|
||||
const label = (typeof statusCode === "number" && HTTP_STATUS_TEXT[statusCode]) || body.code;
|
||||
return typeof statusCode === "number" ? `${statusCode} ${label}: ${body.message}` : (body.message as string);
|
||||
}
|
||||
if (err instanceof Error) return err.message;
|
||||
if (err != null) return String(err);
|
||||
return fallback;
|
||||
}
|
||||
import { HTTP_STATUS_TEXT } from "@/lib/http-status";
|
||||
|
||||
/**
|
||||
* Formats an error the same way across every client component: given the
|
||||
* parsed JSON body of a failed fetch (if any) and/or the caught exception,
|
||||
* produce a single "404 Not Found: message" style string mirroring the
|
||||
* backend's { status: "error", error: { statusCode, code, message } } envelope.
|
||||
*/
|
||||
export function getErrorMessage(err: unknown, data?: unknown, fallback = "Unexpected error"): string {
|
||||
const errorBody = data && typeof data === "object" ? (data as Record<string, unknown>).error : undefined;
|
||||
if (errorBody && typeof errorBody === "object" && typeof (errorBody as Record<string, unknown>).message === "string") {
|
||||
const body = errorBody as Record<string, unknown>;
|
||||
const statusCode = body.statusCode;
|
||||
const label = (typeof statusCode === "number" && HTTP_STATUS_TEXT[statusCode]) || body.code;
|
||||
return typeof statusCode === "number" ? `${statusCode} ${label}: ${body.message}` : (body.message as string);
|
||||
}
|
||||
if (err instanceof Error) return err.message;
|
||||
if (err != null) return String(err);
|
||||
return fallback;
|
||||
}
|
||||
@@ -1,362 +1,362 @@
|
||||
import http from "http";
|
||||
import fs from "fs";
|
||||
import path from "path";
|
||||
|
||||
export function dockerRequest(path: string, method: string, body: any = null): Promise<any> {
|
||||
return new Promise((resolve, reject) => {
|
||||
const options = {
|
||||
socketPath: "/var/run/docker.sock",
|
||||
path: path,
|
||||
method: method,
|
||||
headers: {
|
||||
"Content-Type": "application/json",
|
||||
},
|
||||
};
|
||||
|
||||
const req = http.request(options, (res) => {
|
||||
const chunks: Buffer[] = [];
|
||||
res.on("data", (chunk) => chunks.push(chunk));
|
||||
res.on("end", () => {
|
||||
const resBuffer = Buffer.concat(chunks);
|
||||
const data = resBuffer.toString("utf8");
|
||||
if (res.statusCode && res.statusCode >= 200 && res.statusCode < 300) {
|
||||
try {
|
||||
resolve(data ? JSON.parse(data) : null);
|
||||
} catch (e) {
|
||||
resolve(data);
|
||||
}
|
||||
} else {
|
||||
reject(new Error(`Docker API Error ${res.statusCode}: ${data}`));
|
||||
}
|
||||
});
|
||||
});
|
||||
|
||||
req.on("error", (err) => reject(err));
|
||||
if (body) {
|
||||
req.write(JSON.stringify(body));
|
||||
}
|
||||
req.end();
|
||||
});
|
||||
}
|
||||
|
||||
export function parseDockerStream(buffer: Buffer): { stdout: string; stderr: string } {
|
||||
let stdout = "";
|
||||
let stderr = "";
|
||||
let offset = 0;
|
||||
|
||||
while (offset + 8 <= buffer.length) {
|
||||
const streamType = buffer.readUInt8(offset);
|
||||
const size = buffer.readUInt32BE(offset + 4);
|
||||
|
||||
if (offset + 8 + size > buffer.length) {
|
||||
break;
|
||||
}
|
||||
|
||||
const payload = buffer.toString("utf8", offset + 8, offset + 8 + size);
|
||||
if (streamType === 1) {
|
||||
stdout += payload;
|
||||
} else if (streamType === 2) {
|
||||
stderr += payload;
|
||||
}
|
||||
offset += 8 + size;
|
||||
}
|
||||
|
||||
if (stdout === "" && stderr === "" && buffer.length > 0) {
|
||||
stdout = buffer.toString("utf8");
|
||||
}
|
||||
|
||||
return { stdout, stderr };
|
||||
}
|
||||
|
||||
export function runExec(containerName: string, cmd: string[]): Promise<string> {
|
||||
return new Promise(async (resolve, reject) => {
|
||||
try {
|
||||
const execConfig = {
|
||||
AttachStdout: true,
|
||||
AttachStderr: true,
|
||||
Cmd: cmd,
|
||||
};
|
||||
const createRes = await dockerRequest(`/containers/${containerName}/exec`, "POST", execConfig);
|
||||
const execId = createRes.Id;
|
||||
|
||||
const options = {
|
||||
socketPath: "/var/run/docker.sock",
|
||||
path: `/exec/${execId}/start`,
|
||||
method: "POST",
|
||||
headers: {
|
||||
"Content-Type": "application/json",
|
||||
},
|
||||
};
|
||||
|
||||
const req = http.request(options, (res) => {
|
||||
const chunks: Buffer[] = [];
|
||||
res.on("data", (chunk) => chunks.push(chunk));
|
||||
res.on("end", () => {
|
||||
const streamData = parseDockerStream(Buffer.concat(chunks));
|
||||
resolve(streamData.stdout || streamData.stderr);
|
||||
});
|
||||
});
|
||||
|
||||
req.on("error", (err) => reject(err));
|
||||
req.write(JSON.stringify({ Detach: false, Tty: false }));
|
||||
req.end();
|
||||
} catch (err) {
|
||||
reject(err);
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
export function getProcessName(pid: number): string {
|
||||
try {
|
||||
const commPath = `/proc/${pid}/comm`;
|
||||
if (fs.existsSync(commPath)) {
|
||||
return fs.readFileSync(commPath, "utf8").trim();
|
||||
}
|
||||
} catch (err) {
|
||||
// ignore
|
||||
}
|
||||
return "";
|
||||
}
|
||||
|
||||
export function makeHumanReadableName(procName: string): string {
|
||||
const nameLower = procName.toLowerCase();
|
||||
if (nameLower.includes("rustdesk")) return "RustDesk Remote Desktop";
|
||||
if (nameLower.includes("xorg")) return "Xorg Graphics Server";
|
||||
if (nameLower.includes("vllm") || nameLower.includes("enginecore")) return "vLLM Inference Server";
|
||||
if (nameLower.includes("python")) return "Python / Gradio App";
|
||||
if (nameLower.includes("node")) return "Next.js Web App";
|
||||
if (nameLower.includes("postgres")) return "PostgreSQL Database";
|
||||
if (nameLower.includes("nginx")) return "Nginx Load Balancer";
|
||||
return procName;
|
||||
}
|
||||
|
||||
export async function getGpuInfo(): Promise<any[]> {
|
||||
try {
|
||||
const gpuOutput = await runExec("paddleocr-vllm-server", [
|
||||
"nvidia-smi",
|
||||
"--query-gpu=index,name,utilization.gpu,utilization.memory,memory.total,memory.used,memory.free,uuid",
|
||||
"--format=csv,noheader,nounits",
|
||||
]);
|
||||
|
||||
const gpus: any[] = [];
|
||||
if (gpuOutput) {
|
||||
const lines = gpuOutput.split("\n");
|
||||
for (const line of lines) {
|
||||
if (!line.trim()) continue;
|
||||
const parts = line.split(",").map((p) => p.trim());
|
||||
if (parts.length >= 8) {
|
||||
gpus.push({
|
||||
index: parts[0],
|
||||
name: parts[1],
|
||||
gpu_util: parseInt(parts[2]) || 0,
|
||||
mem_util: parseInt(parts[3]) || 0,
|
||||
mem_total: parseInt(parts[4]) || 0,
|
||||
mem_used: parseInt(parts[5]) || 0,
|
||||
mem_free: parseInt(parts[6]) || 0,
|
||||
uuid: parts[7],
|
||||
processes: [],
|
||||
});
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
const procOutput = await runExec("paddleocr-vllm-server", [
|
||||
"nvidia-smi",
|
||||
"--query-compute-apps=gpu_uuid,pid,process_name,used_memory",
|
||||
"--format=csv,noheader,nounits",
|
||||
]);
|
||||
|
||||
if (procOutput) {
|
||||
const lines = procOutput.split("\n");
|
||||
for (const line of lines) {
|
||||
if (!line.trim()) continue;
|
||||
const parts = line.split(",").map((p) => p.trim());
|
||||
if (parts.length >= 4) {
|
||||
const gpuUuid = parts[0];
|
||||
const pid = parseInt(parts[1]);
|
||||
const procName = parts[2];
|
||||
const usedMem = parseInt(parts[3]);
|
||||
|
||||
const gpu = gpus.find((g) => g.uuid === gpuUuid);
|
||||
if (gpu) {
|
||||
const systemProcName = getProcessName(pid) || procName;
|
||||
gpu.processes.push({
|
||||
pid,
|
||||
name: procName,
|
||||
readable_name: makeHumanReadableName(systemProcName),
|
||||
used_mem: usedMem,
|
||||
});
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return gpus;
|
||||
} catch (err) {
|
||||
console.error("Failed to query GPUs:", err);
|
||||
return [];
|
||||
}
|
||||
}
|
||||
|
||||
export async function getContainerStatus(containerName: string): Promise<string> {
|
||||
try {
|
||||
const info = await dockerRequest(`/containers/${containerName}/json`, "GET");
|
||||
return info.State.Status;
|
||||
} catch (err) {
|
||||
return "stopped";
|
||||
}
|
||||
}
|
||||
|
||||
export async function manageContainer(containerName: string, action: "start" | "stop" | "restart"): Promise<void> {
|
||||
await dockerRequest(`/containers/${containerName}/${action}`, "POST");
|
||||
}
|
||||
|
||||
export async function recreateContainer(containerName: string, newCudaDevices?: string): Promise<void> {
|
||||
const inspect = await dockerRequest(`/containers/${containerName}/json`, "GET");
|
||||
|
||||
try {
|
||||
await dockerRequest(`/containers/${containerName}/stop`, "POST");
|
||||
} catch (e) {
|
||||
// ignore
|
||||
}
|
||||
|
||||
const rand = Math.floor(Math.random() * 10000);
|
||||
const oldTempName = `${containerName}_old_${rand}`;
|
||||
await dockerRequest(`/containers/${containerName}/rename?name=${oldTempName}`, "POST");
|
||||
|
||||
const config: any = {
|
||||
...inspect.Config,
|
||||
HostConfig: inspect.HostConfig,
|
||||
NetworkingConfig: {
|
||||
EndpointsConfig: inspect.NetworkSettings.Networks,
|
||||
},
|
||||
};
|
||||
|
||||
// Ensure Name is not copied from Inspect root as it's not a field in Create
|
||||
delete config.Name;
|
||||
|
||||
if (newCudaDevices && config.Env) {
|
||||
config.Env = config.Env.map((envStr: string) => {
|
||||
if (envStr.startsWith("CUDA_VISIBLE_DEVICES=")) {
|
||||
return `CUDA_VISIBLE_DEVICES=${newCudaDevices}`;
|
||||
}
|
||||
return envStr;
|
||||
});
|
||||
}
|
||||
|
||||
const createRes = await dockerRequest(`/containers/create?name=${containerName}`, "POST", config);
|
||||
const newId = createRes.Id;
|
||||
|
||||
await dockerRequest(`/containers/${newId}/start`, "POST");
|
||||
|
||||
try {
|
||||
await dockerRequest(`/containers/${oldTempName}`, "DELETE");
|
||||
} catch (e) {
|
||||
// ignore
|
||||
}
|
||||
}
|
||||
|
||||
export async function getEnvSettings(): Promise<{ cuda_devices: string }> {
|
||||
const envPath = path.join(process.cwd(), "..", ".env");
|
||||
const settings = { cuda_devices: "1" };
|
||||
try {
|
||||
if (fs.existsSync(envPath)) {
|
||||
const content = fs.readFileSync(envPath, "utf8");
|
||||
const lines = content.split("\n");
|
||||
for (const line of lines) {
|
||||
const trimmed = line.trim();
|
||||
if (!trimmed || trimmed.startsWith("#")) continue;
|
||||
const [k, v] = trimmed.split("=");
|
||||
if (k && k.trim() === "CUDA_VISIBLE_DEVICES" && v) {
|
||||
settings.cuda_devices = v.trim().replace(/['"]/g, "");
|
||||
}
|
||||
}
|
||||
}
|
||||
} catch (err) {
|
||||
console.error("Failed to read env settings:", err);
|
||||
}
|
||||
return settings;
|
||||
}
|
||||
|
||||
export async function saveEnvSettings(cuda_devices: string): Promise<void> {
|
||||
const envPath = path.join(process.cwd(), "..", ".env");
|
||||
try {
|
||||
let lines: string[] = [];
|
||||
if (fs.existsSync(envPath)) {
|
||||
lines = fs.readFileSync(envPath, "utf8").split("\n");
|
||||
}
|
||||
|
||||
let found = false;
|
||||
const newLines = lines.map((line) => {
|
||||
if (line.trim().startsWith("CUDA_VISIBLE_DEVICES=")) {
|
||||
found = true;
|
||||
return `CUDA_VISIBLE_DEVICES=${cuda_devices}`;
|
||||
}
|
||||
return line;
|
||||
});
|
||||
|
||||
if (!found) {
|
||||
newLines.push(`CUDA_VISIBLE_DEVICES=${cuda_devices}`);
|
||||
}
|
||||
|
||||
fs.writeFileSync(envPath, newLines.join("\n"), "utf8");
|
||||
} catch (err) {
|
||||
console.error("Failed to save env settings:", err);
|
||||
throw err;
|
||||
}
|
||||
}
|
||||
|
||||
export async function unloadOtherEngines(): Promise<{ stopped: string[]; failed: string[] }> {
|
||||
const stopped: string[] = [];
|
||||
const failed: string[] = [];
|
||||
|
||||
try {
|
||||
const containers = await dockerRequest("/containers/json", "GET");
|
||||
if (!Array.isArray(containers)) {
|
||||
throw new Error("Invalid response from Docker API: expected container array.");
|
||||
}
|
||||
|
||||
const stopPromises: Promise<void>[] = [];
|
||||
|
||||
for (const container of containers) {
|
||||
if (!container.Names || !Array.isArray(container.Names)) continue;
|
||||
|
||||
const rawName = container.Names[0] || "";
|
||||
const name = rawName.startsWith("/") ? rawName.slice(1) : rawName;
|
||||
const nameLower = name.toLowerCase();
|
||||
|
||||
const matchesEngine =
|
||||
nameLower.includes("lighton") ||
|
||||
nameLower.includes("glm") ||
|
||||
nameLower.includes("dots") ||
|
||||
nameLower.includes("deepseek");
|
||||
|
||||
const isExcluded =
|
||||
nameLower.includes("paddleocr") ||
|
||||
nameLower.includes("nemotron");
|
||||
|
||||
if (matchesEngine && !isExcluded) {
|
||||
console.log(`Queueing unload for container: ${name} (${container.Id})`);
|
||||
|
||||
const stopPromise = dockerRequest(`/containers/${container.Id}/stop`, "POST")
|
||||
.then(() => {
|
||||
stopped.push(name);
|
||||
})
|
||||
.catch((err) => {
|
||||
console.error(`Failed to stop container ${name}:`, err);
|
||||
failed.push(`${name} (${err.message})`);
|
||||
});
|
||||
|
||||
stopPromises.push(stopPromise);
|
||||
}
|
||||
}
|
||||
|
||||
await Promise.all(stopPromises);
|
||||
} catch (err: any) {
|
||||
console.error("Failed to unload other engines:", err);
|
||||
throw err;
|
||||
}
|
||||
|
||||
return { stopped, failed };
|
||||
}
|
||||
|
||||
import http from "http";
|
||||
import fs from "fs";
|
||||
import path from "path";
|
||||
|
||||
export function dockerRequest(path: string, method: string, body: any = null): Promise<any> {
|
||||
return new Promise((resolve, reject) => {
|
||||
const options = {
|
||||
socketPath: "/var/run/docker.sock",
|
||||
path: path,
|
||||
method: method,
|
||||
headers: {
|
||||
"Content-Type": "application/json",
|
||||
},
|
||||
};
|
||||
|
||||
const req = http.request(options, (res) => {
|
||||
const chunks: Buffer[] = [];
|
||||
res.on("data", (chunk) => chunks.push(chunk));
|
||||
res.on("end", () => {
|
||||
const resBuffer = Buffer.concat(chunks);
|
||||
const data = resBuffer.toString("utf8");
|
||||
if (res.statusCode && res.statusCode >= 200 && res.statusCode < 300) {
|
||||
try {
|
||||
resolve(data ? JSON.parse(data) : null);
|
||||
} catch (e) {
|
||||
resolve(data);
|
||||
}
|
||||
} else {
|
||||
reject(new Error(`Docker API Error ${res.statusCode}: ${data}`));
|
||||
}
|
||||
});
|
||||
});
|
||||
|
||||
req.on("error", (err) => reject(err));
|
||||
if (body) {
|
||||
req.write(JSON.stringify(body));
|
||||
}
|
||||
req.end();
|
||||
});
|
||||
}
|
||||
|
||||
export function parseDockerStream(buffer: Buffer): { stdout: string; stderr: string } {
|
||||
let stdout = "";
|
||||
let stderr = "";
|
||||
let offset = 0;
|
||||
|
||||
while (offset + 8 <= buffer.length) {
|
||||
const streamType = buffer.readUInt8(offset);
|
||||
const size = buffer.readUInt32BE(offset + 4);
|
||||
|
||||
if (offset + 8 + size > buffer.length) {
|
||||
break;
|
||||
}
|
||||
|
||||
const payload = buffer.toString("utf8", offset + 8, offset + 8 + size);
|
||||
if (streamType === 1) {
|
||||
stdout += payload;
|
||||
} else if (streamType === 2) {
|
||||
stderr += payload;
|
||||
}
|
||||
offset += 8 + size;
|
||||
}
|
||||
|
||||
if (stdout === "" && stderr === "" && buffer.length > 0) {
|
||||
stdout = buffer.toString("utf8");
|
||||
}
|
||||
|
||||
return { stdout, stderr };
|
||||
}
|
||||
|
||||
export function runExec(containerName: string, cmd: string[]): Promise<string> {
|
||||
return new Promise(async (resolve, reject) => {
|
||||
try {
|
||||
const execConfig = {
|
||||
AttachStdout: true,
|
||||
AttachStderr: true,
|
||||
Cmd: cmd,
|
||||
};
|
||||
const createRes = await dockerRequest(`/containers/${containerName}/exec`, "POST", execConfig);
|
||||
const execId = createRes.Id;
|
||||
|
||||
const options = {
|
||||
socketPath: "/var/run/docker.sock",
|
||||
path: `/exec/${execId}/start`,
|
||||
method: "POST",
|
||||
headers: {
|
||||
"Content-Type": "application/json",
|
||||
},
|
||||
};
|
||||
|
||||
const req = http.request(options, (res) => {
|
||||
const chunks: Buffer[] = [];
|
||||
res.on("data", (chunk) => chunks.push(chunk));
|
||||
res.on("end", () => {
|
||||
const streamData = parseDockerStream(Buffer.concat(chunks));
|
||||
resolve(streamData.stdout || streamData.stderr);
|
||||
});
|
||||
});
|
||||
|
||||
req.on("error", (err) => reject(err));
|
||||
req.write(JSON.stringify({ Detach: false, Tty: false }));
|
||||
req.end();
|
||||
} catch (err) {
|
||||
reject(err);
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
export function getProcessName(pid: number): string {
|
||||
try {
|
||||
const commPath = `/proc/${pid}/comm`;
|
||||
if (fs.existsSync(commPath)) {
|
||||
return fs.readFileSync(commPath, "utf8").trim();
|
||||
}
|
||||
} catch (err) {
|
||||
// ignore
|
||||
}
|
||||
return "";
|
||||
}
|
||||
|
||||
export function makeHumanReadableName(procName: string): string {
|
||||
const nameLower = procName.toLowerCase();
|
||||
if (nameLower.includes("rustdesk")) return "RustDesk Remote Desktop";
|
||||
if (nameLower.includes("xorg")) return "Xorg Graphics Server";
|
||||
if (nameLower.includes("vllm") || nameLower.includes("enginecore")) return "vLLM Inference Server";
|
||||
if (nameLower.includes("python")) return "Python / Gradio App";
|
||||
if (nameLower.includes("node")) return "Next.js Web App";
|
||||
if (nameLower.includes("postgres")) return "PostgreSQL Database";
|
||||
if (nameLower.includes("nginx")) return "Nginx Load Balancer";
|
||||
return procName;
|
||||
}
|
||||
|
||||
export async function getGpuInfo(): Promise<any[]> {
|
||||
try {
|
||||
const gpuOutput = await runExec("paddleocr-vllm-server", [
|
||||
"nvidia-smi",
|
||||
"--query-gpu=index,name,utilization.gpu,utilization.memory,memory.total,memory.used,memory.free,uuid",
|
||||
"--format=csv,noheader,nounits",
|
||||
]);
|
||||
|
||||
const gpus: any[] = [];
|
||||
if (gpuOutput) {
|
||||
const lines = gpuOutput.split("\n");
|
||||
for (const line of lines) {
|
||||
if (!line.trim()) continue;
|
||||
const parts = line.split(",").map((p) => p.trim());
|
||||
if (parts.length >= 8) {
|
||||
gpus.push({
|
||||
index: parts[0],
|
||||
name: parts[1],
|
||||
gpu_util: parseInt(parts[2]) || 0,
|
||||
mem_util: parseInt(parts[3]) || 0,
|
||||
mem_total: parseInt(parts[4]) || 0,
|
||||
mem_used: parseInt(parts[5]) || 0,
|
||||
mem_free: parseInt(parts[6]) || 0,
|
||||
uuid: parts[7],
|
||||
processes: [],
|
||||
});
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
const procOutput = await runExec("paddleocr-vllm-server", [
|
||||
"nvidia-smi",
|
||||
"--query-compute-apps=gpu_uuid,pid,process_name,used_memory",
|
||||
"--format=csv,noheader,nounits",
|
||||
]);
|
||||
|
||||
if (procOutput) {
|
||||
const lines = procOutput.split("\n");
|
||||
for (const line of lines) {
|
||||
if (!line.trim()) continue;
|
||||
const parts = line.split(",").map((p) => p.trim());
|
||||
if (parts.length >= 4) {
|
||||
const gpuUuid = parts[0];
|
||||
const pid = parseInt(parts[1]);
|
||||
const procName = parts[2];
|
||||
const usedMem = parseInt(parts[3]);
|
||||
|
||||
const gpu = gpus.find((g) => g.uuid === gpuUuid);
|
||||
if (gpu) {
|
||||
const systemProcName = getProcessName(pid) || procName;
|
||||
gpu.processes.push({
|
||||
pid,
|
||||
name: procName,
|
||||
readable_name: makeHumanReadableName(systemProcName),
|
||||
used_mem: usedMem,
|
||||
});
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return gpus;
|
||||
} catch (err) {
|
||||
console.error("Failed to query GPUs:", err);
|
||||
return [];
|
||||
}
|
||||
}
|
||||
|
||||
export async function getContainerStatus(containerName: string): Promise<string> {
|
||||
try {
|
||||
const info = await dockerRequest(`/containers/${containerName}/json`, "GET");
|
||||
return info.State.Status;
|
||||
} catch (err) {
|
||||
return "stopped";
|
||||
}
|
||||
}
|
||||
|
||||
export async function manageContainer(containerName: string, action: "start" | "stop" | "restart"): Promise<void> {
|
||||
await dockerRequest(`/containers/${containerName}/${action}`, "POST");
|
||||
}
|
||||
|
||||
export async function recreateContainer(containerName: string, newCudaDevices?: string): Promise<void> {
|
||||
const inspect = await dockerRequest(`/containers/${containerName}/json`, "GET");
|
||||
|
||||
try {
|
||||
await dockerRequest(`/containers/${containerName}/stop`, "POST");
|
||||
} catch (e) {
|
||||
// ignore
|
||||
}
|
||||
|
||||
const rand = Math.floor(Math.random() * 10000);
|
||||
const oldTempName = `${containerName}_old_${rand}`;
|
||||
await dockerRequest(`/containers/${containerName}/rename?name=${oldTempName}`, "POST");
|
||||
|
||||
const config: any = {
|
||||
...inspect.Config,
|
||||
HostConfig: inspect.HostConfig,
|
||||
NetworkingConfig: {
|
||||
EndpointsConfig: inspect.NetworkSettings.Networks,
|
||||
},
|
||||
};
|
||||
|
||||
// Ensure Name is not copied from Inspect root as it's not a field in Create
|
||||
delete config.Name;
|
||||
|
||||
if (newCudaDevices && config.Env) {
|
||||
config.Env = config.Env.map((envStr: string) => {
|
||||
if (envStr.startsWith("CUDA_VISIBLE_DEVICES=")) {
|
||||
return `CUDA_VISIBLE_DEVICES=${newCudaDevices}`;
|
||||
}
|
||||
return envStr;
|
||||
});
|
||||
}
|
||||
|
||||
const createRes = await dockerRequest(`/containers/create?name=${containerName}`, "POST", config);
|
||||
const newId = createRes.Id;
|
||||
|
||||
await dockerRequest(`/containers/${newId}/start`, "POST");
|
||||
|
||||
try {
|
||||
await dockerRequest(`/containers/${oldTempName}`, "DELETE");
|
||||
} catch (e) {
|
||||
// ignore
|
||||
}
|
||||
}
|
||||
|
||||
export async function getEnvSettings(): Promise<{ cuda_devices: string }> {
|
||||
const envPath = path.join(process.cwd(), "..", ".env");
|
||||
const settings = { cuda_devices: "1" };
|
||||
try {
|
||||
if (fs.existsSync(envPath)) {
|
||||
const content = fs.readFileSync(envPath, "utf8");
|
||||
const lines = content.split("\n");
|
||||
for (const line of lines) {
|
||||
const trimmed = line.trim();
|
||||
if (!trimmed || trimmed.startsWith("#")) continue;
|
||||
const [k, v] = trimmed.split("=");
|
||||
if (k && k.trim() === "CUDA_VISIBLE_DEVICES" && v) {
|
||||
settings.cuda_devices = v.trim().replace(/['"]/g, "");
|
||||
}
|
||||
}
|
||||
}
|
||||
} catch (err) {
|
||||
console.error("Failed to read env settings:", err);
|
||||
}
|
||||
return settings;
|
||||
}
|
||||
|
||||
export async function saveEnvSettings(cuda_devices: string): Promise<void> {
|
||||
const envPath = path.join(process.cwd(), "..", ".env");
|
||||
try {
|
||||
let lines: string[] = [];
|
||||
if (fs.existsSync(envPath)) {
|
||||
lines = fs.readFileSync(envPath, "utf8").split("\n");
|
||||
}
|
||||
|
||||
let found = false;
|
||||
const newLines = lines.map((line) => {
|
||||
if (line.trim().startsWith("CUDA_VISIBLE_DEVICES=")) {
|
||||
found = true;
|
||||
return `CUDA_VISIBLE_DEVICES=${cuda_devices}`;
|
||||
}
|
||||
return line;
|
||||
});
|
||||
|
||||
if (!found) {
|
||||
newLines.push(`CUDA_VISIBLE_DEVICES=${cuda_devices}`);
|
||||
}
|
||||
|
||||
fs.writeFileSync(envPath, newLines.join("\n"), "utf8");
|
||||
} catch (err) {
|
||||
console.error("Failed to save env settings:", err);
|
||||
throw err;
|
||||
}
|
||||
}
|
||||
|
||||
export async function unloadOtherEngines(): Promise<{ stopped: string[]; failed: string[] }> {
|
||||
const stopped: string[] = [];
|
||||
const failed: string[] = [];
|
||||
|
||||
try {
|
||||
const containers = await dockerRequest("/containers/json", "GET");
|
||||
if (!Array.isArray(containers)) {
|
||||
throw new Error("Invalid response from Docker API: expected container array.");
|
||||
}
|
||||
|
||||
const stopPromises: Promise<void>[] = [];
|
||||
|
||||
for (const container of containers) {
|
||||
if (!container.Names || !Array.isArray(container.Names)) continue;
|
||||
|
||||
const rawName = container.Names[0] || "";
|
||||
const name = rawName.startsWith("/") ? rawName.slice(1) : rawName;
|
||||
const nameLower = name.toLowerCase();
|
||||
|
||||
const matchesEngine =
|
||||
nameLower.includes("lighton") ||
|
||||
nameLower.includes("glm") ||
|
||||
nameLower.includes("dots") ||
|
||||
nameLower.includes("deepseek");
|
||||
|
||||
const isExcluded =
|
||||
nameLower.includes("paddleocr") ||
|
||||
nameLower.includes("nemotron");
|
||||
|
||||
if (matchesEngine && !isExcluded) {
|
||||
console.log(`Queueing unload for container: ${name} (${container.Id})`);
|
||||
|
||||
const stopPromise = dockerRequest(`/containers/${container.Id}/stop`, "POST")
|
||||
.then(() => {
|
||||
stopped.push(name);
|
||||
})
|
||||
.catch((err) => {
|
||||
console.error(`Failed to stop container ${name}:`, err);
|
||||
failed.push(`${name} (${err.message})`);
|
||||
});
|
||||
|
||||
stopPromises.push(stopPromise);
|
||||
}
|
||||
}
|
||||
|
||||
await Promise.all(stopPromises);
|
||||
} catch (err: any) {
|
||||
console.error("Failed to unload other engines:", err);
|
||||
throw err;
|
||||
}
|
||||
|
||||
return { stopped, failed };
|
||||
}
|
||||
|
||||
@@ -1,117 +1,117 @@
|
||||
export type ParseStatus = "pending" | "done" | "failed";
|
||||
|
||||
export interface DocumentRow {
|
||||
id: number;
|
||||
filename: string;
|
||||
upload_time: Date;
|
||||
parsed: boolean;
|
||||
metadata: any;
|
||||
latitude: any;
|
||||
longitude: any;
|
||||
scan_mode: string | null;
|
||||
parse_error: string | null;
|
||||
confirmed: boolean;
|
||||
}
|
||||
|
||||
export interface OcrItemRow {
|
||||
kode_barang: string | null;
|
||||
nama_barang: string | null;
|
||||
banyak: string | null;
|
||||
jumlah: string | null;
|
||||
}
|
||||
|
||||
// Shared by GET /api/v1/documents (list), GET /api/v1/documents/:id, and the
|
||||
// upload route's dedup-return branch, so the header/shipment/status mapping
|
||||
// only lives in one place.
|
||||
export function mapDocumentRow(doc: DocumentRow, itemRows: OcrItemRow[]) {
|
||||
const metadata = doc.metadata || {};
|
||||
|
||||
const items = itemRows.map((item) => ({
|
||||
nomor_sku: item.kode_barang || "",
|
||||
nama_barang: item.nama_barang || "",
|
||||
banyak: item.banyak || "",
|
||||
jumlah: item.jumlah || ""
|
||||
}));
|
||||
|
||||
let header = {
|
||||
tanggal: "",
|
||||
no_po: "",
|
||||
no_so: "",
|
||||
no_do: ""
|
||||
};
|
||||
|
||||
let shipment = {
|
||||
kepada_yth: "",
|
||||
order_untuk: "",
|
||||
alamat: "",
|
||||
plat_truk: "",
|
||||
nama_driver: "",
|
||||
nama_penerima: ""
|
||||
};
|
||||
|
||||
if (metadata.header) {
|
||||
// Document was updated via mobile app
|
||||
header = {
|
||||
tanggal: metadata.header.tanggal || "",
|
||||
no_po: metadata.header.no_po || "",
|
||||
no_so: metadata.header.no_so || "",
|
||||
no_do: metadata.header.no_do || ""
|
||||
};
|
||||
shipment = {
|
||||
kepada_yth: metadata.shipment?.kepada_yth || "",
|
||||
order_untuk: metadata.shipment?.order_untuk || "",
|
||||
alamat: metadata.shipment?.alamat || "",
|
||||
plat_truk: metadata.shipment?.plat_truk || "",
|
||||
nama_driver: metadata.shipment?.nama_driver || "",
|
||||
nama_penerima: metadata.shipment?.nama_penerima || ""
|
||||
};
|
||||
} else {
|
||||
// Document was freshly uploaded / parsed via web
|
||||
header = {
|
||||
tanggal: metadata.tanggal || "",
|
||||
no_po: metadata.noPO || "",
|
||||
no_so: metadata.noSO || "",
|
||||
no_do: metadata.noDO || doc.filename || ""
|
||||
};
|
||||
shipment = {
|
||||
kepada_yth: metadata.customerInfo || "",
|
||||
order_untuk: metadata.orderUntuk || "",
|
||||
alamat: metadata.alamat || "",
|
||||
plat_truk: metadata.platTruk || "",
|
||||
nama_driver: "",
|
||||
nama_penerima: metadata.headerRemark || ""
|
||||
};
|
||||
}
|
||||
|
||||
const parseStatus: ParseStatus = doc.parsed
|
||||
? "done"
|
||||
: doc.parse_error
|
||||
? "failed"
|
||||
: "pending";
|
||||
|
||||
// scan_mode is the source of truth once persisted (task 9.1); fall back to the
|
||||
// legacy metadata sentinel for rows created before that column existed.
|
||||
const docType = doc.scan_mode || (shipment.order_untuk === "PRODUCT SCAN" ? "Product" : "DO");
|
||||
|
||||
return {
|
||||
id: doc.id.toString(),
|
||||
filePath: doc.filename,
|
||||
createdAt: doc.upload_time.toISOString(),
|
||||
header,
|
||||
shipment,
|
||||
items,
|
||||
parsed: doc.parsed,
|
||||
latitude: doc.latitude ? parseFloat(doc.latitude.toString()) : null,
|
||||
longitude: doc.longitude ? parseFloat(doc.longitude.toString()) : null,
|
||||
parseStatus,
|
||||
docType,
|
||||
confirmed: doc.confirmed,
|
||||
// Full classify+OCR result captured at upload time for Product Scan
|
||||
// documents (gap G3) - lets the editor render immediately instead of
|
||||
// re-running the GPU pipeline on review. `null` for DO documents, and
|
||||
// for Product documents parsed before this existed or already PUT
|
||||
// (the PUT route rebuilds `metadata` from scratch without this key,
|
||||
// which is fine - the editor only needs it during the initial review).
|
||||
productScan: metadata.productScan || null
|
||||
};
|
||||
}
|
||||
export type ParseStatus = "pending" | "done" | "failed";
|
||||
|
||||
export interface DocumentRow {
|
||||
id: number;
|
||||
filename: string;
|
||||
upload_time: Date;
|
||||
parsed: boolean;
|
||||
metadata: any;
|
||||
latitude: any;
|
||||
longitude: any;
|
||||
scan_mode: string | null;
|
||||
parse_error: string | null;
|
||||
confirmed: boolean;
|
||||
}
|
||||
|
||||
export interface OcrItemRow {
|
||||
kode_barang: string | null;
|
||||
nama_barang: string | null;
|
||||
banyak: string | null;
|
||||
jumlah: string | null;
|
||||
}
|
||||
|
||||
// Shared by GET /api/v1/documents (list), GET /api/v1/documents/:id, and the
|
||||
// upload route's dedup-return branch, so the header/shipment/status mapping
|
||||
// only lives in one place.
|
||||
export function mapDocumentRow(doc: DocumentRow, itemRows: OcrItemRow[]) {
|
||||
const metadata = doc.metadata || {};
|
||||
|
||||
const items = itemRows.map((item) => ({
|
||||
nomor_sku: item.kode_barang || "",
|
||||
nama_barang: item.nama_barang || "",
|
||||
banyak: item.banyak || "",
|
||||
jumlah: item.jumlah || ""
|
||||
}));
|
||||
|
||||
let header = {
|
||||
tanggal: "",
|
||||
no_po: "",
|
||||
no_so: "",
|
||||
no_do: ""
|
||||
};
|
||||
|
||||
let shipment = {
|
||||
kepada_yth: "",
|
||||
order_untuk: "",
|
||||
alamat: "",
|
||||
plat_truk: "",
|
||||
nama_driver: "",
|
||||
nama_penerima: ""
|
||||
};
|
||||
|
||||
if (metadata.header) {
|
||||
// Document was updated via mobile app
|
||||
header = {
|
||||
tanggal: metadata.header.tanggal || "",
|
||||
no_po: metadata.header.no_po || "",
|
||||
no_so: metadata.header.no_so || "",
|
||||
no_do: metadata.header.no_do || ""
|
||||
};
|
||||
shipment = {
|
||||
kepada_yth: metadata.shipment?.kepada_yth || "",
|
||||
order_untuk: metadata.shipment?.order_untuk || "",
|
||||
alamat: metadata.shipment?.alamat || "",
|
||||
plat_truk: metadata.shipment?.plat_truk || "",
|
||||
nama_driver: metadata.shipment?.nama_driver || "",
|
||||
nama_penerima: metadata.shipment?.nama_penerima || ""
|
||||
};
|
||||
} else {
|
||||
// Document was freshly uploaded / parsed via web
|
||||
header = {
|
||||
tanggal: metadata.tanggal || "",
|
||||
no_po: metadata.noPO || "",
|
||||
no_so: metadata.noSO || "",
|
||||
no_do: metadata.noDO || doc.filename || ""
|
||||
};
|
||||
shipment = {
|
||||
kepada_yth: metadata.customerInfo || "",
|
||||
order_untuk: metadata.orderUntuk || "",
|
||||
alamat: metadata.alamat || "",
|
||||
plat_truk: metadata.platTruk || "",
|
||||
nama_driver: "",
|
||||
nama_penerima: metadata.headerRemark || ""
|
||||
};
|
||||
}
|
||||
|
||||
const parseStatus: ParseStatus = doc.parsed
|
||||
? "done"
|
||||
: doc.parse_error
|
||||
? "failed"
|
||||
: "pending";
|
||||
|
||||
// scan_mode is the source of truth once persisted (task 9.1); fall back to the
|
||||
// legacy metadata sentinel for rows created before that column existed.
|
||||
const docType = doc.scan_mode || (shipment.order_untuk === "PRODUCT SCAN" ? "Product" : "DO");
|
||||
|
||||
return {
|
||||
id: doc.id.toString(),
|
||||
filePath: doc.filename,
|
||||
createdAt: doc.upload_time.toISOString(),
|
||||
header,
|
||||
shipment,
|
||||
items,
|
||||
parsed: doc.parsed,
|
||||
latitude: doc.latitude ? parseFloat(doc.latitude.toString()) : null,
|
||||
longitude: doc.longitude ? parseFloat(doc.longitude.toString()) : null,
|
||||
parseStatus,
|
||||
docType,
|
||||
confirmed: doc.confirmed,
|
||||
// Full classify+OCR result captured at upload time for Product Scan
|
||||
// documents (gap G3) - lets the editor render immediately instead of
|
||||
// re-running the GPU pipeline on review. `null` for DO documents, and
|
||||
// for Product documents parsed before this existed or already PUT
|
||||
// (the PUT route rebuilds `metadata` from scratch without this key,
|
||||
// which is fine - the editor only needs it during the initial review).
|
||||
productScan: metadata.productScan || null
|
||||
};
|
||||
}
|
||||
Loaded 100 of 315 files, more files were not shown because too many files have changed in this diff.
Show more
Reference in new issue
Block a user