feat: consolidate backend and docker-compose setup
No files matched your search
@@ -0,0 +1,50 @@
|
||||
# Miscellaneous
|
||||
*.class
|
||||
*.log
|
||||
*.pyc
|
||||
*.swp
|
||||
.DS_Store
|
||||
.atom/
|
||||
.build/
|
||||
.buildlog/
|
||||
.history
|
||||
.svn/
|
||||
.swiftpm/
|
||||
migrate_working_dir/
|
||||
|
||||
# IntelliJ related
|
||||
*.iml
|
||||
*.ipr
|
||||
*.iws
|
||||
.idea/
|
||||
|
||||
# The .vscode folder contains launch configuration and tasks you configure in
|
||||
# VS Code which you may wish to be included in version control, so this line
|
||||
# is commented out by default.
|
||||
#.vscode/
|
||||
|
||||
# Flutter/Dart/Pub related
|
||||
**/doc/api/
|
||||
**/ios/Flutter/.last_build_id
|
||||
.dart_tool/
|
||||
.flutter-plugins-dependencies
|
||||
.pub-cache/
|
||||
.pub/
|
||||
/build/
|
||||
/coverage/
|
||||
|
||||
# Symbolication related
|
||||
app.*.symbols
|
||||
|
||||
# Obfuscation related
|
||||
app.*.map.json
|
||||
|
||||
# Android Studio will place build artifacts here
|
||||
/android/app/debug
|
||||
/android/app/profile
|
||||
/android/app/release
|
||||
|
||||
# Node, Next.js, and general JS ignores
|
||||
**/node_modules/
|
||||
**/.next/
|
||||
**/dist/
|
||||
@@ -0,0 +1,45 @@
|
||||
# This file tracks properties of this Flutter project.
|
||||
# Used by Flutter tool to assess capabilities and perform upgrades etc.
|
||||
#
|
||||
# This file should be version controlled and should not be manually edited.
|
||||
|
||||
version:
|
||||
revision: "ad70ec4617166f1c38e5d2bfd388af71fda14f06"
|
||||
channel: "stable"
|
||||
|
||||
project_type: app
|
||||
|
||||
# Tracks metadata for the flutter migrate command
|
||||
migration:
|
||||
platforms:
|
||||
- platform: root
|
||||
create_revision: ad70ec4617166f1c38e5d2bfd388af71fda14f06
|
||||
base_revision: ad70ec4617166f1c38e5d2bfd388af71fda14f06
|
||||
- platform: android
|
||||
create_revision: ad70ec4617166f1c38e5d2bfd388af71fda14f06
|
||||
base_revision: ad70ec4617166f1c38e5d2bfd388af71fda14f06
|
||||
- platform: ios
|
||||
create_revision: ad70ec4617166f1c38e5d2bfd388af71fda14f06
|
||||
base_revision: ad70ec4617166f1c38e5d2bfd388af71fda14f06
|
||||
- platform: linux
|
||||
create_revision: ad70ec4617166f1c38e5d2bfd388af71fda14f06
|
||||
base_revision: ad70ec4617166f1c38e5d2bfd388af71fda14f06
|
||||
- platform: macos
|
||||
create_revision: ad70ec4617166f1c38e5d2bfd388af71fda14f06
|
||||
base_revision: ad70ec4617166f1c38e5d2bfd388af71fda14f06
|
||||
- platform: web
|
||||
create_revision: ad70ec4617166f1c38e5d2bfd388af71fda14f06
|
||||
base_revision: ad70ec4617166f1c38e5d2bfd388af71fda14f06
|
||||
- platform: windows
|
||||
create_revision: ad70ec4617166f1c38e5d2bfd388af71fda14f06
|
||||
base_revision: ad70ec4617166f1c38e5d2bfd388af71fda14f06
|
||||
|
||||
# User provided section
|
||||
|
||||
# List of Local paths (relative to this file) that should be
|
||||
# ignored by the migrate tool.
|
||||
#
|
||||
# Files that are not part of the templates will be ignored by default.
|
||||
unmanaged_files:
|
||||
- 'lib/main.dart'
|
||||
- 'ios/Runner.xcodeproj/project.pbxproj'
|
||||
@@ -0,0 +1,17 @@
|
||||
# app_pfm_ocr_v2
|
||||
|
||||
A new Flutter project.
|
||||
|
||||
## Getting Started
|
||||
|
||||
This project is a starting point for a Flutter application.
|
||||
|
||||
A few resources to get you started if this is your first Flutter project:
|
||||
|
||||
- [Learn Flutter](https://docs.flutter.dev/get-started/learn-flutter)
|
||||
- [Write your first Flutter app](https://docs.flutter.dev/get-started/codelab)
|
||||
- [Flutter learning resources](https://docs.flutter.dev/reference/learning-resources)
|
||||
|
||||
For help getting started with Flutter development, view the
|
||||
[online documentation](https://docs.flutter.dev/), which offers tutorials,
|
||||
samples, guidance on mobile development, and a full API reference.
|
||||
@@ -0,0 +1,28 @@
|
||||
# This file configures the analyzer, which statically analyzes Dart code to
|
||||
# check for errors, warnings, and lints.
|
||||
#
|
||||
# The issues identified by the analyzer are surfaced in the UI of Dart-enabled
|
||||
# IDEs (https://dart.dev/tools#ides-and-editors). The analyzer can also be
|
||||
# invoked from the command line by running `flutter analyze`.
|
||||
|
||||
# The following line activates a set of recommended lints for Flutter apps,
|
||||
# packages, and plugins designed to encourage good coding practices.
|
||||
include: package:flutter_lints/flutter.yaml
|
||||
|
||||
linter:
|
||||
# The lint rules applied to this project can be customized in the
|
||||
# section below to disable rules from the `package:flutter_lints/flutter.yaml`
|
||||
# included above or to enable additional rules. A list of all available lints
|
||||
# and their documentation is published at https://dart.dev/lints.
|
||||
#
|
||||
# Instead of disabling a lint rule for the entire project in the
|
||||
# section below, it can also be suppressed for a single line of code
|
||||
# or a specific dart file by using the `// ignore: name_of_lint` and
|
||||
# `// ignore_for_file: name_of_lint` syntax on the line or in the file
|
||||
# producing the lint.
|
||||
rules:
|
||||
# avoid_print: false # Uncomment to disable the `avoid_print` rule
|
||||
# prefer_single_quotes: true # Uncomment to enable the `prefer_single_quotes` rule
|
||||
|
||||
# Additional information about this file can be found at
|
||||
# https://dart.dev/guides/language/analysis-options
|
||||
@@ -0,0 +1,14 @@
|
||||
gradle-wrapper.jar
|
||||
/.gradle
|
||||
/captures/
|
||||
/gradlew
|
||||
/gradlew.bat
|
||||
/local.properties
|
||||
GeneratedPluginRegistrant.java
|
||||
.cxx/
|
||||
|
||||
# Remember to never publicly share your keystore.
|
||||
# See https://flutter.dev/to/reference-keystore
|
||||
key.properties
|
||||
**/*.keystore
|
||||
**/*.jks
|
||||
@@ -0,0 +1,2 @@
|
||||
#Fri Jun 26 15:57:59 WIB 2026
|
||||
java.home=D\:\\Android\\jbr
|
||||
@@ -0,0 +1,45 @@
|
||||
plugins {
|
||||
id("com.android.application")
|
||||
// The Flutter Gradle Plugin must be applied after the Android and Kotlin Gradle plugins.
|
||||
id("dev.flutter.flutter-gradle-plugin")
|
||||
}
|
||||
|
||||
android {
|
||||
namespace = "com.databisnis.app_pfm_ocr_v2"
|
||||
compileSdk = flutter.compileSdkVersion
|
||||
ndkVersion = flutter.ndkVersion
|
||||
|
||||
compileOptions {
|
||||
sourceCompatibility = JavaVersion.VERSION_17
|
||||
targetCompatibility = JavaVersion.VERSION_17
|
||||
}
|
||||
|
||||
defaultConfig {
|
||||
// TODO: Specify your own unique Application ID (https://developer.android.com/studio/build/application-id.html).
|
||||
applicationId = "com.databisnis.app_pfm_ocr_v2"
|
||||
// You can update the following values to match your application needs.
|
||||
// For more information, see: https://flutter.dev/to/review-gradle-config.
|
||||
minSdk = flutter.minSdkVersion
|
||||
targetSdk = flutter.targetSdkVersion
|
||||
versionCode = flutter.versionCode
|
||||
versionName = flutter.versionName
|
||||
}
|
||||
|
||||
buildTypes {
|
||||
release {
|
||||
// TODO: Add your own signing config for the release build.
|
||||
// Signing with the debug keys for now, so `flutter run --release` works.
|
||||
signingConfig = signingConfigs.getByName("debug")
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
kotlin {
|
||||
compilerOptions {
|
||||
jvmTarget = org.jetbrains.kotlin.gradle.dsl.JvmTarget.JVM_17
|
||||
}
|
||||
}
|
||||
|
||||
flutter {
|
||||
source = "../.."
|
||||
}
|
||||
@@ -0,0 +1,8 @@
|
||||
## This file must *NOT* be checked into Version Control Systems,
|
||||
# as it contains information specific to your local configuration.
|
||||
#
|
||||
# Location of the SDK. This is only used by Gradle.
|
||||
# For customization when using a Version Control System, please read the
|
||||
# header note.
|
||||
#Fri Jun 26 15:57:59 WIB 2026
|
||||
sdk.dir=C\:\\Users\\rafha\\AppData\\Local\\Android\\Sdk
|
||||
@@ -0,0 +1,7 @@
|
||||
<manifest xmlns:android="http://schemas.android.com/apk/res/android">
|
||||
<!-- The INTERNET permission is required for development. Specifically,
|
||||
the Flutter tool needs it to communicate with the running application
|
||||
to allow setting breakpoints, to provide hot reload, etc.
|
||||
-->
|
||||
<uses-permission android:name="android.permission.INTERNET"/>
|
||||
</manifest>
|
||||
@@ -0,0 +1,50 @@
|
||||
<manifest xmlns:android="http://schemas.android.com/apk/res/android">
|
||||
<uses-permission android:name="android.permission.INTERNET" />
|
||||
<uses-permission android:name="android.permission.CAMERA" />
|
||||
<uses-permission android:name="android.permission.ACCESS_FINE_LOCATION" />
|
||||
<uses-permission android:name="android.permission.ACCESS_COARSE_LOCATION" />
|
||||
|
||||
<application
|
||||
android:label="app_pfm_ocr_v2"
|
||||
android:name="${applicationName}"
|
||||
android:icon="@mipmap/launcher_icon">
|
||||
<activity
|
||||
android:name=".MainActivity"
|
||||
android:exported="true"
|
||||
android:launchMode="singleTop"
|
||||
android:taskAffinity=""
|
||||
android:theme="@style/LaunchTheme"
|
||||
android:configChanges="orientation|keyboardHidden|keyboard|screenSize|smallestScreenSize|locale|layoutDirection|fontScale|screenLayout|density|uiMode"
|
||||
android:hardwareAccelerated="true"
|
||||
android:windowSoftInputMode="adjustResize">
|
||||
<!-- Specifies an Android theme to apply to this Activity as soon as
|
||||
the Android process has started. This theme is visible to the user
|
||||
while the Flutter UI initializes. After that, this theme continues
|
||||
to determine the Window background behind the Flutter UI. -->
|
||||
<meta-data
|
||||
android:name="io.flutter.embedding.android.NormalTheme"
|
||||
android:resource="@style/NormalTheme"
|
||||
/>
|
||||
<intent-filter>
|
||||
<action android:name="android.intent.action.MAIN"/>
|
||||
<category android:name="android.intent.category.LAUNCHER"/>
|
||||
</intent-filter>
|
||||
</activity>
|
||||
<!-- Don't delete the meta-data below.
|
||||
This is used by the Flutter tool to generate GeneratedPluginRegistrant.java -->
|
||||
<meta-data
|
||||
android:name="flutterEmbedding"
|
||||
android:value="2" />
|
||||
</application>
|
||||
<!-- Required to query activities that can process text, see:
|
||||
https://developer.android.com/training/package-visibility and
|
||||
https://developer.android.com/reference/android/content/Intent#ACTION_PROCESS_TEXT.
|
||||
|
||||
In particular, this is used by the Flutter engine in io.flutter.plugin.text.ProcessTextPlugin. -->
|
||||
<queries>
|
||||
<intent>
|
||||
<action android:name="android.intent.action.PROCESS_TEXT"/>
|
||||
<data android:mimeType="text/plain"/>
|
||||
</intent>
|
||||
</queries>
|
||||
</manifest>
|
||||
@@ -0,0 +1,5 @@
|
||||
package com.databisnis.app_pfm_ocr_v2
|
||||
|
||||
import io.flutter.embedding.android.FlutterActivity
|
||||
|
||||
class MainActivity : FlutterActivity()
|
||||
@@ -0,0 +1,12 @@
|
||||
<?xml version="1.0" encoding="utf-8"?>
|
||||
<!-- Modify this file to customize your launch splash screen -->
|
||||
<layer-list xmlns:android="http://schemas.android.com/apk/res/android">
|
||||
<item android:drawable="?android:colorBackground" />
|
||||
|
||||
<!-- You can insert your own image assets here -->
|
||||
<!-- <item>
|
||||
<bitmap
|
||||
android:gravity="center"
|
||||
android:src="@mipmap/launch_image" />
|
||||
</item> -->
|
||||
</layer-list>
|
||||
@@ -0,0 +1,12 @@
|
||||
<?xml version="1.0" encoding="utf-8"?>
|
||||
<!-- Modify this file to customize your launch splash screen -->
|
||||
<layer-list xmlns:android="http://schemas.android.com/apk/res/android">
|
||||
<item android:drawable="@android:color/white" />
|
||||
|
||||
<!-- You can insert your own image assets here -->
|
||||
<!-- <item>
|
||||
<bitmap
|
||||
android:gravity="center"
|
||||
android:src="@mipmap/launch_image" />
|
||||
</item> -->
|
||||
</layer-list>
|
||||
|
After Width: | Height: | Size: 544 B |
|
After Width: | Height: | Size: 6.4 KiB |
|
After Width: | Height: | Size: 442 B |
|
After Width: | Height: | Size: 4.0 KiB |
|
After Width: | Height: | Size: 721 B |
|
After Width: | Height: | Size: 8.9 KiB |
|
After Width: | Height: | Size: 1.0 KiB |
|
After Width: | Height: | Size: 14 KiB |
|
After Width: | Height: | Size: 1.4 KiB |
|
After Width: | Height: | Size: 18 KiB |
@@ -0,0 +1,18 @@
|
||||
<?xml version="1.0" encoding="utf-8"?>
|
||||
<resources>
|
||||
<!-- Theme applied to the Android Window while the process is starting when the OS's Dark Mode setting is on -->
|
||||
<style name="LaunchTheme" parent="@android:style/Theme.Black.NoTitleBar">
|
||||
<!-- Show a splash screen on the activity. Automatically removed when
|
||||
the Flutter engine draws its first frame -->
|
||||
<item name="android:windowBackground">@drawable/launch_background</item>
|
||||
</style>
|
||||
<!-- Theme applied to the Android Window as soon as the process has started.
|
||||
This theme determines the color of the Android Window while your
|
||||
Flutter UI initializes, as well as behind your Flutter UI while its
|
||||
running.
|
||||
|
||||
This Theme is only used starting with V2 of Flutter's Android embedding. -->
|
||||
<style name="NormalTheme" parent="@android:style/Theme.Black.NoTitleBar">
|
||||
<item name="android:windowBackground">?android:colorBackground</item>
|
||||
</style>
|
||||
</resources>
|
||||
@@ -0,0 +1,18 @@
|
||||
<?xml version="1.0" encoding="utf-8"?>
|
||||
<resources>
|
||||
<!-- Theme applied to the Android Window while the process is starting when the OS's Dark Mode setting is off -->
|
||||
<style name="LaunchTheme" parent="@android:style/Theme.Light.NoTitleBar">
|
||||
<!-- Show a splash screen on the activity. Automatically removed when
|
||||
the Flutter engine draws its first frame -->
|
||||
<item name="android:windowBackground">@drawable/launch_background</item>
|
||||
</style>
|
||||
<!-- Theme applied to the Android Window as soon as the process has started.
|
||||
This theme determines the color of the Android Window while your
|
||||
Flutter UI initializes, as well as behind your Flutter UI while its
|
||||
running.
|
||||
|
||||
This Theme is only used starting with V2 of Flutter's Android embedding. -->
|
||||
<style name="NormalTheme" parent="@android:style/Theme.Light.NoTitleBar">
|
||||
<item name="android:windowBackground">?android:colorBackground</item>
|
||||
</style>
|
||||
</resources>
|
||||
@@ -0,0 +1,7 @@
|
||||
<manifest xmlns:android="http://schemas.android.com/apk/res/android">
|
||||
<!-- The INTERNET permission is required for development. Specifically,
|
||||
the Flutter tool needs it to communicate with the running application
|
||||
to allow setting breakpoints, to provide hot reload, etc.
|
||||
-->
|
||||
<uses-permission android:name="android.permission.INTERNET"/>
|
||||
</manifest>
|
||||
@@ -0,0 +1,24 @@
|
||||
allprojects {
|
||||
repositories {
|
||||
google()
|
||||
mavenCentral()
|
||||
}
|
||||
}
|
||||
|
||||
val newBuildDir: Directory =
|
||||
rootProject.layout.buildDirectory
|
||||
.dir("../../build")
|
||||
.get()
|
||||
rootProject.layout.buildDirectory.value(newBuildDir)
|
||||
|
||||
subprojects {
|
||||
val newSubprojectBuildDir: Directory = newBuildDir.dir(project.name)
|
||||
project.layout.buildDirectory.value(newSubprojectBuildDir)
|
||||
}
|
||||
subprojects {
|
||||
project.evaluationDependsOn(":app")
|
||||
}
|
||||
|
||||
tasks.register<Delete>("clean") {
|
||||
delete(rootProject.layout.buildDirectory)
|
||||
}
|
||||
@@ -0,0 +1,6 @@
|
||||
org.gradle.jvmargs=-Xmx8G -XX:MaxMetaspaceSize=4G -XX:ReservedCodeCacheSize=512m -XX:+HeapDumpOnOutOfMemoryError
|
||||
android.useAndroidX=true
|
||||
# This newDsl flag was added by the Flutter template
|
||||
android.newDsl=false
|
||||
# This builtInKotlin flag was added by the Flutter template
|
||||
android.builtInKotlin=false
|
||||
@@ -0,0 +1,5 @@
|
||||
distributionBase=GRADLE_USER_HOME
|
||||
distributionPath=wrapper/dists
|
||||
zipStoreBase=GRADLE_USER_HOME
|
||||
zipStorePath=wrapper/dists
|
||||
distributionUrl=https\://services.gradle.org/distributions/gradle-9.1.0-all.zip
|
||||
@@ -0,0 +1,26 @@
|
||||
pluginManagement {
|
||||
val flutterSdkPath =
|
||||
run {
|
||||
val properties = java.util.Properties()
|
||||
file("local.properties").inputStream().use { properties.load(it) }
|
||||
val flutterSdkPath = properties.getProperty("flutter.sdk")
|
||||
require(flutterSdkPath != null) { "flutter.sdk not set in local.properties" }
|
||||
flutterSdkPath
|
||||
}
|
||||
|
||||
includeBuild("$flutterSdkPath/packages/flutter_tools/gradle")
|
||||
|
||||
repositories {
|
||||
google()
|
||||
mavenCentral()
|
||||
gradlePluginPortal()
|
||||
}
|
||||
}
|
||||
|
||||
plugins {
|
||||
id("dev.flutter.flutter-plugin-loader") version "1.0.0"
|
||||
id("com.android.application") version "9.0.1" apply false
|
||||
id("org.jetbrains.kotlin.android") version "2.3.20" apply false
|
||||
}
|
||||
|
||||
include(":app")
|
||||
|
After Width: | Height: | Size: 48 KiB |
@@ -0,0 +1,14 @@
|
||||
.git
|
||||
.github
|
||||
.venv
|
||||
.venv-api
|
||||
PaddleOCR-VL-1.6_Online_Demo/.venv
|
||||
**/__pycache__
|
||||
**/*.pyc
|
||||
.cache
|
||||
.python-version
|
||||
issues
|
||||
*.md
|
||||
pfm-web-app/node_modules
|
||||
pfm-web-app/.next
|
||||
|
||||
@@ -0,0 +1,5 @@
|
||||
# Port to serve the application on the host (routed via Nginx)
|
||||
APP_PORT=8000
|
||||
|
||||
# GPU index to allocate (e.g. 0, 1, or 0,1)
|
||||
CUDA_VISIBLE_DEVICES=1
|
||||
@@ -0,0 +1,17 @@
|
||||
.venv/
|
||||
.venv-api/
|
||||
.env
|
||||
__pycache__/
|
||||
*.pyc
|
||||
.python-version
|
||||
.antigravitycli/
|
||||
PaddleOCR-VL-1.6_Online_Demo/examples/do-pfm/*.jpg
|
||||
pfm-web-app/public/do-pfm/*
|
||||
uploads/*
|
||||
pfm-web-app/public/produk-pfm/**/*.jpeg
|
||||
pfm-web-app/public/produk-pfm/yolo_dataset/*
|
||||
pfm-web-app/public/produk-pfm/runs/*
|
||||
pfm-web-app/public/produk-pfm/models/*
|
||||
*.pt
|
||||
test_img.jpeg
|
||||
screenshots/*.jpg
|
||||
@@ -0,0 +1,18 @@
|
||||
[submodule "deepseek-ocr-2-demo-2026"]
|
||||
path = deepseek-ocr-2-demo-2026
|
||||
url = https://github.com/abdshomad/deepseek-ocr-2-demo-2026.git
|
||||
[submodule "LightOnOCR-2-1B-Demo-2026"]
|
||||
path = LightOnOCR-2-1B-Demo-2026
|
||||
url = https://github.com/abdshomad/LightOnOCR-2-1B-Demo-2026.git
|
||||
[submodule "nvidia-nemotron-ocr-v2-demo-2026"]
|
||||
path = nvidia-nemotron-ocr-v2-demo-2026
|
||||
url = https://github.com/abdshomad/nvidia-nemotron-ocr-v2-demo-2026.git
|
||||
[submodule "dots.ocr-demo-2026"]
|
||||
path = dots.ocr-demo-2026
|
||||
url = https://github.com/abdshomad/dots.ocr-demo-2026.git
|
||||
[submodule "glm-ocr-demo-2026"]
|
||||
path = glm-ocr-demo-2026
|
||||
url = https://github.com/abdshomad/glm-ocr-demo-2026.git
|
||||
[submodule "andrej-karpathy-skills"]
|
||||
path = andrej-karpathy-skills
|
||||
url = https://github.com/multica-ai/andrej-karpathy-skills.git
|
||||
@@ -0,0 +1,237 @@
|
||||
# AGENTS: PaddleOCR-VL-1.6 vLLM Service
|
||||
|
||||
This repository serves **PaddleOCR-VL-1.6** as a dedicated VLM inference backend using **vLLM**. All Python workflows use **uv** (never bare `pip` or system Python).
|
||||
|
||||
## Architecture
|
||||
|
||||
```
|
||||
Client (PaddleOCR pipeline) --> HTTP /v1 --> paddleocr genai_server (vLLM backend)
|
||||
```
|
||||
|
||||
This service exposes only the VLM stage. Clients connect with `vl_rec_backend="vllm-server"` and `vl_rec_server_url="http://<host>:8118/v1"`.
|
||||
|
||||
## Prerequisites
|
||||
|
||||
- Linux with NVIDIA GPU (CC >= 8.0 recommended; CUDA 12.6+ driver support)
|
||||
- [uv](https://docs.astral.sh/uv/) installed (`uv --version`)
|
||||
- ~16 GB GPU VRAM for default settings (tune via `config/vllm_config.yaml`)
|
||||
|
||||
## Quick start
|
||||
|
||||
```bash
|
||||
cd <YOUR-WORKING-DIR>/ai-ocr-pfm-2026
|
||||
|
||||
# 1) Create Python 3.12 venv and install dependencies
|
||||
./scripts/install.sh
|
||||
|
||||
# 2) Start the vLLM-backed genai server
|
||||
./scripts/serve.sh
|
||||
```
|
||||
|
||||
Default endpoint: `http://0.0.0.0:8118/v1`
|
||||
|
||||
## uv conventions (always follow)
|
||||
|
||||
| Task | Command |
|
||||
|------|---------|
|
||||
| Create/sync env | `uv sync` |
|
||||
| Run any Python | `uv run <command>` |
|
||||
| Add a package | `uv add <package>` |
|
||||
| Run server | `./scripts/serve.sh` or `uv run paddleocr genai_server ...` |
|
||||
|
||||
Never use `python -m pip`, `pip install`, or `python -m venv` directly in this repo.
|
||||
|
||||
## Issue recording (always follow)
|
||||
|
||||
**Every problem encountered** during install, serve, debug, or client integration must be written to `issues/` before moving on — even if it was resolved in the same session.
|
||||
|
||||
### Naming
|
||||
|
||||
```
|
||||
issues/{NN}-{slug}.md
|
||||
```
|
||||
|
||||
| Part | Rule | Example |
|
||||
|------|------|---------|
|
||||
| `{NN}` | Two-digit running number (`01`, `02`, …). Increment from the highest existing file. | `03` |
|
||||
| `{slug}` | Lowercase kebab-case summary of the problem | `gpu-memory-startup-failure` |
|
||||
|
||||
Full example: `issues/04-gpu-memory-startup-failure.md`
|
||||
|
||||
### When to create a file
|
||||
|
||||
- Install or dependency errors (flash-attn, vLLM, uv conflicts)
|
||||
- Server startup or runtime failures (OOM, port bind, model load)
|
||||
- Client integration bugs or misconfiguration
|
||||
- Workarounds that took non-obvious steps to discover
|
||||
|
||||
Do **not** rely on chat history or inline comments alone — if it blocked progress, it belongs in `issues/`.
|
||||
|
||||
### File template
|
||||
|
||||
```markdown
|
||||
# Issue {NN}: {Short title}
|
||||
|
||||
## Problem
|
||||
What failed, with exact error message or symptom.
|
||||
|
||||
## Context
|
||||
Environment, command run, relevant config (`.env`, `config/vllm_config.yaml`).
|
||||
|
||||
## Solution
|
||||
What fixed it, or current workaround / open status.
|
||||
|
||||
## References
|
||||
Links, related issue files, or AGENTS.md sections.
|
||||
```
|
||||
|
||||
### Index
|
||||
|
||||
Check `issues/` for the next number:
|
||||
|
||||
```bash
|
||||
ls issues/*.md 2>/dev/null | sort
|
||||
```
|
||||
|
||||
See [issues/](issues/) for recorded problems and fixes from this project.
|
||||
|
||||
## Environment variables
|
||||
|
||||
Copy `.env.example` to `.env` and adjust as needed:
|
||||
|
||||
| Variable | Default | Description |
|
||||
|----------|---------|-------------|
|
||||
| `GENAI_HOST` | `0.0.0.0` | Bind address |
|
||||
| `GENAI_PORT` | `8118` | Service port |
|
||||
| `GENAI_MODEL` | `PaddleOCR-VL-1.6-0.9B` | Model name for `genai_server` |
|
||||
| `GENAI_BACKEND` | `vllm` | Inference backend |
|
||||
| `VLLM_CONFIG` | `config/vllm_config.yaml` | vLLM backend YAML config |
|
||||
| `CUDA_VISIBLE_DEVICES` | `1` (see `.env.example`) | GPU index(es) to use |
|
||||
|
||||
On dual-GPU hosts, pick the GPU with more free VRAM. If startup fails with a memory error, lower `gpu-memory-utilization` in `config/vllm_config.yaml`.
|
||||
|
||||
## Client usage
|
||||
|
||||
After the server is running:
|
||||
|
||||
```bash
|
||||
# CLI
|
||||
uv run paddleocr doc_parser \
|
||||
--input https://paddle-model-ecology.bj.bcebos.com/paddlex/imgs/demo_image/paddleocr_vl_demo.png \
|
||||
--vl_rec_backend vllm-server \
|
||||
--vl_rec_server_url http://localhost:8118/v1
|
||||
```
|
||||
|
||||
```python
|
||||
from paddleocr import PaddleOCRVL
|
||||
|
||||
pipeline = PaddleOCRVL(
|
||||
vl_rec_backend="vllm-server",
|
||||
vl_rec_server_url="http://127.0.0.1:8118/v1",
|
||||
)
|
||||
output = pipeline.predict("path/to/image.png")
|
||||
```
|
||||
|
||||
Note: The full PaddleOCR-VL client should run in a **separate** environment if it needs PaddlePaddle GPU + Transformers. This repo is the isolated vLLM server only.
|
||||
|
||||
## Tuning vLLM
|
||||
|
||||
Edit `config/vllm_config.yaml`:
|
||||
|
||||
```yaml
|
||||
gpu-memory-utilization: 0.8
|
||||
max-num-seqs: 128
|
||||
```
|
||||
|
||||
Reference: [PaddleOCR-VL vLLM parameter tuning](https://www.paddleocr.ai/latest/en/version3.x/pipeline_usage/PaddleOCR-VL.html#331-server-side-parameter-adjustment)
|
||||
|
||||
## Troubleshooting
|
||||
|
||||
See `issues/` for full write-ups. Quick pointers:
|
||||
|
||||
| Symptom | Issue file |
|
||||
|---------|------------|
|
||||
| `paddleocr install_genai_server_deps` / `No module named pip` | [01-genai-server-deps-pip-in-uv-venv.md](issues/01-genai-server-deps-pip-in-uv-venv.md) |
|
||||
| flash-attn wheel incompatible with Python version | [02-flash-attn-wheel-python-version-mismatch.md](issues/02-flash-attn-wheel-python-version-mismatch.md) |
|
||||
| `uv pip` targets wrong venv from another project | [03-active-virtual-env-from-other-project.md](issues/03-active-virtual-env-from-other-project.md) |
|
||||
| Free memory below `gpu-memory-utilization` on startup | [04-gpu-memory-startup-failure.md](issues/04-gpu-memory-startup-failure.md) |
|
||||
| `TokenizersBackend has no attribute all_special_tokens_extended` | [05-transformers-tokenizers-incompatibility.md](issues/05-transformers-tokenizers-incompatibility.md) |
|
||||
| Extracted images not shown in Gradio demo (raw base64 in markdown) | [06-extracted-images-raw-base64-not-displayed.md](issues/06-extracted-images-raw-base64-not-displayed.md) |
|
||||
|
||||
### flash-attn build failures
|
||||
|
||||
Install the prebuilt wheel after `uv sync` (see `scripts/install.sh`):
|
||||
|
||||
```bash
|
||||
FLASH_ATTN_WHEEL="https://github.com/mjun0812/flash-attention-prebuild-wheels/releases/download/v0.3.14/flash_attn-2.8.2+cu128torch2.8-cp312-cp312-linux_x86_64.whl" \
|
||||
./scripts/install.sh
|
||||
```
|
||||
|
||||
Pick the wheel matching your Python and CUDA versions from [flash-attention prebuild wheels](https://mjunya.com/flash-attention-prebuild-wheels/). Details: [02-flash-attn-wheel-python-version-mismatch.md](issues/02-flash-attn-wheel-python-version-mismatch.md).
|
||||
|
||||
Note: `paddleocr install_genai_server_deps` uses `pip` internally and is incompatible with uv-managed venvs. See [01-genai-server-deps-pip-in-uv-venv.md](issues/01-genai-server-deps-pip-in-uv-venv.md). This repo installs the vLLM stack via `uv sync` + `uv pip`.
|
||||
|
||||
### `TokenizersBackend has no attribute all_special_tokens_extended`
|
||||
|
||||
Pin transformers (already in `pyproject.toml`):
|
||||
|
||||
```bash
|
||||
uv pip install "transformers==4.57.6"
|
||||
```
|
||||
|
||||
See [05-transformers-tokenizers-incompatibility.md](issues/05-transformers-tokenizers-incompatibility.md).
|
||||
|
||||
### Do not install `paddlepaddle-gpu` in this venv
|
||||
|
||||
vLLM and PaddlePaddle GPU conflict. This server env uses `paddleocr[doc-parser]` without Paddle GPU.
|
||||
|
||||
### GPU memory on startup
|
||||
|
||||
If vLLM reports free memory below `gpu-memory-utilization`, either:
|
||||
|
||||
- Set `CUDA_VISIBLE_DEVICES` to a less-busy GPU
|
||||
- Lower `gpu-memory-utilization` in `config/vllm_config.yaml` (e.g. `0.75` or `0.7`)
|
||||
|
||||
See [04-gpu-memory-startup-failure.md](issues/04-gpu-memory-startup-failure.md).
|
||||
|
||||
### Health check
|
||||
|
||||
```bash
|
||||
curl -s http://localhost:8118/v1/models | jq .
|
||||
```
|
||||
|
||||
## File map
|
||||
|
||||
| Path | Purpose |
|
||||
|------|---------|
|
||||
| `issues/` | Recorded problems and fixes (`{NN}-{slug}.md`) |
|
||||
| `pyproject.toml` | uv project metadata and base dependencies |
|
||||
| `scripts/install.sh` | Bootstrap venv + vLLM server deps |
|
||||
| `scripts/serve.sh` | Start `paddleocr genai_server` |
|
||||
| `config/vllm_config.yaml` | vLLM backend tuning |
|
||||
| `.env.example` | Environment variable template |
|
||||
|
||||
## References
|
||||
|
||||
- [PaddleOCR-VL usage tutorial](https://www.paddleocr.ai/latest/en/version3.x/pipeline_usage/PaddleOCR-VL.html)
|
||||
- [PaddleOCR genai_server FAQ](https://github.com/PaddlePaddle/PaddleOCR/discussions/16822)
|
||||
|
||||
## Coding Guidelines (always follow)
|
||||
|
||||
We use the [karpathy-guidelines](file:///home/user/LABS/OCR/paddle-ocr-vl-1-6-using-vllm-2026/andrej-karpathy-skills/skills/karpathy-guidelines/SKILL.md) skill to reduce common LLM coding mistakes. Refer to [SKILL.md](file:///home/user/LABS/OCR/paddle-ocr-vl-1-6-using-vllm-2026/andrej-karpathy-skills/skills/karpathy-guidelines/SKILL.md) for details:
|
||||
1. **Think Before Coding**: Explicitly state assumptions and surface tradeoffs instead of making silent choices.
|
||||
2. **Simplicity First**: Write the minimum amount of code to solve the problem with zero speculative configurations.
|
||||
3. **Surgical Changes**: Edit only what is required and match the existing coding style exactly.
|
||||
4. **Goal-Driven Execution**: Define verifiable success criteria and run automated tests/screenshots to confirm correctness.
|
||||
|
||||
## Path Guidelines (always follow)
|
||||
|
||||
Never use full paths containing the user's logged-in name (e.g., `/home/{uid}/path`). Always use relative paths instead (e.g., `.` or `./path` relative to the workspace root).
|
||||
|
||||
## App Testing Guidelines (always follow)
|
||||
|
||||
When the user intentionally asks to test the app:
|
||||
- Use browser tools to test the app.
|
||||
- Take a screenshot for each sample image, each step, and each variant/option (if any), until the OCR result appears.
|
||||
- Save the screenshots in the `/screenshots/` folder.
|
||||
- Follow the file naming convention: `{2-digit-number}-{step#}-{variant_or_options_if_any}-{slug}.jpg` (e.g., `01-step1-default-upload.jpg`).
|
||||
@@ -0,0 +1,82 @@
|
||||
# Stage 0: GPU Base image
|
||||
FROM nvidia/cuda:12.6.0-devel-ubuntu22.04 AS base-gpu
|
||||
|
||||
ENV DEBIAN_FRONTEND=noninteractive
|
||||
ENV PATH="/root/.local/bin:$PATH"
|
||||
|
||||
# Install system dependencies (libgl and libglib are required for OpenCV)
|
||||
RUN apt-get update && apt-get install -y --no-install-recommends \
|
||||
curl \
|
||||
git \
|
||||
libgl1 \
|
||||
libglib2.0-0 \
|
||||
&& rm -rf /var/lib/apt/lists/*
|
||||
|
||||
# Install uv
|
||||
RUN curl -LsSf https://astral.sh/uv/install.sh | sh
|
||||
|
||||
# --- vLLM Server Stage ---
|
||||
FROM base-gpu AS vllm-server
|
||||
WORKDIR /app
|
||||
|
||||
# Install project dependencies
|
||||
COPY pyproject.toml uv.lock ./
|
||||
RUN uv python pin 3.12 && uv sync --frozen --no-dev
|
||||
|
||||
# Install prebuilt flash-attention wheel
|
||||
ARG FLASH_ATTN_WHEEL=https://github.com/mjun0812/flash-attention-prebuild-wheels/releases/download/v0.3.14/flash_attn-2.8.2+cu128torch2.8-cp312-cp312-linux_x86_64.whl
|
||||
RUN uv pip install --python .venv "${FLASH_ATTN_WHEEL}"
|
||||
|
||||
COPY . /app
|
||||
RUN chmod +x /app/scripts/serve.sh
|
||||
|
||||
EXPOSE 8118
|
||||
CMD ["./scripts/serve.sh"]
|
||||
|
||||
# --- Pipeline API Stage ---
|
||||
FROM base-gpu AS pipeline-api
|
||||
WORKDIR /app
|
||||
|
||||
# Build paddlepaddle and paddlex virtual env
|
||||
RUN uv venv .venv-api --python 3.12
|
||||
RUN uv pip install --python .venv-api paddlepaddle-gpu -i https://www.paddlepaddle.org.cn/packages/stable/cu126/
|
||||
RUN uv pip install --python .venv-api "paddleocr[doc-parser]>=3.3.0"
|
||||
RUN uv pip install --python .venv-api "aiohttp>=3.9" "filetype>=1.2" "fastapi>=0.110" "starlette>=0.36" "uvicorn>=0.16" "ultralytics>=8.0"
|
||||
|
||||
COPY . /app
|
||||
RUN chmod +x /app/scripts/serve-pipeline.sh
|
||||
|
||||
EXPOSE 8090
|
||||
CMD ["./scripts/serve-pipeline.sh"]
|
||||
|
||||
# --- Gradio UI Stage ---
|
||||
FROM python:3.12-slim AS gradio-ui
|
||||
WORKDIR /app/PaddleOCR-VL-1.6_Online_Demo
|
||||
|
||||
RUN apt-get update && apt-get install -y --no-install-recommends \
|
||||
curl \
|
||||
&& rm -rf /var/lib/apt/lists/*
|
||||
|
||||
COPY PaddleOCR-VL-1.6_Online_Demo/requirements.txt ./
|
||||
RUN pip install --no-cache-dir -r requirements.txt
|
||||
|
||||
COPY PaddleOCR-VL-1.6_Online_Demo ./
|
||||
|
||||
EXPOSE 7870
|
||||
|
||||
ENV GRADIO_SERVER_NAME="0.0.0.0"
|
||||
ENV GRADIO_SERVER_PORT="7870"
|
||||
|
||||
CMD ["python", "app.py"]
|
||||
|
||||
# --- Next.js Web App Stage ---
|
||||
FROM node:20-slim AS pfm-web-app
|
||||
WORKDIR /app
|
||||
COPY pfm-web-app/package.json pfm-web-app/package-lock.json ./
|
||||
RUN npm ci
|
||||
COPY pfm-web-app/ ./
|
||||
ENV NODE_ENV=production
|
||||
RUN npm run build
|
||||
EXPOSE 3000
|
||||
CMD ["npm", "start"]
|
||||
|
||||
@@ -0,0 +1,151 @@
|
||||
# PaddleOCR-VL-1.6 on vLLM
|
||||
|
||||
Local deployment of [PaddleOCR-VL-1.6](https://www.paddleocr.ai/latest/en/version3.x/pipeline_usage/PaddleOCR-VL.html) using **vLLM** as the VLM inference backend. All Python workflows use **[uv](https://docs.astral.sh/uv/)**.
|
||||
|
||||
## Architecture
|
||||
|
||||
```
|
||||
Gradio demo (7870)
|
||||
│
|
||||
▼
|
||||
Pipeline API (8090) ── layout + preprocessing (PaddlePaddle GPU)
|
||||
│
|
||||
▼
|
||||
vLLM genai server (8118) ── PaddleOCR-VL-1.6 VLM
|
||||
```
|
||||
|
||||
| Service | Script | Default URL |
|
||||
|---------|--------|-------------|
|
||||
| vLLM VLM server | `./scripts/serve.sh` | `http://127.0.0.1:8118/v1` |
|
||||
| Full pipeline API | `./scripts/serve-pipeline.sh` | `http://127.0.0.1:8090/layout-parsing` |
|
||||
| Online demo UI | `./scripts/run-demo.sh` | `http://127.0.0.1:7870` |
|
||||
|
||||
The vLLM server exposes only the VLM stage. For HTTP document parsing (layout + OCR), run the pipeline API, which calls vLLM via `config/pipeline_config_vllm.yaml`.
|
||||
|
||||
## Prerequisites
|
||||
|
||||
- Linux with NVIDIA GPU (CC ≥ 8.0 recommended; CUDA 12.6+ driver)
|
||||
- [uv](https://docs.astral.sh/uv/) installed
|
||||
- ~16 GB GPU VRAM for default vLLM settings (tune in `config/vllm_config.yaml`)
|
||||
|
||||
## Quick start
|
||||
|
||||
```bash
|
||||
git clone <repo-url> ai-ocr-pfm-2026
|
||||
cd ai-ocr-pfm-2026
|
||||
|
||||
cp .env.example .env # adjust CUDA_VISIBLE_DEVICES if needed
|
||||
|
||||
# 1) Install vLLM server (.venv)
|
||||
FLASH_ATTN_WHEEL="https://github.com/mjun0812/flash-attention-prebuild-wheels/releases/download/v0.3.14/flash_attn-2.8.2+cu128torch2.8-cp312-cp312-linux_x86_64.whl" \
|
||||
./scripts/install.sh
|
||||
|
||||
# 2) Install pipeline API (.venv-api) — optional, needed for demo / full HTTP API
|
||||
./scripts/install-pipeline.sh
|
||||
```
|
||||
|
||||
Start services (three terminals, or background each):
|
||||
|
||||
```bash
|
||||
./scripts/serve.sh # vLLM on :8118
|
||||
./scripts/serve-pipeline.sh # pipeline on :8090
|
||||
./scripts/run-demo.sh # Gradio on :7870
|
||||
```
|
||||
|
||||
Health checks:
|
||||
|
||||
```bash
|
||||
curl -s http://127.0.0.1:8118/v1/models | jq .
|
||||
curl -s http://127.0.0.1:8090/health
|
||||
curl -s -o /dev/null -w "%{http_code}\n" http://127.0.0.1:7870/
|
||||
```
|
||||
|
||||
## Configuration
|
||||
|
||||
Copy `.env.example` to `.env`:
|
||||
|
||||
| Variable | Default | Description |
|
||||
|----------|---------|-------------|
|
||||
| `GENAI_HOST` | `0.0.0.0` | vLLM bind address |
|
||||
| `GENAI_PORT` | `8118` | vLLM port |
|
||||
| `GENAI_MODEL` | `PaddleOCR-VL-1.6-0.9B` | Model name |
|
||||
| `VLLM_CONFIG` | `config/vllm_config.yaml` | vLLM tuning |
|
||||
| `CUDA_VISIBLE_DEVICES` | `1` | GPU for vLLM (use least-busy GPU) |
|
||||
| `PIPELINE_PORT` | `8090` | Pipeline API port |
|
||||
| `PIPELINE_DEVICE` | `gpu:0` | GPU for layout/preprocessing |
|
||||
| `GRADIO_PORT` | `7870` | Demo UI port |
|
||||
|
||||
vLLM tuning (`config/vllm_config.yaml`):
|
||||
|
||||
```yaml
|
||||
gpu-memory-utilization: 0.75
|
||||
max-num-seqs: 128
|
||||
```
|
||||
|
||||
## Client usage
|
||||
|
||||
### Python (vLLM only)
|
||||
|
||||
```python
|
||||
from paddleocr import PaddleOCRVL
|
||||
|
||||
pipeline = PaddleOCRVL(
|
||||
vl_rec_backend="vllm-server",
|
||||
vl_rec_server_url="http://127.0.0.1:8118/v1",
|
||||
)
|
||||
output = pipeline.predict("path/to/image.png")
|
||||
```
|
||||
|
||||
Run the client in a **separate** environment if it needs PaddlePaddle GPU alongside Transformers.
|
||||
|
||||
### CLI
|
||||
|
||||
```bash
|
||||
uv run paddleocr doc_parser \
|
||||
--input demo.png \
|
||||
--vl_rec_backend vllm-server \
|
||||
--vl_rec_server_url http://127.0.0.1:8118/v1
|
||||
```
|
||||
|
||||
### HTTP (full pipeline)
|
||||
|
||||
```bash
|
||||
curl -X POST http://127.0.0.1:8090/layout-parsing \
|
||||
-H "Content-Type: application/json" \
|
||||
-d '{"file":"<base64>", "fileType": 1, "useLayoutDetection": true}'
|
||||
```
|
||||
|
||||
## Project layout
|
||||
|
||||
```
|
||||
config/
|
||||
vllm_config.yaml # vLLM backend tuning
|
||||
pipeline_config_vllm.yaml # pipeline → vLLM server URL
|
||||
scripts/
|
||||
install.sh # bootstrap .venv (vLLM)
|
||||
install-pipeline.sh # bootstrap .venv-api (pipeline)
|
||||
serve.sh # start vLLM genai server
|
||||
serve-pipeline.sh # start pipeline API
|
||||
run-demo.sh # start Gradio demo
|
||||
PaddleOCR-VL-1.6_Online_Demo/ # bundled Hugging Face-style demo
|
||||
issues/ # recorded problems and fixes
|
||||
AGENTS.md # agent / contributor guide
|
||||
```
|
||||
|
||||
## Troubleshooting
|
||||
|
||||
See [issues/](issues/) for detailed write-ups. Common fixes:
|
||||
|
||||
| Symptom | Fix |
|
||||
|---------|-----|
|
||||
| GPU OOM on vLLM startup | Lower `gpu-memory-utilization` or set `CUDA_VISIBLE_DEVICES` to a free GPU |
|
||||
| flash-attn build failure | Use prebuilt wheel via `FLASH_ATTN_WHEEL=... ./scripts/install.sh` |
|
||||
| Port 8080 in use | Pipeline defaults to **8090**; demo defaults to **7870** |
|
||||
|
||||
Agent conventions and issue-recording rules: [AGENTS.md](AGENTS.md).
|
||||
|
||||
## References
|
||||
|
||||
- [PaddleOCR-VL usage tutorial](https://www.paddleocr.ai/latest/en/version3.x/pipeline_usage/PaddleOCR-VL.html)
|
||||
- [PaddleOCR genai_server FAQ](https://github.com/PaddlePaddle/PaddleOCR/discussions/16822)
|
||||
- [flash-attention prebuild wheels](https://mjunya.com/flash-attention-prebuild-wheels/)
|
||||
@@ -0,0 +1 @@
|
||||
ai-ocr-pfm-2026
|
||||
@@ -0,0 +1,592 @@
|
||||
import base64
|
||||
import io
|
||||
import os
|
||||
import re
|
||||
import traceback
|
||||
from datetime import date
|
||||
from pathlib import Path
|
||||
|
||||
import requests
|
||||
from fastapi import FastAPI, HTTPException
|
||||
from fastapi.middleware.cors import CORSMiddleware
|
||||
from pydantic import BaseModel
|
||||
from PIL import Image
|
||||
import numpy as np
|
||||
from ultralytics import YOLO
|
||||
from paddleocr import PaddleOCR
|
||||
|
||||
app = FastAPI(title="PFM Product Classifier and OCR API")
|
||||
|
||||
CLASSIFIER_WEIGHTS_GLOB = "produk-pfm-classifier-26n-*e-*.pt"
|
||||
CLASSIFIER_DATE_IN_NAME = re.compile(
|
||||
r"produk-pfm-classifier-26n-\d+e-(\d{4}-\d{2}-\d{2})\.pt$"
|
||||
)
|
||||
|
||||
|
||||
def _classifier_date_from_name(path: Path) -> date | None:
|
||||
match = CLASSIFIER_DATE_IN_NAME.match(path.name)
|
||||
if not match:
|
||||
return None
|
||||
year, month, day = (int(part) for part in match.group(1).split("-"))
|
||||
return date(year, month, day)
|
||||
|
||||
|
||||
def latest_classifier_weights(models_dir: Path) -> Path | None:
|
||||
"""Return the newest produk-pfm-classifier .pt weights in models/."""
|
||||
if not models_dir.is_dir():
|
||||
return None
|
||||
|
||||
candidates = list(models_dir.glob(CLASSIFIER_WEIGHTS_GLOB))
|
||||
if not candidates:
|
||||
return None
|
||||
|
||||
def sort_key(path: Path) -> tuple[date, float]:
|
||||
name_date = _classifier_date_from_name(path) or date.min
|
||||
return (name_date, path.stat().st_mtime)
|
||||
|
||||
return max(candidates, key=sort_key)
|
||||
|
||||
|
||||
def resolve_classifier_models_dir() -> Path | None:
|
||||
"""Locate produk-pfm/models (env override, repo path, or Docker mount)."""
|
||||
env_dir = os.environ.get("CLASSIFIER_MODELS_DIR")
|
||||
if env_dir:
|
||||
path = Path(env_dir)
|
||||
if path.is_dir():
|
||||
return path
|
||||
|
||||
repo_root = Path(__file__).resolve().parent.parent
|
||||
for candidate in (
|
||||
repo_root / "pfm-web-app/public/produk-pfm/models",
|
||||
Path("/app/pfm-web-app/public/produk-pfm/models"),
|
||||
Path(__file__).resolve().parent,
|
||||
):
|
||||
if candidate.is_dir():
|
||||
return candidate
|
||||
return None
|
||||
|
||||
|
||||
def resolve_classifier_weights_path() -> Path | None:
|
||||
"""Resolve YOLO weights: CLASSIFIER_MODEL_PATH or newest file in models/."""
|
||||
explicit = os.environ.get("CLASSIFIER_MODEL_PATH")
|
||||
if explicit:
|
||||
path = Path(explicit)
|
||||
if path.is_file():
|
||||
return path
|
||||
print(f"CLASSIFIER_MODEL_PATH not found: {path}")
|
||||
|
||||
models_dir = resolve_classifier_models_dir()
|
||||
if models_dir is None:
|
||||
return None
|
||||
|
||||
weights = latest_classifier_weights(models_dir)
|
||||
if weights is None:
|
||||
print(f"No classifier weights matching {CLASSIFIER_WEIGHTS_GLOB} in {models_dir}")
|
||||
return weights
|
||||
|
||||
|
||||
# Enable CORS
|
||||
app.add_middleware(
|
||||
CORSMiddleware,
|
||||
allow_origins=["*"],
|
||||
allow_credentials=True,
|
||||
allow_methods=["*"],
|
||||
allow_headers=["*"],
|
||||
)
|
||||
|
||||
# Load models on startup
|
||||
print("Loading YOLO model...")
|
||||
yolo_model = None
|
||||
try:
|
||||
classifier_weights = resolve_classifier_weights_path()
|
||||
if classifier_weights is None:
|
||||
raise FileNotFoundError(
|
||||
"No produk-pfm classifier weights found. Train with train_classifier.py "
|
||||
"or set CLASSIFIER_MODEL_PATH / CLASSIFIER_MODELS_DIR."
|
||||
)
|
||||
print(f"Using classifier weights: {classifier_weights}")
|
||||
yolo_model = YOLO(str(classifier_weights))
|
||||
print("YOLO model loaded successfully.")
|
||||
except Exception as e:
|
||||
print(f"Error loading YOLO model: {e}")
|
||||
yolo_model = None
|
||||
|
||||
print("Loading PaddleOCR...")
|
||||
try:
|
||||
# Use standard textline orientation detection for PaddleOCR 3.x
|
||||
ocr = PaddleOCR(use_textline_orientation=True, lang='en')
|
||||
print("PaddleOCR loaded successfully.")
|
||||
except Exception as e:
|
||||
print(f"Error loading PaddleOCR: {e}")
|
||||
ocr = None
|
||||
|
||||
class ScanRequest(BaseModel):
|
||||
image_base64: str
|
||||
|
||||
def clean_ocr_text(text: str) -> str:
|
||||
return re.sub(r'^[^\w\s./-]+|[^\w\s./-]+$', '', text).strip()
|
||||
|
||||
EXP_KEYWORD_RE = re.compile(
|
||||
r'(?:exp(?:\.|ired)?|tgl(?:\s*exp)?|expiry|bbd|best\s*before|before|best|baik\s*digunakan|\bbb\b)',
|
||||
re.IGNORECASE,
|
||||
)
|
||||
DD_MM_YYYY_RE = re.compile(
|
||||
r'(?<!\d)(0[1-9]|[12]\d|3[01]).*?(0[1-9]|1[0-2]).*?(20\d{2})(?!\d)'
|
||||
)
|
||||
DDMMYYYY_RE = re.compile(
|
||||
r'(?<!\d)(0[1-9]|[12]\d|3[01])(0[1-9]|1[0-2])(20\d{2})(?!\d)'
|
||||
)
|
||||
BB_ATTACHED_DATE_RE = re.compile(
|
||||
r'\b(?:bb|bestbefore)\s*[:.-]?\s*(0[1-9]|[12]\d|3[01])(0[1-9]|1[0-2])(20\d{2})(?!\d)',
|
||||
re.IGNORECASE,
|
||||
)
|
||||
KEYWORD_DIGITS_RE = re.compile(
|
||||
r'(?:exp|expired|tgl|expiry|bbd|before|best|bb|baik|digunakan)\s*[:.-]?\s*(\d{6,8})\b',
|
||||
re.IGNORECASE,
|
||||
)
|
||||
LENIENT_DATE_RE = re.compile(
|
||||
r'(?<!\d)(\d{1,2}).*?(\d{1,2}).*?((?:20)?\d{2})(?!\d)'
|
||||
)
|
||||
|
||||
def is_valid_ddmmyyyy_digits(val: str) -> bool:
|
||||
if len(val) != 8 or not val.isdigit():
|
||||
return False
|
||||
day, month, year = int(val[0:2]), int(val[2:4]), int(val[4:8])
|
||||
return 1 <= day <= 31 and 1 <= month <= 12 and 2000 <= year <= 2099
|
||||
|
||||
def format_ddmmyyyy(val: str) -> str:
|
||||
if is_valid_ddmmyyyy_digits(val):
|
||||
return f"{val[0:2]}/{val[2:4]}/{val[4:8]}"
|
||||
return val.upper()
|
||||
|
||||
def format_ddmmyy(val: str) -> str:
|
||||
if len(val) == 6 and val.isdigit():
|
||||
day, month = int(val[0:2]), int(val[2:4])
|
||||
if 1 <= day <= 31 and 1 <= month <= 12:
|
||||
return f"{val[0:2]}/{val[2:4]}/{val[4:6]}"
|
||||
return val.upper()
|
||||
|
||||
def line_has_exp_keyword(line: str) -> bool:
|
||||
if EXP_KEYWORD_RE.search(line):
|
||||
return True
|
||||
# BB05032027 — keyword directly followed by digits
|
||||
return bool(re.search(r'(?i)\b(?:bb|bestbefore)(?:\s*[:.-]?\s*)?\d', line))
|
||||
|
||||
def clean_date_line(line: str) -> str:
|
||||
# 1) Replace "1)" with "0"
|
||||
cleaned = line.replace("1)", "0")
|
||||
|
||||
# 2) Replace "()" with "0"
|
||||
cleaned = cleaned.replace("()", "0")
|
||||
|
||||
# Clean BB misrecognitions (convert B8, 8B, 88 to BB when followed by digits)
|
||||
cleaned = re.sub(r'\b(?:B8|8B|88)(?=\d)', 'BB', cleaned)
|
||||
cleaned = re.sub(r'^(?:B8|8B|88)(?=\d)', 'BB', cleaned)
|
||||
|
||||
# Clean 012 month misrecognition (e.g. 020122027 -> 02022027)
|
||||
cleaned = re.sub(r'(?<!\d)(\d{1,2})012(20\d{2})(?!\d)', r'\g<1>02\g<2>', cleaned)
|
||||
cleaned = re.sub(r'(?<!\d)(\d{1,2})([-./\s]+)012([-./\s]+)(20\d{2})(?!\d)', r'\g<1>\g<2>02\g<3>\g<4>', cleaned)
|
||||
|
||||
# Clean 112 month misrecognition (e.g. 021122027 -> 02022027)
|
||||
cleaned = re.sub(r'(?<!\d)(\d{1,2})112(20\d{2})(?!\d)', r'\g<1>02\g<2>', cleaned)
|
||||
cleaned = re.sub(r'(?<!\d)(\d{1,2})([-./\s]+)112([-./\s]+)(20\d{2})(?!\d)', r'\g<1>\g<2>02\g<3>\g<4>', cleaned)
|
||||
|
||||
# Run contextual replacements
|
||||
for _ in range(3):
|
||||
# letter o/O flanked by digits or boundary -> 0
|
||||
cleaned = re.compile(r'(\d)[oO](\d|\b)').sub(r'\g<1>0\g<2>', cleaned)
|
||||
cleaned = re.compile(r'(\b|\d)[oO](\d)').sub(r'\g<1>0\g<2>', cleaned)
|
||||
|
||||
# letter I/i/l/| flanked by digits -> 1
|
||||
cleaned = re.compile(r'(\d)[Ii|l](\d|\b)').sub(r'\g<1>1\g<2>', cleaned)
|
||||
cleaned = re.compile(r'(\b|\d)[Ii|l](\d)').sub(r'\g<1>1\g<2>', cleaned)
|
||||
|
||||
# letter S/s flanked by digits -> 5
|
||||
cleaned = re.compile(r'(\d)[Ss](\d|\b)').sub(r'\g<1>5\g<2>', cleaned)
|
||||
cleaned = re.compile(r'(\b|\d)[Ss](\d)').sub(r'\g<1>5\g<2>', cleaned)
|
||||
|
||||
# letter Z/z flanked by digits -> 2
|
||||
cleaned = re.compile(r'(\d)[Zz](\d|\b)').sub(r'\g<1>2\g<2>', cleaned)
|
||||
cleaned = re.compile(r'(\b|\d)[Zz](\d)').sub(r'\g<1>2\g<2>', cleaned)
|
||||
|
||||
# letter B flanked by digits -> 8
|
||||
cleaned = re.compile(r'(\d)B(\d|\b)').sub(r'\g<1>8\g<2>', cleaned)
|
||||
cleaned = re.compile(r'(\b|\d)B(\d)').sub(r'\g<1>8\g<2>', cleaned)
|
||||
|
||||
return cleaned
|
||||
|
||||
def extract_expired_date(text_lines):
|
||||
"""Return (formatted_date, line_index, source_line). Prioritises BB/EXP + DDMMYYYY or DD MM YYYY."""
|
||||
if not text_lines:
|
||||
return None, None, None
|
||||
|
||||
cleaned_lines = [clean_date_line(line) for line in text_lines]
|
||||
|
||||
def pick(match, idx, cleaned_line, formatter=None):
|
||||
raw = match.group(0)
|
||||
original_line = text_lines[idx].strip()
|
||||
if match.lastindex and match.lastindex >= 3:
|
||||
formatted = f"{match.group(1)}/{match.group(2)}/{match.group(3)}"
|
||||
elif match.lastindex and match.lastindex >= 1 and match.group(1).isdigit():
|
||||
digits = match.group(1)
|
||||
if len(digits) == 8:
|
||||
formatted = format_ddmmyyyy(digits)
|
||||
elif len(digits) == 6:
|
||||
formatted = format_ddmmyy(digits)
|
||||
else:
|
||||
formatted = digits
|
||||
elif formatter:
|
||||
formatted = formatter(raw)
|
||||
else:
|
||||
formatted = raw.strip().upper()
|
||||
return formatted, idx, original_line
|
||||
|
||||
# 1) BB/EXP keyword lines — compact DDMMYYYY (e.g. BB05032027, EXP 05032027)
|
||||
for idx, line in enumerate(cleaned_lines):
|
||||
if not line_has_exp_keyword(line):
|
||||
continue
|
||||
match = BB_ATTACHED_DATE_RE.search(line) or DDMMYYYY_RE.search(line)
|
||||
if match:
|
||||
return pick(match, idx, line)
|
||||
|
||||
# 2) BB/EXP keyword lines — spaced DD MM YYYY (e.g. BB 05 03 2027)
|
||||
for idx, line in enumerate(cleaned_lines):
|
||||
if not line_has_exp_keyword(line):
|
||||
continue
|
||||
match = DD_MM_YYYY_RE.search(line)
|
||||
if match:
|
||||
return pick(match, idx, line)
|
||||
|
||||
# 3) Keyword + 6–8 digit run (BB05032027 via keyword_digits)
|
||||
for idx, line in enumerate(cleaned_lines):
|
||||
match = KEYWORD_DIGITS_RE.search(line)
|
||||
if match:
|
||||
digits = match.group(1)
|
||||
if len(digits) == 8 and is_valid_ddmmyyyy_digits(digits):
|
||||
return format_ddmmyyyy(digits), idx, text_lines[idx].strip()
|
||||
if len(digits) == 6:
|
||||
return format_ddmmyy(digits), idx, text_lines[idx].strip()
|
||||
|
||||
# 3.5) BB/EXP keyword lines — lenient check for unclear/noisy date formats (e.g. BB 02J 132027)
|
||||
for idx, line in enumerate(cleaned_lines):
|
||||
if not line_has_exp_keyword(line):
|
||||
continue
|
||||
match = LENIENT_DATE_RE.search(line)
|
||||
if match:
|
||||
return pick(match, idx, line)
|
||||
|
||||
# 4) Any line — spaced DD MM YYYY
|
||||
for idx, line in enumerate(cleaned_lines):
|
||||
match = DD_MM_YYYY_RE.search(line)
|
||||
if match:
|
||||
return pick(match, idx, line)
|
||||
|
||||
# 5) Any line — compact DDMMYYYY (skip likely SKU: same line has 8-digit product code context)
|
||||
for idx, line in enumerate(cleaned_lines):
|
||||
for match in DDMMYYYY_RE.finditer(line):
|
||||
digits = f"{match.group(1)}{match.group(2)}{match.group(3)}"
|
||||
if is_valid_ddmmyyyy_digits(digits):
|
||||
# Skip if this 8-digit block is the only digits and looks like SKU on label top
|
||||
if re.search(r'\b\d{8}\b', line) and not line_has_exp_keyword(line):
|
||||
if re.search(r'(?:nugget|chicken|fiesta|champ|okey|akumo|frozen|gr)', line, re.I):
|
||||
continue
|
||||
return format_ddmmyyyy(digits), idx, text_lines[idx].strip()
|
||||
|
||||
# 6) Legacy patterns (slashes, month names, etc.)
|
||||
date_patterns = [
|
||||
r'\b\d{2}[-./]\d{2}[-./]\d{2,4}\b',
|
||||
r'\b\d{4}[-./]\d{2}[-./]\d{2}\b',
|
||||
r'\b\d{2}\s+(?:JAN|FEB|MAR|APR|MAY|JUN|JUL|AUG|SEP|OCT|NOV|DEC)[a-zA-Z]*\s+\d{2,4}\b',
|
||||
]
|
||||
for idx, line in enumerate(cleaned_lines):
|
||||
if not line_has_exp_keyword(line):
|
||||
continue
|
||||
for pat in date_patterns:
|
||||
match = re.search(pat, line, re.IGNORECASE)
|
||||
if match:
|
||||
return match.group(0).upper(), idx, text_lines[idx].strip()
|
||||
|
||||
return None, None, None
|
||||
|
||||
def line_contains_expired_date(line: str, expired_date: str) -> bool:
|
||||
if not line or not expired_date:
|
||||
return False
|
||||
digits_only = re.sub(r"\D", "", expired_date)
|
||||
line_digits = re.sub(r"\D", "", line)
|
||||
if len(digits_only) >= 6 and digits_only in line_digits:
|
||||
return True
|
||||
compact = expired_date.replace("/", "")
|
||||
return compact in line.replace(" ", "") or expired_date in line
|
||||
|
||||
def find_expired_crop_index(text_lines, expired_idx, expired_date, polys_len):
|
||||
"""Pick OCR box index for cropping; prefer the line that actually contains the date."""
|
||||
if not expired_date or polys_len <= 0:
|
||||
return None
|
||||
|
||||
if (
|
||||
expired_idx is not None
|
||||
and expired_idx < polys_len
|
||||
and expired_idx < len(text_lines)
|
||||
and line_contains_expired_date(text_lines[expired_idx], expired_date)
|
||||
):
|
||||
return expired_idx
|
||||
|
||||
keyword_match = None
|
||||
for idx, line in enumerate(text_lines):
|
||||
if idx >= polys_len:
|
||||
break
|
||||
if not line_contains_expired_date(line, expired_date):
|
||||
continue
|
||||
if line_has_exp_keyword(line):
|
||||
return idx
|
||||
if keyword_match is None:
|
||||
keyword_match = idx
|
||||
|
||||
if keyword_match is not None:
|
||||
return keyword_match
|
||||
|
||||
if expired_idx is not None and expired_idx < polys_len:
|
||||
return expired_idx
|
||||
return None
|
||||
|
||||
def ocr_coordinate_image(res_entry, fallback_image: Image.Image) -> Image.Image:
|
||||
"""Image in the same pixel space as rec_polys (after doc orientation + unwarping)."""
|
||||
dpr = res_entry.get("doc_preprocessor_res") or {}
|
||||
output_arr = dpr.get("output_img")
|
||||
if output_arr is not None:
|
||||
return Image.fromarray(np.asarray(output_arr)).convert("RGB")
|
||||
return fallback_image
|
||||
|
||||
def ocr_text_polys(res_entry):
|
||||
"""Recognition polygons — 1:1 aligned with rec_texts."""
|
||||
return res_entry.get("rec_polys") or res_entry.get("dt_polys") or []
|
||||
|
||||
def extract_sku(text_lines):
|
||||
# SKU is usually an 8-digit number (e.g. 12010111)
|
||||
for line in text_lines:
|
||||
match = re.search(r'\b(\d{8})\b', line)
|
||||
if match:
|
||||
return match.group(1)
|
||||
|
||||
# Try finding 7-9 digit numbers
|
||||
for line in text_lines:
|
||||
match = re.search(r'\b(\d{7,9})\b', line)
|
||||
if match:
|
||||
return match.group(1)
|
||||
return None
|
||||
|
||||
def crop_poly_region(image, poly, padding=8):
|
||||
x_coords = [float(p[0]) for p in poly]
|
||||
y_coords = [float(p[1]) for p in poly]
|
||||
|
||||
x_min = max(0, int(min(x_coords)))
|
||||
y_min = max(0, int(min(y_coords)))
|
||||
x_max = min(image.width, int(max(x_coords)))
|
||||
y_max = min(image.height, int(max(y_coords)))
|
||||
|
||||
crop_left = max(0, x_min - padding)
|
||||
crop_top = max(0, y_min - padding)
|
||||
crop_right = min(image.width, x_max + padding)
|
||||
crop_bottom = min(image.height, y_max + padding)
|
||||
|
||||
if crop_right <= crop_left or crop_bottom <= crop_top:
|
||||
return None
|
||||
|
||||
cropped = image.crop((crop_left, crop_top, crop_right, crop_bottom))
|
||||
buffered = io.BytesIO()
|
||||
cropped.save(buffered, format="JPEG")
|
||||
return "data:image/jpeg;base64," + base64.b64encode(buffered.getvalue()).decode("utf-8")
|
||||
|
||||
def extract_product_name(text_lines, classified_name=None):
|
||||
keywords = ['NUGGET', 'CHICKEN', 'CHAMP', 'FIESTA', 'AKUMO', 'ASIMO', 'OKEY', 'FRIES', 'BURGER', 'SAUSAGE', 'SOSIS', 'KARAGE', 'SIOMAY', 'BUMBU', 'RACIK']
|
||||
matches = []
|
||||
for line in text_lines:
|
||||
line_upper = line.upper()
|
||||
if any(kw in line_upper for kw in keywords):
|
||||
cleaned = re.sub(r'\b\d{8}\b', '', line)
|
||||
cleaned = re.sub(r'(?:exp|expired|tgl|expiry|bbd|before)[^ \n]*', '', cleaned, flags=re.IGNORECASE)
|
||||
cleaned = re.sub(r'\b\d{2}[-./]\d{2}[-./]\d{2,4}\b', '', cleaned)
|
||||
cleaned = cleaned.strip()
|
||||
if len(cleaned) > 3:
|
||||
matches.append(cleaned)
|
||||
|
||||
if matches:
|
||||
return max(matches, key=len)
|
||||
|
||||
if classified_name:
|
||||
return classified_name
|
||||
|
||||
candidate_lines = [l for l in text_lines if not re.match(r'^\d+$', l) and len(l) > 3]
|
||||
if candidate_lines:
|
||||
return max(candidate_lines, key=len)
|
||||
|
||||
return "Unknown Product"
|
||||
|
||||
@app.post("/classify-ocr")
|
||||
async def classify_ocr(payload: ScanRequest):
|
||||
try:
|
||||
# Decode image
|
||||
from PIL import ImageOps
|
||||
img_data = base64.b64decode(payload.image_base64.split(",")[-1])
|
||||
raw_image = Image.open(io.BytesIO(img_data))
|
||||
image = ImageOps.exif_transpose(raw_image).convert("RGB")
|
||||
|
||||
# 1. Run YOLO Classification
|
||||
classification_result = {}
|
||||
top1_name = None
|
||||
if yolo_model:
|
||||
results = yolo_model(image)
|
||||
probs = results[0].probs
|
||||
top1_idx = probs.top1
|
||||
top1_conf = float(probs.top1conf)
|
||||
top1_name = results[0].names[top1_idx]
|
||||
|
||||
all_probs = []
|
||||
for idx, val in enumerate(probs.data):
|
||||
all_probs.append({
|
||||
"name": results[0].names[idx],
|
||||
"confidence": float(val)
|
||||
})
|
||||
all_probs.sort(key=lambda x: x["confidence"], reverse=True)
|
||||
|
||||
classification_result = {
|
||||
"top1_name": top1_name,
|
||||
"top1_confidence": top1_conf,
|
||||
"all_probabilities": all_probs
|
||||
}
|
||||
else:
|
||||
classification_result = {
|
||||
"error": "YOLO model not loaded"
|
||||
}
|
||||
|
||||
# 2. Run PaddleOCR
|
||||
ocr_result = {}
|
||||
if ocr:
|
||||
img_arr = np.array(image)
|
||||
# Use predict method and convert generator to list
|
||||
res_list = list(ocr.predict(img_arr))
|
||||
|
||||
text_lines = []
|
||||
if res_list and len(res_list) > 0:
|
||||
text_lines = res_list[0].get("rec_texts", [])
|
||||
|
||||
sku = extract_sku(text_lines)
|
||||
expired_date, expired_idx, expired_source_line = extract_expired_date(text_lines)
|
||||
product_name = extract_product_name(text_lines, top1_name)
|
||||
|
||||
res_entry = res_list[0] if res_list else {}
|
||||
coord_image = ocr_coordinate_image(res_entry, image)
|
||||
text_polys = ocr_text_polys(res_entry)
|
||||
crop_idx = find_expired_crop_index(
|
||||
text_lines, expired_idx, expired_date, len(text_polys)
|
||||
)
|
||||
|
||||
# Create visual OCR image with bounding boxes
|
||||
vis_image_b64 = None
|
||||
try:
|
||||
vis_image = coord_image.copy()
|
||||
from PIL import ImageDraw, ImageFont
|
||||
draw = ImageDraw.Draw(vis_image)
|
||||
|
||||
try:
|
||||
font = ImageFont.load_default()
|
||||
except:
|
||||
font = None
|
||||
|
||||
for idx, poly in enumerate(text_polys):
|
||||
is_expired = (crop_idx is not None and idx == crop_idx)
|
||||
pts = [(float(p[0]), float(p[1])) for p in poly]
|
||||
|
||||
if is_expired:
|
||||
color = (245, 158, 11) # Amber
|
||||
label = "EXP"
|
||||
else:
|
||||
color = (13, 148, 136) # Teal
|
||||
label = "TEXT"
|
||||
|
||||
draw.polygon(pts, outline=color, width=3)
|
||||
|
||||
x0, y0 = pts[0]
|
||||
label_w = 32 if label == "EXP" else 38
|
||||
draw.rectangle([x0, y0 - 15, x0 + label_w, y0], fill=color)
|
||||
|
||||
if font:
|
||||
draw.text((x0 + 4, y0 - 14), label, fill=(255, 255, 255), font=font)
|
||||
else:
|
||||
draw.text((x0 + 4, y0 - 14), label, fill=(255, 255, 255))
|
||||
|
||||
buffered = io.BytesIO()
|
||||
vis_image.save(buffered, format="JPEG")
|
||||
vis_image_b64 = "data:image/jpeg;base64," + base64.b64encode(buffered.getvalue()).decode("utf-8")
|
||||
except Exception as draw_err:
|
||||
print(f"Error drawing visual OCR: {draw_err}")
|
||||
traceback.print_exc()
|
||||
|
||||
# Crop expired date OCR region for summary verification
|
||||
expired_date_crop_b64 = None
|
||||
try:
|
||||
if crop_idx is not None and crop_idx < len(text_polys):
|
||||
expired_date_crop_b64 = crop_poly_region(coord_image, text_polys[crop_idx])
|
||||
except Exception as crop_err:
|
||||
print(f"Error cropping expired date image: {crop_err}")
|
||||
traceback.print_exc()
|
||||
|
||||
# 3. Call Spotting API
|
||||
spotting_image_b64 = None
|
||||
try:
|
||||
img_b64_only = payload.image_base64.split(",")[-1]
|
||||
spotting_payload = {
|
||||
"file": img_b64_only,
|
||||
"matchHistoryJob": False,
|
||||
"useLayoutDetection": False,
|
||||
"fileType": 1,
|
||||
"useDocUnwarping": False,
|
||||
"useDocOrientationClassify": False,
|
||||
"promptLabel": "spotting"
|
||||
}
|
||||
spotting_url = "http://localhost:8090/layout-parsing"
|
||||
spotting_resp = requests.post(spotting_url, json=spotting_payload, timeout=60)
|
||||
if spotting_resp.status_code == 200:
|
||||
spotting_data = spotting_resp.json()
|
||||
if spotting_data.get("errorCode") == 0:
|
||||
layout_results = spotting_data.get("result", {}).get("layoutParsingResults", [])
|
||||
if layout_results:
|
||||
page0 = layout_results[0]
|
||||
out_imgs = page0.get("outputImages", {})
|
||||
spotting_img = out_imgs.get("spotting_res_img")
|
||||
if spotting_img:
|
||||
spotting_image_b64 = "data:image/jpeg;base64," + spotting_img
|
||||
else:
|
||||
print(f"Spotting API error: {spotting_resp.text}")
|
||||
except Exception as spotting_err:
|
||||
print(f"Error calling spotting API: {spotting_err}")
|
||||
traceback.print_exc()
|
||||
|
||||
ocr_result = {
|
||||
"text_lines": text_lines,
|
||||
"extracted_product_name": product_name,
|
||||
"extracted_sku": sku,
|
||||
"extracted_expired_date": expired_date,
|
||||
"expired_line_index": crop_idx,
|
||||
"expired_source_line": expired_source_line,
|
||||
"expired_date_crop_base64": expired_date_crop_b64,
|
||||
"vis_image_base64": vis_image_b64,
|
||||
"spotting_image_base64": spotting_image_b64
|
||||
}
|
||||
else:
|
||||
ocr_result = {
|
||||
"error": "PaddleOCR not loaded"
|
||||
}
|
||||
|
||||
return {
|
||||
"classification": classification_result,
|
||||
"ocr": ocr_result
|
||||
}
|
||||
|
||||
except Exception as e:
|
||||
traceback.print_exc()
|
||||
raise HTTPException(status_code=500, detail=str(e))
|
||||
|
||||
if __name__ == "__main__":
|
||||
import uvicorn
|
||||
uvicorn.run(app, host="0.0.0.0", port=8120)
|
||||
@@ -0,0 +1,85 @@
|
||||
pipeline_name: PaddleOCR-VL-1.6
|
||||
|
||||
batch_size: 64
|
||||
|
||||
use_queues: True
|
||||
|
||||
use_doc_preprocessor: True
|
||||
use_layout_detection: True
|
||||
use_chart_recognition: False
|
||||
use_seal_recognition: False
|
||||
format_block_content: False
|
||||
merge_layout_blocks: True
|
||||
markdown_ignore_labels:
|
||||
- number
|
||||
- footnote
|
||||
- header
|
||||
- header_image
|
||||
- footer
|
||||
- footer_image
|
||||
- aside_text
|
||||
|
||||
SubModules:
|
||||
LayoutDetection:
|
||||
module_name: layout_detection
|
||||
model_name: PP-DocLayoutV3
|
||||
model_dir: null
|
||||
batch_size: 8
|
||||
threshold: 0.3
|
||||
layout_nms: True
|
||||
layout_unclip_ratio: [1.0, 1.0]
|
||||
layout_merge_bboxes_mode:
|
||||
0: "union"
|
||||
1: "union"
|
||||
2: "union"
|
||||
3: "large"
|
||||
4: "union"
|
||||
5: "large"
|
||||
6: "large"
|
||||
7: "union"
|
||||
8: "union"
|
||||
9: "union"
|
||||
10: "union"
|
||||
11: "union"
|
||||
12: "union"
|
||||
13: "union"
|
||||
14: "union"
|
||||
15: "large"
|
||||
16: "union"
|
||||
17: "large"
|
||||
18: "union"
|
||||
19: "union"
|
||||
20: "union"
|
||||
21: "union"
|
||||
22: "union"
|
||||
23: "union"
|
||||
24: "union"
|
||||
VLRecognition:
|
||||
module_name: vl_recognition
|
||||
model_name: PaddleOCR-VL-1.6-0.9B
|
||||
model_dir: null
|
||||
batch_size: 4096
|
||||
genai_config:
|
||||
backend: vllm-server
|
||||
server_url: http://127.0.0.1:8118/v1
|
||||
|
||||
SubPipelines:
|
||||
DocPreprocessor:
|
||||
pipeline_name: doc_preprocessor
|
||||
batch_size: 8
|
||||
use_doc_orientation_classify: True
|
||||
use_doc_unwarping: True
|
||||
SubModules:
|
||||
DocOrientationClassify:
|
||||
module_name: doc_text_orientation
|
||||
model_name: PP-LCNet_x1_0_doc_ori
|
||||
model_dir: null
|
||||
batch_size: 8
|
||||
DocUnwarping:
|
||||
module_name: image_unwarping
|
||||
model_name: UVDoc
|
||||
model_dir: null
|
||||
|
||||
Serving:
|
||||
extra:
|
||||
max_num_input_imgs: null
|
||||
@@ -0,0 +1,20 @@
|
||||
from PIL import Image
|
||||
import numpy as np
|
||||
from paddleocr import PaddleOCR
|
||||
|
||||
ocr = PaddleOCR(use_textline_orientation=True, lang='en')
|
||||
image = Image.open('/app/config/test_img.jpeg').convert('RGB')
|
||||
img_arr = np.array(image)
|
||||
res_list = list(ocr.predict(img_arr))
|
||||
|
||||
texts = res_list[0].get('rec_texts', [])
|
||||
dt_polys = res_list[0].get('dt_polys', [])
|
||||
|
||||
for idx, (text, poly) in enumerate(zip(texts, dt_polys)):
|
||||
if 'BB05032027' in text or 'BB' in text:
|
||||
print(f"Match: {text}")
|
||||
print("Raw poly:")
|
||||
print(poly)
|
||||
print("Pts computed:")
|
||||
pts = [(float(p[0]), float(p[1])) for p in poly]
|
||||
print(pts)
|
||||
@@ -0,0 +1,69 @@
|
||||
import re
|
||||
|
||||
def clean_date_line(line: str) -> str:
|
||||
# 1) Replace "1)" with "0"
|
||||
cleaned = line.replace("1)", "0")
|
||||
|
||||
# 2) Replace "()" with "0"
|
||||
cleaned = cleaned.replace("()", "0")
|
||||
|
||||
# Clean BB misrecognitions (convert B8, 8B, 88 to BB when followed by digits)
|
||||
cleaned = re.sub(r'\b(?:B8|8B|88)(?=\d)', 'BB', cleaned)
|
||||
cleaned = re.sub(r'^(?:B8|8B|88)(?=\d)', 'BB', cleaned)
|
||||
|
||||
# Clean 012 month misrecognition (e.g. 020122027 -> 02022027)
|
||||
cleaned = re.sub(r'(?<!\d)(\d{1,2})012(20\d{2})(?!\d)', r'\g<1>02\g<2>', cleaned)
|
||||
cleaned = re.sub(r'(?<!\d)(\d{1,2})([-./\s]+)012([-./\s]+)(20\d{2})(?!\d)', r'\g<1>\g<2>02\g<3>\g<4>', cleaned)
|
||||
|
||||
# Clean 112 month misrecognition (e.g. 021122027 -> 02022027)
|
||||
cleaned = re.sub(r'(?<!\d)(\d{1,2})112(20\d{2})(?!\d)', r'\g<1>02\g<2>', cleaned)
|
||||
cleaned = re.sub(r'(?<!\d)(\d{1,2})([-./\s]+)112([-./\s]+)(20\d{2})(?!\d)', r'\g<1>\g<2>02\g<3>\g<4>', cleaned)
|
||||
|
||||
# Run contextual replacements
|
||||
for _ in range(3):
|
||||
# letter o/O flanked by digits or boundary -> 0
|
||||
cleaned = re.compile(r'(\d)[oO](\d|\b)').sub(r'\g<1>0\g<2>', cleaned)
|
||||
cleaned = re.compile(r'(\b|\d)[oO](\d)').sub(r'\g<1>0\g<2>', cleaned)
|
||||
|
||||
# letter I/i/l/| flanked by digits -> 1
|
||||
cleaned = re.compile(r'(\d)[Ii|l](\d|\b)').sub(r'\g<1>1\g<2>', cleaned)
|
||||
cleaned = re.compile(r'(\b|\d)[Ii|l](\d)').sub(r'\g<1>1\g<2>', cleaned)
|
||||
|
||||
# letter S/s flanked by digits -> 5
|
||||
cleaned = re.compile(r'(\d)[Ss](\d|\b)').sub(r'\g<1>5\g<2>', cleaned)
|
||||
cleaned = re.compile(r'(\b|\d)[Ss](\d)').sub(r'\g<1>5\g<2>', cleaned)
|
||||
|
||||
# letter Z/z flanked by digits -> 2
|
||||
cleaned = re.compile(r'(\d)[Zz](\d|\b)').sub(r'\g<1>2\g<2>', cleaned)
|
||||
cleaned = re.compile(r'(\b|\d)[Zz](\d)').sub(r'\g<1>2\g<2>', cleaned)
|
||||
|
||||
# letter B flanked by digits -> 8
|
||||
cleaned = re.compile(r'(\d)B(\d|\b)').sub(r'\g<1>8\g<2>', cleaned)
|
||||
cleaned = re.compile(r'(\b|\d)B(\d)').sub(r'\g<1>8\g<2>', cleaned)
|
||||
|
||||
return cleaned
|
||||
|
||||
test_cases = [
|
||||
"231)92026",
|
||||
"23o92026",
|
||||
"23O92026",
|
||||
"2309202l",
|
||||
"2309202I",
|
||||
"230920Z6",
|
||||
"230920s6",
|
||||
"2309202B",
|
||||
"BB 231)92026",
|
||||
"BB: 23()92026",
|
||||
"12010111",
|
||||
"B8021122027",
|
||||
"88021122027",
|
||||
"020122027",
|
||||
"BB 02/012/2027",
|
||||
"021122027",
|
||||
"BB 02/112/2027"
|
||||
]
|
||||
|
||||
for tc in test_cases:
|
||||
cleaned = clean_date_line(tc)
|
||||
print(f"Original: {tc:<18} -> Cleaned: {cleaned}")
|
||||
|
||||
@@ -0,0 +1,7 @@
|
||||
# vLLM backend tuning for paddleocr genai_server
|
||||
# Docs: https://www.paddleocr.ai/latest/en/version3.x/pipeline_usage/PaddleOCR-VL.html#331-server-side-parameter-adjustment
|
||||
gpu-memory-utilization: 0.6
|
||||
max-num-seqs: 4
|
||||
enforce-eager: true
|
||||
max-model-len: 2048
|
||||
max-num-batched-tokens: 2048
|
||||
@@ -0,0 +1,8 @@
|
||||
#!/bin/bash
|
||||
# Create the target directory inside the Next.js app
|
||||
mkdir -p pfm-web-app/public/do-pfm
|
||||
|
||||
# Copy DO-PFM images
|
||||
cp -v PaddleOCR-VL-1.6_Online_Demo/examples/do-pfm/*.jpg pfm-web-app/public/do-pfm/
|
||||
|
||||
echo "DO-PFM examples copied successfully!"
|
||||
@@ -0,0 +1,110 @@
|
||||
# Database Entity Relationship Diagram (ERD)
|
||||
|
||||
This document describes the PostgreSQL database schema used to store OCR documents, parsed layout elements, inline cell edits, and row flagging status for the DO-PFM system.
|
||||
|
||||
## Relationship Diagram
|
||||
|
||||
```mermaid
|
||||
erDiagram
|
||||
documents {
|
||||
integer id PK "SERIAL"
|
||||
varchar filename UK "VARCHAR(255)"
|
||||
timestamp upload_time "TIMESTAMP"
|
||||
integer size "INTEGER"
|
||||
boolean parsed "BOOLEAN"
|
||||
jsonb metadata "JSONB"
|
||||
jsonb layout_parsing_result "JSONB"
|
||||
boolean is_sample "BOOLEAN"
|
||||
varchar file_hash "VARCHAR(64)"
|
||||
}
|
||||
|
||||
ocr_items {
|
||||
integer id PK "SERIAL"
|
||||
integer document_id FK "INTEGER"
|
||||
integer row_index "INTEGER"
|
||||
varchar kode_barang_original "VARCHAR(255)"
|
||||
varchar kode_barang "VARCHAR(255)"
|
||||
varchar nama_barang "VARCHAR(255)"
|
||||
varchar banyak_original "VARCHAR(255)"
|
||||
varchar banyak "VARCHAR(255)"
|
||||
varchar jumlah_original "VARCHAR(255)"
|
||||
varchar jumlah "VARCHAR(255)"
|
||||
boolean is_flagged "BOOLEAN"
|
||||
varchar remark "VARCHAR(1000)"
|
||||
}
|
||||
|
||||
documents ||--o{ ocr_items : "has"
|
||||
|
||||
vendors {
|
||||
integer id PK "SERIAL"
|
||||
varchar name UK "VARCHAR(255)"
|
||||
timestamp created_at "TIMESTAMP"
|
||||
}
|
||||
|
||||
customers {
|
||||
integer id PK "SERIAL"
|
||||
varchar name UK "VARCHAR(255)"
|
||||
timestamp created_at "TIMESTAMP"
|
||||
}
|
||||
```
|
||||
|
||||
## Schema Definitions
|
||||
|
||||
### 1. `documents` Table
|
||||
Stores parsed OCR files (both static sample pages and user-uploaded invoices/documents).
|
||||
|
||||
| Column | Type | Constraints | Description |
|
||||
|---|---|---|---|
|
||||
| `id` | `SERIAL` | `PRIMARY KEY` | Unique autoincrement ID of the document. |
|
||||
| `filename` | `VARCHAR(255)` | `UNIQUE`, `NOT NULL` | The unique name of the document file. |
|
||||
| `upload_time` | `TIMESTAMP` | `DEFAULT NOW()`, `NOT NULL` | The timestamp of the file upload. |
|
||||
| `size` | `INTEGER` | `DEFAULT 0`, `NOT NULL` | The file size in bytes. |
|
||||
| `parsed` | `BOOLEAN` | `DEFAULT FALSE`, `NOT NULL` | Indicates whether the document layout parsing has completed. |
|
||||
| `metadata` | `JSONB` | | Structured general metadata (Vendor, Customer, PO, SO, DO, etc.). |
|
||||
| `layout_parsing_result` | `JSONB` | | Raw layout parser response JSON from pipeline backend. |
|
||||
| `is_sample` | `BOOLEAN` | `DEFAULT FALSE`, `NOT NULL` | True if the file belongs to the pre-seeded static sample pages. |
|
||||
| `file_hash` | `VARCHAR(64)` | | SHA-256 hash of the document file contents. |
|
||||
|
||||
---
|
||||
|
||||
### 2. `ocr_items` Table
|
||||
Stores the extracted row items from tabular components of the document, supporting inline modifications and flagging details.
|
||||
|
||||
| Column | Type | Constraints | Description |
|
||||
|---|---|---|---|
|
||||
| `id` | `SERIAL` | `PRIMARY KEY` | Unique autoincrement ID of the item row. |
|
||||
| `document_id` | `INTEGER` | `REFERENCES documents(id) ON DELETE CASCADE`, `NOT NULL` | The associated document ID. |
|
||||
| `row_index` | `INTEGER` | `NOT NULL` | The index of the item row in the document table list (0-indexed). |
|
||||
| `kode_barang_original` | `VARCHAR(255)` | | The initial "Kode Barang" value extracted directly from OCR. |
|
||||
| `kode_barang` | `VARCHAR(255)` | | The edited/current "Kode Barang" value. |
|
||||
| `nama_barang` | `VARCHAR(255)` | | The "Nama Barang" value (read-only reference). |
|
||||
| `banyak_original` | `VARCHAR(255)` | | The initial "Banyak" value extracted from OCR. |
|
||||
| `banyak` | `VARCHAR(255)` | | The edited/current "Banyak" value. |
|
||||
| `jumlah_original` | `VARCHAR(255)` | | The initial "Jumlah" value extracted from OCR. |
|
||||
| `jumlah` | `VARCHAR(255)` | | The edited/current "Jumlah" value. |
|
||||
| `is_flagged` | `BOOLEAN` | `DEFAULT FALSE`, `NOT NULL` | True if the line item is flagged/strikethrough ("dicoret"). |
|
||||
| `remark` | `VARCHAR(1000)` | | Custom notes/remarks provided for flagging. |
|
||||
|
||||
* **Unique Constraints**: A unique index on `(document_id, row_index)` prevents duplicate indexes for the same page.
|
||||
|
||||
---
|
||||
|
||||
### 3. `vendors` Table
|
||||
Stores the Vendor Master registry.
|
||||
|
||||
| Column | Type | Constraints | Description |
|
||||
|---|---|---|---|
|
||||
| `id` | `SERIAL` | `PRIMARY KEY` | Unique autoincrement ID of the vendor. |
|
||||
| `name` | `VARCHAR(255)` | `UNIQUE`, `NOT NULL` | The unique name of the vendor (e.g. including kawasan/address). |
|
||||
| `created_at` | `TIMESTAMP` | `DEFAULT NOW()`, `NOT NULL` | The registration timestamp. |
|
||||
|
||||
---
|
||||
|
||||
### 4. `customers` Table
|
||||
Stores the Customer Master registry.
|
||||
|
||||
| Column | Type | Constraints | Description |
|
||||
|---|---|---|---|
|
||||
| `id` | `SERIAL` | `PRIMARY KEY` | Unique autoincrement ID of the customer. |
|
||||
| `name` | `VARCHAR(255)` | `UNIQUE`, `NOT NULL` | The unique name of the customer (e.g. including branch/address). |
|
||||
| `created_at` | `TIMESTAMP` | `DEFAULT NOW()`, `NOT NULL` | The registration timestamp. |
|
||||
@@ -0,0 +1,29 @@
|
||||
-- Migration: 001_init_schema
|
||||
-- Description: Initialize schema for documents and ocr_items
|
||||
|
||||
CREATE TABLE IF NOT EXISTS documents (
|
||||
id SERIAL PRIMARY KEY,
|
||||
filename VARCHAR(255) UNIQUE NOT NULL,
|
||||
upload_time TIMESTAMP NOT NULL DEFAULT NOW(),
|
||||
size INTEGER NOT NULL DEFAULT 0,
|
||||
parsed BOOLEAN NOT NULL DEFAULT FALSE,
|
||||
metadata JSONB,
|
||||
layout_parsing_result JSONB,
|
||||
is_sample BOOLEAN NOT NULL DEFAULT FALSE
|
||||
);
|
||||
|
||||
CREATE TABLE IF NOT EXISTS ocr_items (
|
||||
id SERIAL PRIMARY KEY,
|
||||
document_id INTEGER NOT NULL REFERENCES documents(id) ON DELETE CASCADE,
|
||||
row_index INTEGER NOT NULL,
|
||||
kode_barang_original VARCHAR(255),
|
||||
kode_barang VARCHAR(255),
|
||||
nama_barang VARCHAR(255),
|
||||
banyak_original VARCHAR(255),
|
||||
banyak VARCHAR(255),
|
||||
jumlah_original VARCHAR(255),
|
||||
jumlah VARCHAR(255),
|
||||
is_flagged BOOLEAN NOT NULL DEFAULT FALSE,
|
||||
remark VARCHAR(1000),
|
||||
UNIQUE(document_id, row_index)
|
||||
);
|
||||
@@ -0,0 +1,4 @@
|
||||
-- Migration: 002_add_file_hash
|
||||
-- Description: Add file_hash column to documents table for duplicate content detection
|
||||
|
||||
ALTER TABLE documents ADD COLUMN IF NOT EXISTS file_hash VARCHAR(64);
|
||||
@@ -0,0 +1,12 @@
|
||||
-- Migration: 003_create_vendor_master
|
||||
-- Description: Create vendors table and seed the initial vendor entry
|
||||
|
||||
CREATE TABLE IF NOT EXISTS vendors (
|
||||
id SERIAL PRIMARY KEY,
|
||||
name VARCHAR(255) UNIQUE NOT NULL,
|
||||
created_at TIMESTAMP NOT NULL DEFAULT NOW()
|
||||
);
|
||||
|
||||
INSERT INTO vendors (name)
|
||||
VALUES ('PT. CHAROEN POKPHAND INDONESIA Tbk KAWASAN INDUSTRI MODERN, BANTEN')
|
||||
ON CONFLICT (name) DO NOTHING;
|
||||
@@ -0,0 +1,12 @@
|
||||
-- Migration: 004_create_customer_master
|
||||
-- Description: Create customers table and seed the initial customer entry
|
||||
|
||||
CREATE TABLE IF NOT EXISTS customers (
|
||||
id SERIAL PRIMARY KEY,
|
||||
name VARCHAR(255) UNIQUE NOT NULL,
|
||||
created_at TIMESTAMP NOT NULL DEFAULT NOW()
|
||||
);
|
||||
|
||||
INSERT INTO customers (name)
|
||||
VALUES ('PT.PRIMAFOOD INTERNATIONAL, JL. ANCOL BARAT VIII/1, ANCOL, PADEMANGAN, JAKARTA UTARA, 14430')
|
||||
ON CONFLICT (name) DO NOTHING;
|
||||
@@ -0,0 +1,244 @@
|
||||
-- Migration: 005_create_sku_master
|
||||
-- Description: Create sku_master table and seed the initial SKU entries
|
||||
|
||||
CREATE TABLE IF NOT EXISTS sku_master (
|
||||
id SERIAL PRIMARY KEY,
|
||||
no_sku VARCHAR(255) UNIQUE NOT NULL,
|
||||
nama_item VARCHAR(255) NOT NULL,
|
||||
created_at TIMESTAMP NOT NULL DEFAULT NOW()
|
||||
);
|
||||
|
||||
INSERT INTO sku_master (no_sku, nama_item) VALUES
|
||||
('11048006', 'BEBEK PARTING-NEW(*)'),
|
||||
('11110059', 'CEKER BERKUKU FROZEN PACK 1 KG(*)'),
|
||||
('11110074', 'CEKER 1 KG FROZEN (20 PAC/KARUNG)(*)'),
|
||||
('11140051', 'AMPELA FROZEN PACK 1 KG(*)'),
|
||||
('11140062', 'AMPELA 1 KG FROZEN (20 PAC/KARUNG)(*)'),
|
||||
('11148002', 'AMPELA BEBEK FROZEN 1 KG/PACK (NEW)(*)'),
|
||||
('11150052', 'HATI FROZEN PACK 1 KG(*)'),
|
||||
('11150055', 'JANTUNG FROZEN PACK 1 KG(*)'),
|
||||
('11150064', 'HATI 1 KG FROZEN (20 PAC/KARUNG)(*)'),
|
||||
('11150065', 'JANTUNG 1 KG FROZEN (20 PAC/KARUNG)(*)'),
|
||||
('11310012', 'AYAM SIZE 0 (0.6-0.7)KG(*)'),
|
||||
('11310013', 'AYAM SIZE1 FROZEN (0.75-0.8) KG(*)'),
|
||||
('11310014', 'AYAM SIZE 2 FROZEN (0.8-0.9)KG(*)'),
|
||||
('11310016', 'AYAM SIZE Z PR FROZ(*)'),
|
||||
('11310017', 'AYAM SIZE 0 PR FROZEN(*)'),
|
||||
('11310018', 'AYAM SIZE 1 PR FROZEN(*)'),
|
||||
('11310019', 'AYAM SIZE 2 PR FROZEN(*)'),
|
||||
('11310021', 'AYAM SIZE BESAR (B) FROZ (1-1.1)KG/PC(*)'),
|
||||
('11310022', 'AYAM SIZE A PR (0.9-1)KG/PC(*)'),
|
||||
('11310024', 'AYAM SIZE A FROZEN (0.9-1)KG/PC(*)'),
|
||||
('11310025', 'AYAM SIZE SUPER (C) FROZ(1.1-1.2)KG/ PC(*)'),
|
||||
('11310026', 'AYAM SIZE JUMBO (D) FROZ (1.2- 1.3)KG/PC(*)'),
|
||||
('11318301', 'BEBEK MUDA-BD1(1.0-1.1 KG)-NEW(*)'),
|
||||
('11318306', 'CP DUCK PEKING 1.5-1.6 KG/PC(*)'),
|
||||
('11318308', 'BEBEK PEKING SPR BD5(1.7 -1.8 )Kg-NEW(*)'),
|
||||
('11410043', 'PARTING 10 SIZE D FRESH BENSU 1.25 KG/PAC(*)'),
|
||||
('11420055', 'PARTING 12 ALL SIZE FROZ/PAC(*)'),
|
||||
('11600053', 'BONELESS LEG FROZEN 1 KG(*)'),
|
||||
('11620056', 'SBL (FILLET PAHA) 1 KG(*)'),
|
||||
('11640053', 'PAHA UTUH (1 KG)(*)'),
|
||||
('11650053', 'PAHA ATAS 1 KG(*)'),
|
||||
('11660050', 'PAHA BAWAH (1 KG)(*)'),
|
||||
('11690053', 'SBB (FILLET DADA )1 KG(*)'),
|
||||
('11690081', 'SBB JUMBO FZ (2.0 - 2.2 KG/PAC)(*)'),
|
||||
('11710051', 'DADA UTUH (1 KG)(*)'),
|
||||
('11720055', 'FULL WING FROZ PACK 1 KG(*)'),
|
||||
('11730050', 'MIDDLE WING FROZ PACK 1 KG(*)'),
|
||||
('11750050', 'FILLET MITRA 1 KG(*)'),
|
||||
('11818300', 'CP-BEBEK GORENG 400GR/PAC'),
|
||||
('11840002', 'AYAM JANTAN BKKL SZ 0 (600-700) GR/PC(*)'),
|
||||
('11959937', 'SATE AYAM FRESHMART 360 GR (PAC)'),
|
||||
('12010111', 'FIESTA CRISPY BUBBLE 400 GR/PAC'),
|
||||
('12010112', 'FIESTA CHICKEN NUGGET 400 GR/PAC'),
|
||||
('12010113', 'FIESTA CHICKEN NUGGET 200 GR/PAC'),
|
||||
('12010115', 'FIESTA NUGGET ZOO 400 GR/PAC'),
|
||||
('12010116', 'FIESTA NUGGET DINO 400 GR/PAC'),
|
||||
('12010117', 'FIESTA NUGGET HAPPY STAR 400 GR/PAC'),
|
||||
('12010119', 'FIESTA NUGGET CHEESE 123 400 GR/PAC'),
|
||||
('12010121', 'FIESTA NUGGET PIZZABC 400 GR/PAC'),
|
||||
('12010122', 'FIESTA CHEESY LOVER 400 GR/PAC'),
|
||||
('12010123', 'FIESTA GARLIC CHEESE 400 GR/PAC'),
|
||||
('12010124', 'FIESTA CHEESY CHIC W/BROCCOLI 400 GR/PAC'),
|
||||
('12010127', 'FIESTA SPICY NUGGET 400 GR/PAC'),
|
||||
('12010128', 'FIESTA VOLCANO CHEESE 400 GR/PAC'),
|
||||
('12010129', 'FIESTA CHEESY BOMBS CHICKEN NUGGET 400 GR'),
|
||||
('12010402', 'GOLDEN FIESTA NUGGET W/PINEAPPLE SAUCE 500 GR'),
|
||||
('12010509', 'CHAMP CRUNCHY NUGGET 450 GR/PAC'),
|
||||
('12010510', 'CHAMP NUGGET AYAM 225 GR/PAC'),
|
||||
('12010511', 'CHAMP NUGGET AYAM 450 GR/PAC'),
|
||||
('12010512', 'CHAMP NUGGET AYAM 900 GR/PAC'),
|
||||
('12010513', 'CHAMP NUGGET ABC KOMBINASI 225 GR/PAC'),
|
||||
('12010514', 'CHAMP NUGGET ABC KOMBINASI 450 GR/PAC'),
|
||||
('12010515', 'CHAMP KOIN KOMBINASI 450 GR/PAC'),
|
||||
('12010516', 'CHAMP KOIN KOMBINASI 200 GR/PAC'),
|
||||
('12010517', 'CHAMP NUGGET STICK 225 GR/PAC'),
|
||||
('12010518', 'CHAMP NUGGET STICK 450 GR/PAC'),
|
||||
('12010519', 'CHAMP NUGGET STICK 900 GR/PAC'),
|
||||
('12010520', 'CHAMP CHICKEN NUGGET BENTUK 123 450 GR/PAC'),
|
||||
('12010521', 'CHAMP NUGGET HOTZZ LEVEL 5 450 GR/PAC'),
|
||||
('12010606', 'CHAMP CRUNCHY NUGGET 225 GR/PAC'),
|
||||
('12010707', 'CHAMP MITRA NUGGET COIN 200 GR (NEW)'),
|
||||
('12010801', 'OKEY NUGGET 500GR'),
|
||||
('12012201', 'ASIMO NUGGET KOMBINASI 500 GR/PAC'),
|
||||
('12012202', 'ASIMO NUGGET KOMBINASI 1 KG/PAC'),
|
||||
('12012203', 'ASIMO NUGGET KOMBINASI 250 GR/PAC'),
|
||||
('12012501', 'AKUMO CHICKEN NAGET 250 GR'),
|
||||
('12012502', 'AKUMO CHICKEN NUGGET 500 GR'),
|
||||
('12012503', 'AKUMO CHICKEN NUGGET 1000 GR'),
|
||||
('12012504', 'AKUMO COIN 200 GR/PAC'),
|
||||
('12012505', 'AKUMO KOIN 400 GR/PAC'),
|
||||
('12020102', 'FIESTA SPICY WING 400 GR/PAC'),
|
||||
('12020401', 'GOLDEN FIESTA SP WING 500 GR'),
|
||||
('12030101', 'FIESTA STIKIE 400 GR/PAC'),
|
||||
('12030102', 'FIESTA STIKIE 200 GR/PAC'),
|
||||
('12030403', 'GOLDEN FIESTA STIKIE W/ SWEET CHILLI SAUCE 500GR'),
|
||||
('12030801', 'OKEY STICK 1000 GR'),
|
||||
('12030802', 'OKEY STICK 500GR'),
|
||||
('12032201', 'ASIMO STICK KOMBINASI 500 GR/PAC'),
|
||||
('12032202', 'ASIMO STICK KOMBINASI 1000 GR/PAC'),
|
||||
('12032203', 'ASIMO STIK KOMBINASI 250 GR/PAC'),
|
||||
('12032501', 'AKUMO CHICKEN STICK 250 GR'),
|
||||
('12032502', 'AKUMO CHICKEN STIK 500 GR'),
|
||||
('12032503', 'AKUMO CHICKEN STICK 1000 GR'),
|
||||
('12040101', 'FIESTA SCHNITZEL 400 GR/PAC'),
|
||||
('12040102', 'FIESTA CRISPY BUBBLE KATSU 400 GR/PAC'),
|
||||
('12040404', 'GOLDEN FIESTA CORDON BLEU BBQ SAUCE 500 GR'),
|
||||
('12040406', 'GOLDEN FIESTA KATSU W/CHEESE SAUCE 500 GR/PAC'),
|
||||
('12050103', 'FIESTA FRIED CHICKEN 400 GR/PAC'),
|
||||
('12050104', 'FIESTA HOT & CRISPY FRIED CHICKEN 400 GR/PAC'),
|
||||
('12050401', 'GOLDEN FIESTA CRISPY WING W/SP GLAZ SC 500 GR/PAC'),
|
||||
('12060103', 'FIESTA KARAGE 200 GR/PAC'),
|
||||
('12060104', 'FIESTA KARAGE 400 GR/PAC'),
|
||||
('12060105', 'FIESTA SPICY KARAGE 400 GR/PAC'),
|
||||
('12060402', 'GOLDEN FIESTA KARAGE CHILI SAUCE 500GR'),
|
||||
('12070101', 'FIESTA POK-POK 400 GR/PAC (NEW)'),
|
||||
('12080101', 'FIESTA SPICY CHICK 400 GR/PAC'),
|
||||
('12130102', 'FIESTA CRISPY BURGER 360 GR (NEW)'),
|
||||
('12130504', 'CHAMP BURGER 315 GR (NEW)'),
|
||||
('12140105', 'FIESTA CHICK TOFU 400 GR/PAC'),
|
||||
('12150201', 'FIESTA DS CRISPY CRUNCH 300 GR/PAC'),
|
||||
('12150501', 'CHAMP CRUNCHY HOTZZ 300 GR/PAC'),
|
||||
('12190103', 'FIESTA DELISTRIPE 400 GR/PAC'),
|
||||
('12240102', 'FIESTA CHEESY ITALIAN R/BITES 400 GR/PAC'),
|
||||
('12240103', 'FIESTA YAKINIKU R/BITES 400 GR/PAC'),
|
||||
('13010101', 'FIESTA CHICK SSG 300 GR'),
|
||||
('13010102', 'FIESTA CHICK SSG 500 GR'),
|
||||
('13010103', 'FIESTA CHICK SSG 200 GR/PAC'),
|
||||
('13010111', 'FIESTA SOSIS BRATWURST 300 GR'),
|
||||
('13010112', 'FIESTA CHEESE SSG 300 GR'),
|
||||
('13010113', 'FIESTA SOSIS CURRYWURST 300 GR'),
|
||||
('13010114', 'FIESTA SSG BOCKWURST 300GR'),
|
||||
('13010115', 'FIESTA SSG WIENER 300GR'),
|
||||
('13010116', 'FIESTA SSG ORIGINAL 300 GR'),
|
||||
('13010117', 'FIESTA SSG FRANKFURTER 300GR'),
|
||||
('13010118', 'FIESTA RTG SSG 65 GR/PAC'),
|
||||
('13010119', 'FIESTA RTG C/SPICY KOREAN 60 GR/PAC'),
|
||||
('13010120', 'FIESTA RTG C/CHEESY MELTS 65 GR/PAC'),
|
||||
('13010122', 'FIESTA RTG SAUSAGE WITH HOT LAVA 60G'),
|
||||
('13010123', 'FIESTA RTG SAUSAGE WITH CHEESE LAVA 60G'),
|
||||
('13010124', 'FIESTA RTG SAUSAGE HICKORY SAUCE 60GR'),
|
||||
('13010125', 'FIESTA RTG SAUSAGE WITH MENTAI LAVA 60GR'),
|
||||
('13010510', 'CHAMP CHICK SSG 75 GR'),
|
||||
('13010513', 'CHAMP CHICK SSG 375 GR'),
|
||||
('13010514', 'CHAMP CHICK SSG 1000 GR'),
|
||||
('13010518', 'CHAMP SSG JUMBO BAKAR 500 GR/PAC'),
|
||||
('13010519', 'CHAMP SSG BAKAR MINI 500 GR/PAC-INACT'),
|
||||
('13010521', 'CHAMP CHICK SSG 150 GR/PAC (NEW)'),
|
||||
('13010523', 'CHAMP CHICK SSG AYAM MADU 300 GR/PAC'),
|
||||
('13010524', 'CHAMP SSG JUMBO BAKAR 500 GR/PAC (NEW)'),
|
||||
('13010525', 'CHAMP SSG BAKAR MINI 500 GR/PAC (NEW)'),
|
||||
('13010809', 'OKEY CHICK SSG 500GR-INACT'),
|
||||
('13010815', 'OKEY SSG BAKAR JUMBO 500 GR/PAC (NEW)'),
|
||||
('13010816', 'OKEY SSG BAKAR MINI 500 GR/PAC (NEW)'),
|
||||
('13010817', 'OKEY SSG CHICK KOMBINASI 500 GR (EXTRA1)'),
|
||||
('13010818', 'OKEY SSG CHICK KOMBINASI 1 KG (EXTR$A2)'),
|
||||
('13012205', 'ASIMO SOSIS AYAM KOMBINASI 375 GR (PAC)'),
|
||||
('13012206', 'ASIMO SOSIS AYAM KOMBINASI 500 GR'),
|
||||
('13012207', 'ASIMO SOSIS AYAM KOMBINASI 750 GR'),
|
||||
('13012208', 'ASIMO SOSIS AYAM KOMBINASI 1000 GR'),
|
||||
('13012209', 'ASIMO SSG CHICK KOMBINASI 500 GR (EXTRA1)'),
|
||||
('13012210', 'ASIMO SSG CHICK KOMBINASI 1 KG (EXTR$A2)'),
|
||||
('13030101', 'FIESTA CHICK MEAT BALL 300 GR'),
|
||||
('13030102', 'FIESTA CHICK MEATBALL 500 GR'),
|
||||
('13030501', 'CHAMP CHICK MEATBALL 200 GR'),
|
||||
('13030502', 'CHAMP CHICK MEATBALL 500 GR'),
|
||||
('13050101', 'FIESTA SCB 250 GR'),
|
||||
('13050105', 'FIESTA CHICKEN SLICE 300 GR'),
|
||||
('13050106', 'FIESTA BEEF SLICE 300 GR'),
|
||||
('13070501', 'CHAMP BEEF SSG SERBAGUNA 150 GR'),
|
||||
('13070502', 'CHAMP BEEF SSG SERBAGUNA 375GR'),
|
||||
('13070505', 'CHAMP BEEF SSG GORENG 375 GR'),
|
||||
('13070506', 'CHAMP FRANKFURTER SSG 375GR'),
|
||||
('13100512', 'CHAMP CHICK SSG S/SANTAP ORIG 546GR (CAN)'),
|
||||
('13110504', 'CHAMP BEEF BALL 500GR'),
|
||||
('13170510', 'CHAMP BEEF BBQ SSG S/SANTAP 546GR (CAN)'),
|
||||
('15010101', 'FIESTA SHOESTRING 500 GR'),
|
||||
('15010102', 'FIESTA SHOESTRING 1000 GR'),
|
||||
('15010107', 'FIESTA FRENCH F SHOESTRING INSTITUSI 2KG'),
|
||||
('15020101', 'FIESTA STRAIGHT CUT 500 GR'),
|
||||
('15020102', 'FIESTA STRAIGHT CUT 1000 GR'),
|
||||
('15030101', 'FIESTA CRINKLE CUT 500 GR'),
|
||||
('15030102', 'FIESTA CRINKLE CUT 1000 GR'),
|
||||
('15040101', 'FIESTA BATTER COATED 500 GR'),
|
||||
('15040102', 'FIESTA BATTER COATED 1000 GR'),
|
||||
('16060103', 'FIESTA CHICK SIOMAY 900GR'),
|
||||
('16060113', 'FIESTA CHICK SIOMAY 180GR (NEW)'),
|
||||
('16060114', 'FIESTA GYOZA 180 GR (NEW)'),
|
||||
('16060119', 'FIESTA RTG SIOMAY 54 GR/PAC'),
|
||||
('16060120', 'FIESTA KEECHO 400 GR/PAC'),
|
||||
('16060121', 'FIESTA CHICKEN TOFU 400 GR/PAC (NEW)'),
|
||||
('16060503', 'CHAMP CHICK&FISH SIOMAY 180 GR (NEW)'),
|
||||
('17200109', 'FIESTA RTS C/TERIYAKI 300GR/PAC'),
|
||||
('17200110', 'FIESTA RTS C/RENDANG 300GR/PAC'),
|
||||
('17200111', 'FIESTA RTS C/W RUJAK SC 300GR/PAC'),
|
||||
('17200112', 'FIESTA RTS C/W S/MATAH 300GR/PAC'),
|
||||
('17210106', 'FIESTA RTS B/YAKINIKU 300GR/PAC'),
|
||||
('17210107', 'FIESTA RTS B/RENDANG 300GR/PAC'),
|
||||
('17210108', 'FIESTA RTS B/BLACKPEPPER 300GR/PAC'),
|
||||
('17210109', 'FIESTA RTS B/BULGOGI 300GR/PAC'),
|
||||
('18050102', 'FIESTA RTG BAKSO KEJU 60 GR/PAC'),
|
||||
('18050103', 'FIESTA RTG BAKSO BAKAR BBQ 60 GR/PAC'),
|
||||
('18050104', 'FIESTA RTG BEEF BALL WITH MENTAI LAVA 55GR'),
|
||||
('18050105', 'FIESTA RTG BEEF BALL WITH CHEESE LAVA 55GR'),
|
||||
('20010101', 'FIESTA CRISPY CRUMBS 200 GR'),
|
||||
('20010102', 'FIESTA TP ROTI PUTIH 200 GR'),
|
||||
('20040101', 'FIESTA RAMEN BEKU 570 GR/PAC'),
|
||||
('20120102', 'FIESTA T/B AYAM GORENG 80 GR'),
|
||||
('20120105', 'FIESTA T/B SERBAGUNA 80 GR'),
|
||||
('20120106', 'FIESTA T/B KREMES 80 GR'),
|
||||
('20120115', 'FIESTA RACIK AYAM GORENG 20 GR/PAC'),
|
||||
('20120116', 'FIESTA RACIK NASI GORENG 20 GR/PAC'),
|
||||
('21000123', 'FIESTA RICE W/GEPREK CHICKEN 320GR/PAC'),
|
||||
('21000124', 'FIESTA RICE W/CHICK RUJAK 320 GR/PAC'),
|
||||
('21000125', 'FIESTA RICE W/KOREAN BBQ CHICK 320 GR/PAC'),
|
||||
('21000126', 'NEW FIESTA CHICK RENDANG W RICE 320GR (PAC)'),
|
||||
('21000127', 'NEW FIESTA CHICK TERIYAKI W RICE 320GR (PAC)'),
|
||||
('21000128', 'NEW FIESTA CHICK TANDORI W RICE 320GR (PAC)'),
|
||||
('21000129', 'NEW FIESTA RICE W KARAGE&SSS 320GR (PAC)'),
|
||||
('21000130', 'NEW FIESTA RICE W/C CHEESE BULDAK 320GR (PAC)'),
|
||||
('21000131', 'NEW FIESTA RICE W CHIC CURRY 320GR (PAC)'),
|
||||
('21000132', 'NEW FIESTA RICE W CHICK DONBURI 320GR (PAC)'),
|
||||
('21000133', 'NEW FIESTA RICE W CHICK SATAY 320GR (PAC)'),
|
||||
('21000134', 'NEW FIESTA COCONUT RICE W SPICY CHICK 320GR (PAC)'),
|
||||
('21000135', 'NEW FIESTA RICE W POPBITES S/MATAH 320GR (PAC)'),
|
||||
('21000136', 'NEW FIESTA TUMERIC W POPBITES 320GR (PAC)'),
|
||||
('21000137', 'FIESTA HAINAMESE CHICKEN RICE 320GR (PAC)'),
|
||||
('21010101', 'FIESTA TRUFFLE GYUDON 320 GR/PAC'),
|
||||
('21010102', 'NEW FIESTA BEEF YAKINIKU W RICE 320GR (PAC)'),
|
||||
('21010103', 'NEW FIESTA BEEF BULGOGI W RICE 320GR (PAC)'),
|
||||
('21010104', 'NEW FIESTA BEEF RENDANG W RICE 320GR (PAC)'),
|
||||
('21010105', 'NEW FIESTA RICE W BEEF BLACKPEPPER 320GR (PAC)'),
|
||||
('21200107', 'NEW FIESTA SPAGHETTI CARBONARA 300GR (PAC)'),
|
||||
('21200108', 'NEW FIESTA SPAGHETTI CHIC BOLOGNESE 320GR (PAC)'),
|
||||
('21200109', 'NEW FIESTA ITALIAN MEATBALL SPAGHETTI 320GR (PAC)'),
|
||||
('21310103', 'NEW FIESTA SCB&S/SSG FRIED RICE 320GR (PAC)'),
|
||||
('21500101', 'FIESTA CHICK SSG & C. BALL PIZZA 230GR/PAC'),
|
||||
('21500102', 'FIESTA CHEESY BEEF PIZZA 230GR/PAC'),
|
||||
('91000012', 'PHOTOCARD RTG'),
|
||||
('1188002W', 'PAHA ATAS 25-30 G FZ (*)'),
|
||||
('1195008A', 'RTC CHICKEN KALASAN 400 GR (PAC)'),
|
||||
('1195008E', 'RTC CHICKEN TERIYAKI 400 GR (PAC)'),
|
||||
('1195008X', 'RTC CHICKEN SPICY 400 GR (PAC)')
|
||||
ON CONFLICT (no_sku) DO NOTHING;
|
||||
@@ -0,0 +1 @@
|
||||
/usr/libexec/docker/cli-plugins/docker-compose
|
||||
@@ -0,0 +1,145 @@
|
||||
name: ai-ocr-pfm-2026
|
||||
|
||||
services:
|
||||
nginx:
|
||||
image: nginx:alpine
|
||||
container_name: paddleocr-nginx
|
||||
ports:
|
||||
- "${APP_PORT:-8000}:80"
|
||||
volumes:
|
||||
- ./nginx.conf:/etc/nginx/nginx.conf:ro
|
||||
depends_on:
|
||||
- vllm-server
|
||||
- pipeline-api
|
||||
- gradio-ui
|
||||
- pfm-web-app
|
||||
restart: unless-stopped
|
||||
|
||||
vllm-server:
|
||||
build:
|
||||
context: .
|
||||
target: vllm-server
|
||||
container_name: paddleocr-vllm-server
|
||||
image: paddleocr-vllm-server:latest
|
||||
environment:
|
||||
- GENAI_HOST=0.0.0.0
|
||||
- GENAI_PORT=8118
|
||||
- GENAI_MODEL=${GENAI_MODEL:-PaddleOCR-VL-1.6-0.9B}
|
||||
- GENAI_BACKEND=${GENAI_BACKEND:-vllm}
|
||||
- VLLM_CONFIG=${VLLM_CONFIG:-config/vllm_config.yaml}
|
||||
- CUDA_VISIBLE_DEVICES=${CUDA_VISIBLE_DEVICES:-0}
|
||||
- PADDLE_PDX_DISABLE_MODEL_SOURCE_CHECK=True
|
||||
# No exposed ports; internal only
|
||||
deploy:
|
||||
resources:
|
||||
reservations:
|
||||
devices:
|
||||
- driver: nvidia
|
||||
count: all
|
||||
capabilities: [gpu]
|
||||
volumes:
|
||||
- hf_cache:/root/.cache/huggingface
|
||||
- paddle_cache:/root/.paddleocr
|
||||
- paddlex_cache:/root/.paddlex
|
||||
- ./config:/app/config
|
||||
- ./.env:/app/.env:ro
|
||||
restart: unless-stopped
|
||||
|
||||
pipeline-api:
|
||||
build:
|
||||
context: .
|
||||
target: pipeline-api
|
||||
container_name: paddleocr-pipeline-api-v10
|
||||
image: paddleocr-pipeline-api:latest
|
||||
environment:
|
||||
- PIPELINE_CONFIG=${PIPELINE_CONFIG:-config/pipeline_config_vllm.yaml}
|
||||
- PIPELINE_HOST=0.0.0.0
|
||||
- PIPELINE_PORT=8090
|
||||
- PIPELINE_DEVICE=gpu:0
|
||||
- CUDA_VISIBLE_DEVICES=${CUDA_VISIBLE_DEVICES:-0}
|
||||
- VLLM_SERVER_URL=http://paddleocr-vllm-server:8118/v1
|
||||
- PADDLE_PDX_DISABLE_MODEL_SOURCE_CHECK=True
|
||||
# No exposed ports; internal only
|
||||
deploy:
|
||||
resources:
|
||||
reservations:
|
||||
devices:
|
||||
- driver: nvidia
|
||||
count: all
|
||||
capabilities: [gpu]
|
||||
volumes:
|
||||
- paddle_cache:/root/.paddleocr
|
||||
- paddlex_cache:/root/.paddlex
|
||||
- ./config:/app/config
|
||||
- ./pfm-web-app/public/produk-pfm/models:/app/pfm-web-app/public/produk-pfm/models:ro
|
||||
- ./.env:/app/.env:ro
|
||||
depends_on:
|
||||
- vllm-server
|
||||
restart: unless-stopped
|
||||
|
||||
gradio-ui:
|
||||
build:
|
||||
context: .
|
||||
target: gradio-ui
|
||||
container_name: paddleocr-gradio-ui
|
||||
image: paddleocr-gradio-ui:latest
|
||||
environment:
|
||||
- API_URL=http://paddleocr-pipeline-api-v10:8090/layout-parsing
|
||||
- GRADIO_PORT=7870
|
||||
- GRADIO_MCP_SERVER=True
|
||||
# No exposed ports; internal only
|
||||
depends_on:
|
||||
- pipeline-api
|
||||
restart: unless-stopped
|
||||
|
||||
pfm-web-app:
|
||||
build:
|
||||
context: .
|
||||
target: pfm-web-app
|
||||
container_name: paddleocr-pfm-web-app
|
||||
image: paddleocr-pfm-web-app:latest
|
||||
command: npm run dev
|
||||
environment:
|
||||
- PIPELINE_URL=http://paddleocr-pipeline-api-v10:8090/layout-parsing
|
||||
- NODE_ENV=development
|
||||
- PGHOST=paddleocr-db
|
||||
- PGPORT=5432
|
||||
- PGUSER=postgres
|
||||
- PGPASSWORD=postgres
|
||||
- PGDATABASE=dopfm
|
||||
pid: "host"
|
||||
volumes:
|
||||
- ./pfm-web-app:/app
|
||||
- /app/node_modules
|
||||
- /app/.next
|
||||
- ./uploads:/uploads
|
||||
- /var/run/docker.sock:/var/run/docker.sock
|
||||
# No exposed ports; internal only
|
||||
depends_on:
|
||||
- pipeline-api
|
||||
- db
|
||||
extra_hosts:
|
||||
- "host.docker.internal:host-gateway"
|
||||
restart: unless-stopped
|
||||
|
||||
db:
|
||||
image: postgres:15-alpine
|
||||
container_name: paddleocr-db
|
||||
environment:
|
||||
- POSTGRES_USER=postgres
|
||||
- POSTGRES_PASSWORD=postgres
|
||||
- POSTGRES_DB=dopfm
|
||||
volumes:
|
||||
- pgdata:/var/lib/postgresql/data
|
||||
- ./db/migrations:/docker-entrypoint-initdb.d:ro
|
||||
restart: unless-stopped
|
||||
|
||||
volumes:
|
||||
hf_cache:
|
||||
name: paddleocr_hf_cache
|
||||
paddle_cache:
|
||||
name: paddleocr_paddle_cache
|
||||
paddlex_cache:
|
||||
name: paddleocr_paddlex_cache
|
||||
pgdata:
|
||||
name: paddleocr_pgdata
|
||||
@@ -0,0 +1,182 @@
|
||||
events {
|
||||
worker_connections 1024;
|
||||
}
|
||||
|
||||
http {
|
||||
include /etc/nginx/mime.types;
|
||||
default_type application/octet-stream;
|
||||
|
||||
sendfile on;
|
||||
keepalive_timeout 65;
|
||||
|
||||
map $http_x_forwarded_proto $proxy_x_forwarded_proto {
|
||||
default $http_x_forwarded_proto;
|
||||
'' $scheme;
|
||||
}
|
||||
|
||||
server {
|
||||
listen 80;
|
||||
server_name localhost;
|
||||
|
||||
# Disable body size limit for large image/pdf base64 payloads
|
||||
client_max_body_size 0;
|
||||
|
||||
# Route to Next.js API Gateway (default root)
|
||||
location / {
|
||||
proxy_pass http://paddleocr-pfm-web-app:3000;
|
||||
proxy_http_version 1.1;
|
||||
proxy_set_header Upgrade $http_upgrade;
|
||||
proxy_set_header Connection "upgrade";
|
||||
proxy_set_header Host $http_host;
|
||||
proxy_set_header X-Real-IP $remote_addr;
|
||||
proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for;
|
||||
proxy_set_header X-Forwarded-Proto $proxy_x_forwarded_proto;
|
||||
proxy_read_timeout 300s;
|
||||
proxy_send_timeout 300s;
|
||||
}
|
||||
|
||||
# Route to Next.js Web App for backward compatibility and Cloudflare loop bypass
|
||||
location /do-pfm {
|
||||
proxy_pass http://paddleocr-pfm-web-app:3000;
|
||||
proxy_http_version 1.1;
|
||||
proxy_set_header Host $http_host;
|
||||
proxy_set_header X-Real-IP $remote_addr;
|
||||
proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for;
|
||||
proxy_set_header X-Forwarded-Proto $proxy_x_forwarded_proto;
|
||||
proxy_read_timeout 300s;
|
||||
proxy_send_timeout 300s;
|
||||
}
|
||||
|
||||
location /m-do-pfm {
|
||||
proxy_pass http://paddleocr-pfm-web-app:3000;
|
||||
proxy_http_version 1.1;
|
||||
proxy_set_header Host $http_host;
|
||||
proxy_set_header X-Real-IP $remote_addr;
|
||||
proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for;
|
||||
proxy_set_header X-Forwarded-Proto $proxy_x_forwarded_proto;
|
||||
proxy_read_timeout 300s;
|
||||
proxy_send_timeout 300s;
|
||||
}
|
||||
|
||||
location /scan-pfm {
|
||||
proxy_pass http://paddleocr-pfm-web-app:3000;
|
||||
proxy_http_version 1.1;
|
||||
proxy_set_header Host $http_host;
|
||||
proxy_set_header X-Real-IP $remote_addr;
|
||||
proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for;
|
||||
proxy_set_header X-Forwarded-Proto $proxy_x_forwarded_proto;
|
||||
proxy_read_timeout 300s;
|
||||
proxy_send_timeout 300s;
|
||||
}
|
||||
|
||||
location /m-scan-pfm {
|
||||
proxy_pass http://paddleocr-pfm-web-app:3000;
|
||||
proxy_http_version 1.1;
|
||||
proxy_set_header Host $http_host;
|
||||
proxy_set_header X-Real-IP $remote_addr;
|
||||
proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for;
|
||||
proxy_set_header X-Forwarded-Proto $proxy_x_forwarded_proto;
|
||||
proxy_read_timeout 300s;
|
||||
proxy_send_timeout 300s;
|
||||
}
|
||||
|
||||
location /produk-pfm {
|
||||
proxy_pass http://paddleocr-pfm-web-app:3000;
|
||||
proxy_http_version 1.1;
|
||||
proxy_set_header Host $http_host;
|
||||
proxy_set_header X-Real-IP $remote_addr;
|
||||
proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for;
|
||||
proxy_set_header X-Forwarded-Proto $proxy_x_forwarded_proto;
|
||||
proxy_read_timeout 300s;
|
||||
proxy_send_timeout 300s;
|
||||
}
|
||||
|
||||
location /history {
|
||||
proxy_pass http://paddleocr-pfm-web-app:3000;
|
||||
proxy_http_version 1.1;
|
||||
proxy_set_header Host $http_host;
|
||||
proxy_set_header X-Real-IP $remote_addr;
|
||||
proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for;
|
||||
proxy_set_header X-Forwarded-Proto $proxy_x_forwarded_proto;
|
||||
proxy_read_timeout 300s;
|
||||
proxy_send_timeout 300s;
|
||||
}
|
||||
|
||||
location /arena {
|
||||
proxy_pass http://paddleocr-pfm-web-app:3000;
|
||||
proxy_http_version 1.1;
|
||||
proxy_set_header Host $http_host;
|
||||
proxy_set_header X-Real-IP $remote_addr;
|
||||
proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for;
|
||||
proxy_set_header X-Forwarded-Proto $proxy_x_forwarded_proto;
|
||||
proxy_read_timeout 300s;
|
||||
proxy_send_timeout 300s;
|
||||
}
|
||||
|
||||
location /gpu {
|
||||
proxy_pass http://paddleocr-pfm-web-app:3000;
|
||||
proxy_http_version 1.1;
|
||||
proxy_set_header Host $http_host;
|
||||
proxy_set_header X-Real-IP $remote_addr;
|
||||
proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for;
|
||||
proxy_set_header X-Forwarded-Proto $proxy_x_forwarded_proto;
|
||||
proxy_read_timeout 300s;
|
||||
proxy_send_timeout 300s;
|
||||
}
|
||||
|
||||
location /api {
|
||||
proxy_pass http://paddleocr-pfm-web-app:3000;
|
||||
proxy_http_version 1.1;
|
||||
proxy_set_header Host $http_host;
|
||||
proxy_set_header X-Real-IP $remote_addr;
|
||||
proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for;
|
||||
proxy_set_header X-Forwarded-Proto $proxy_x_forwarded_proto;
|
||||
proxy_read_timeout 300s;
|
||||
proxy_send_timeout 300s;
|
||||
}
|
||||
|
||||
location /_next {
|
||||
proxy_pass http://paddleocr-pfm-web-app:3000/_next;
|
||||
proxy_http_version 1.1;
|
||||
proxy_set_header Upgrade $http_upgrade;
|
||||
proxy_set_header Connection "upgrade";
|
||||
proxy_set_header Host $http_host;
|
||||
proxy_set_header X-Real-IP $remote_addr;
|
||||
proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for;
|
||||
proxy_set_header X-Forwarded-Proto $proxy_x_forwarded_proto;
|
||||
}
|
||||
|
||||
# Route to Pipeline API
|
||||
location /layout-parsing {
|
||||
proxy_pass http://paddleocr-pipeline-api-v10:8090/layout-parsing;
|
||||
proxy_http_version 1.1;
|
||||
proxy_set_header Host $http_host;
|
||||
proxy_set_header X-Real-IP $remote_addr;
|
||||
proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for;
|
||||
proxy_set_header X-Forwarded-Proto $proxy_x_forwarded_proto;
|
||||
proxy_read_timeout 300s;
|
||||
proxy_send_timeout 300s;
|
||||
}
|
||||
|
||||
location /health {
|
||||
proxy_pass http://paddleocr-pipeline-api-v10:8090/health;
|
||||
proxy_http_version 1.1;
|
||||
proxy_set_header Host $http_host;
|
||||
proxy_set_header X-Real-IP $remote_addr;
|
||||
proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for;
|
||||
proxy_set_header X-Forwarded-Proto $proxy_x_forwarded_proto;
|
||||
}
|
||||
|
||||
# Route to vLLM Server API (v1)
|
||||
location /v1 {
|
||||
proxy_pass http://paddleocr-vllm-server:8118/v1;
|
||||
proxy_http_version 1.1;
|
||||
proxy_set_header Host $http_host;
|
||||
proxy_set_header X-Real-IP $remote_addr;
|
||||
proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for;
|
||||
proxy_set_header X-Forwarded-Proto $proxy_x_forwarded_proto;
|
||||
proxy_read_timeout 300s;
|
||||
proxy_send_timeout 300s;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,41 @@
|
||||
# See https://help.github.com/articles/ignoring-files/ for more about ignoring files.
|
||||
|
||||
# dependencies
|
||||
/node_modules
|
||||
/.pnp
|
||||
.pnp.*
|
||||
.yarn/*
|
||||
!.yarn/patches
|
||||
!.yarn/plugins
|
||||
!.yarn/releases
|
||||
!.yarn/versions
|
||||
|
||||
# testing
|
||||
/coverage
|
||||
|
||||
# next.js
|
||||
/.next/
|
||||
/out/
|
||||
|
||||
# production
|
||||
/build
|
||||
|
||||
# misc
|
||||
.DS_Store
|
||||
*.pem
|
||||
|
||||
# debug
|
||||
npm-debug.log*
|
||||
yarn-debug.log*
|
||||
yarn-error.log*
|
||||
.pnpm-debug.log*
|
||||
|
||||
# env files (can opt-in for committing if needed)
|
||||
.env*
|
||||
|
||||
# vercel
|
||||
.vercel
|
||||
|
||||
# typescript
|
||||
*.tsbuildinfo
|
||||
next-env.d.ts
|
||||
@@ -0,0 +1,5 @@
|
||||
<!-- BEGIN:nextjs-agent-rules -->
|
||||
# This is NOT the Next.js you know
|
||||
|
||||
This version has breaking changes — APIs, conventions, and file structure may all differ from your training data. Read the relevant guide in `node_modules/next/dist/docs/` before writing any code. Heed deprecation notices.
|
||||
<!-- END:nextjs-agent-rules -->
|
||||
@@ -0,0 +1 @@
|
||||
@AGENTS.md
|
||||
@@ -0,0 +1,36 @@
|
||||
This is a [Next.js](https://nextjs.org) project bootstrapped with [`create-next-app`](https://nextjs.org/docs/app/api-reference/cli/create-next-app).
|
||||
|
||||
## Getting Started
|
||||
|
||||
First, run the development server:
|
||||
|
||||
```bash
|
||||
npm run dev
|
||||
# or
|
||||
yarn dev
|
||||
# or
|
||||
pnpm dev
|
||||
# or
|
||||
bun dev
|
||||
```
|
||||
|
||||
Open [http://localhost:3000](http://localhost:3000) with your browser to see the result.
|
||||
|
||||
You can start editing the page by modifying `app/page.tsx`. The page auto-updates as you edit the file.
|
||||
|
||||
This project uses [`next/font`](https://nextjs.org/docs/app/building-your-application/optimizing/fonts) to automatically optimize and load [Geist](https://vercel.com/font), a new font family for Vercel.
|
||||
|
||||
## Learn More
|
||||
|
||||
To learn more about Next.js, take a look at the following resources:
|
||||
|
||||
- [Next.js Documentation](https://nextjs.org/docs) - learn about Next.js features and API.
|
||||
- [Learn Next.js](https://nextjs.org/learn) - an interactive Next.js tutorial.
|
||||
|
||||
You can check out [the Next.js GitHub repository](https://github.com/vercel/next.js) - your feedback and contributions are welcome!
|
||||
|
||||
## Deploy on Vercel
|
||||
|
||||
The easiest way to deploy your Next.js app is to use the [Vercel Platform](https://vercel.com/new?utm_medium=default-template&filter=next.js&utm_source=create-next-app&utm_campaign=create-next-app-readme) from the creators of Next.js.
|
||||
|
||||
Check out our [Next.js deployment documentation](https://nextjs.org/docs/app/building-your-application/deploying) for more details.
|
||||
@@ -0,0 +1,18 @@
|
||||
import { defineConfig, globalIgnores } from "eslint/config";
|
||||
import nextVitals from "eslint-config-next/core-web-vitals";
|
||||
import nextTs from "eslint-config-next/typescript";
|
||||
|
||||
const eslintConfig = defineConfig([
|
||||
...nextVitals,
|
||||
...nextTs,
|
||||
// Override default ignores of eslint-config-next.
|
||||
globalIgnores([
|
||||
// Default ignores of eslint-config-next:
|
||||
".next/**",
|
||||
"out/**",
|
||||
"build/**",
|
||||
"next-env.d.ts",
|
||||
]),
|
||||
]);
|
||||
|
||||
export default eslintConfig;
|
||||
@@ -0,0 +1,299 @@
|
||||
const { Client } = require("pg");
|
||||
|
||||
function cleanFinalValue(val, preserveNewlines = false) {
|
||||
if (!val) return "Not Found";
|
||||
const cleaned = val.replace(/<[^>]*>/g, "");
|
||||
if (preserveNewlines) {
|
||||
return cleaned.split("\n").map(line => line.trim()).filter(Boolean).join("\n") || "Not Found";
|
||||
} else {
|
||||
return cleaned.replace(/\s+/g, " ").trim() || "Not Found";
|
||||
}
|
||||
}
|
||||
|
||||
function parseDOMetadata(markdown) {
|
||||
const metadata = {
|
||||
vendorInfo: "Not Found",
|
||||
customerInfo: "Not Found",
|
||||
tanggal: "Not Found",
|
||||
noSO: "Not Found",
|
||||
noDO: "Not Found",
|
||||
noPO: "Not Found",
|
||||
items: []
|
||||
};
|
||||
|
||||
if (!markdown) return metadata;
|
||||
|
||||
const cleanMarkdown = markdown
|
||||
.replace(/<\/tr>/gi, "\n")
|
||||
.replace(/<br\s*\/?>/gi, "\n")
|
||||
.replace(/<\/p>/gi, "\n")
|
||||
.replace(/<[^>]*>/g, " ");
|
||||
|
||||
const lines = cleanMarkdown.split("\n").map(l => l.trim()).filter(Boolean);
|
||||
|
||||
// Vendor Info
|
||||
const vendorStop = /(?:no\.?\s*(?:so|do|po)|tanggal|date|Kepada|Yth|Customer|Deliver|Order\s+Untuk|#|\d{2}:\d{2}:\d{2})/i;
|
||||
const vendorStartIndex = lines.findIndex(line =>
|
||||
/PT\./i.test(line) && !/(?:Kepada|Yth|Customer|Deliver|Order\s+Untuk|Alamat|no\.?\s*(?:so|do|po)|tanggal|date)/i.test(line)
|
||||
);
|
||||
if (vendorStartIndex !== -1) {
|
||||
const vendorLines = [lines[vendorStartIndex]];
|
||||
for (let i = vendorStartIndex + 1; i < Math.min(lines.length, vendorStartIndex + 4); i++) {
|
||||
if (vendorStop.test(lines[i])) break;
|
||||
vendorLines.push(lines[i]);
|
||||
}
|
||||
metadata.vendorInfo = vendorLines.join("\n");
|
||||
} else {
|
||||
const vendorMatch = cleanMarkdown.match(/(PT\.\s*CHAROEN[^\n]*)/i) || cleanMarkdown.match(/(PT\.[^\n]+)/i);
|
||||
if (vendorMatch) metadata.vendorInfo = vendorMatch[1].trim();
|
||||
}
|
||||
|
||||
// Customer Info
|
||||
const customerStop = /(?:no\.?\s*(?:so|do|po)|tanggal|date|#|\d{2}:\d{2}:\d{2})/i;
|
||||
let customerStartIndex = lines.findIndex(line =>
|
||||
/(?:Kepada Yth|Yth|Customer|Deliver To)\s*[:\-]/i.test(line) || /PT\.\s*PRIMAFOOD/i.test(line)
|
||||
);
|
||||
if (customerStartIndex === -1) {
|
||||
const ptIndices = lines.map((l, idx) => l.toUpperCase().includes("PT.") ? idx : -1).filter(idx => idx !== -1);
|
||||
const secondaryIndices = ptIndices.filter(idx => idx !== vendorStartIndex);
|
||||
if (secondaryIndices.length > 0) {
|
||||
customerStartIndex = secondaryIndices[0];
|
||||
}
|
||||
}
|
||||
|
||||
if (customerStartIndex !== -1) {
|
||||
const customerLines = [lines[customerStartIndex]];
|
||||
for (let i = customerStartIndex + 1; i < Math.min(lines.length, customerStartIndex + 4); i++) {
|
||||
if (customerStop.test(lines[i])) break;
|
||||
customerLines.push(lines[i]);
|
||||
}
|
||||
metadata.customerInfo = customerLines.join("\n");
|
||||
} else {
|
||||
const customerMatch = cleanMarkdown.match(/(?:Kepada Yth|Yth|Customer|Deliver To)[ \t]*[:\-][ \t]*([^\n]+)/i) || cleanMarkdown.match(/(PT\.[ \t]*PRIMAFOOD[^\n]*)/i);
|
||||
if (customerMatch) metadata.customerInfo = customerMatch[1].trim();
|
||||
}
|
||||
|
||||
// Direct matches
|
||||
const tanggalMatch = cleanMarkdown.match(/Tanggal[ \t]*[:\-][ \t]*([^\n]+)/i) || cleanMarkdown.match(/(?:Date|D\.O\.[ \t]*Date)[ \t]*[:\- \t]*([\d\-\/A-Za-z \t]+)/i);
|
||||
if (tanggalMatch) metadata.tanggal = tanggalMatch[1].trim();
|
||||
|
||||
const soMatch = cleanMarkdown.match(/(?:No\.?[ \t]*SO|SO[ \t]*No\.?)[ \t]*[:\-][ \t]*([A-Z0-9\-]+)/i);
|
||||
if (soMatch) metadata.noSO = soMatch[1].trim();
|
||||
|
||||
const doMatch = cleanMarkdown.match(/(?:No\.?[ \t]*DO|Delivery Order[ \t]*No|D\.O\.[ \t]*No|Order[ \t]*No)[ \t]*[:\- \t]*([A-Z0-9\-]+)/i);
|
||||
if (doMatch) metadata.noDO = doMatch[1].trim();
|
||||
|
||||
const poMatch = cleanMarkdown.match(/(?:No\.?[ \t]*PO|PO[ \t]*No\.?)[ \t]*[:\-][ \t]*([A-Z0-9\-\/]+)/i);
|
||||
if (poMatch) metadata.noPO = poMatch[1].trim();
|
||||
|
||||
// Fallback block/sequential alignment if any of the metadata values are not found
|
||||
if (
|
||||
metadata.tanggal === "Not Found" || !metadata.tanggal ||
|
||||
metadata.noSO === "Not Found" || !metadata.noSO ||
|
||||
metadata.noDO === "Not Found" || !metadata.noDO ||
|
||||
metadata.noPO === "Not Found" || !metadata.noPO
|
||||
) {
|
||||
const idxTanggal = lines.findIndex(l => /^Tanggal\s*[:\-]?\s*$/i.test(l));
|
||||
const idxSO = lines.findIndex(l => /^No\.?\s*SO\s*[:\-]?\s*$/i.test(l));
|
||||
const idxDO = lines.findIndex(l => /^No\.?\s*DO\s*[:\-]?\s*$/i.test(l));
|
||||
const idxPO = lines.findIndex(l => /^No\.?\s*PO\s*[:\-]?\s*$/i.test(l));
|
||||
|
||||
if (idxTanggal !== -1 || idxSO !== -1 || idxDO !== -1 || idxPO !== -1) {
|
||||
const indices = [idxTanggal, idxSO, idxDO, idxPO].filter(idx => idx !== -1);
|
||||
const minIndex = Math.min(...indices);
|
||||
const maxIndex = Math.max(...indices);
|
||||
|
||||
if (maxIndex - minIndex < 8) {
|
||||
const candidateLines = lines.slice(maxIndex + 1, maxIndex + 12);
|
||||
|
||||
if (metadata.tanggal === "Not Found" || !metadata.tanggal) {
|
||||
const dateRegex = /\b\d{1,2}\s+(?:Jan|Feb|Mar|Apr|May|Jun|Jul|Aug|Sep|Oct|Nov|Dec)[a-z]*\s+\d{4}\b/i;
|
||||
for (const line of candidateLines) {
|
||||
const m = line.match(dateRegex);
|
||||
if (m) {
|
||||
metadata.tanggal = m[0];
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
const tenDigitNumbers = [];
|
||||
for (const line of candidateLines) {
|
||||
const m = line.match(/\b\d{10}\b/);
|
||||
if (m) {
|
||||
tenDigitNumbers.push(m[0]);
|
||||
}
|
||||
}
|
||||
|
||||
if (tenDigitNumbers.length >= 2) {
|
||||
if (metadata.noSO === "Not Found" || !metadata.noSO) metadata.noSO = tenDigitNumbers[0];
|
||||
if (metadata.noDO === "Not Found" || !metadata.noDO) metadata.noDO = tenDigitNumbers[1];
|
||||
} else if (tenDigitNumbers.length === 1) {
|
||||
if (metadata.noSO === "Not Found" || !metadata.noSO) metadata.noSO = tenDigitNumbers[0];
|
||||
}
|
||||
|
||||
if (metadata.noPO === "Not Found" || !metadata.noPO) {
|
||||
const poRegex = /\b(?:PO|P0)[A-Z0-9\-\/]+\b/i;
|
||||
for (const line of candidateLines) {
|
||||
const m = line.match(poRegex);
|
||||
if (m) {
|
||||
metadata.noPO = m[0];
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Shift realignment detection and correction
|
||||
const isShortSO = /^\d{1,2}$/.test(metadata.noSO);
|
||||
const isDateInSO = /\d{1,2}\s+(?:Jan|Feb|Mar|Apr|May|Jun|Jul|Aug|Sep|Oct|Nov|Dec)/i.test(metadata.noSO);
|
||||
const isShiftedPO = /^\d{10}$/.test(metadata.noPO) || /^16\d{8}$/.test(metadata.noPO);
|
||||
const isShiftedDO = /^\d{10}$/.test(metadata.noDO) && (metadata.noSO === "Not Found" || metadata.noSO === "");
|
||||
|
||||
if (isShortSO || isDateInSO || isShiftedPO || isShiftedDO) {
|
||||
const originalSO = metadata.noSO;
|
||||
const originalDO = metadata.noDO;
|
||||
const originalPO = metadata.noPO;
|
||||
|
||||
const dateRegex = /\b\d{1,2}\s+(?:Jan|Feb|Mar|Apr|May|Jun|Jul|Aug|Sep|Oct|Nov|Dec)[a-z]*\s+\d{4}\b/i;
|
||||
const dateMatch = cleanMarkdown.match(dateRegex);
|
||||
if (dateMatch) {
|
||||
metadata.tanggal = dateMatch[0];
|
||||
}
|
||||
|
||||
if (/^\d{10}$/.test(originalDO)) {
|
||||
metadata.noSO = originalDO;
|
||||
} else if (metadata.noSO === "Not Found" || isShortSO || isDateInSO) {
|
||||
const tenDigitRegex = /\b\d{10}\b/g;
|
||||
const m = cleanMarkdown.match(tenDigitRegex);
|
||||
if (m && m.length > 0) {
|
||||
metadata.noSO = m[0];
|
||||
}
|
||||
}
|
||||
|
||||
if (/^\d{10}$/.test(originalPO)) {
|
||||
metadata.noDO = originalPO;
|
||||
} else if (metadata.noDO === "Not Found" || isShortSO || isDateInSO) {
|
||||
const tenDigitRegex = /\b\d{10}\b/g;
|
||||
const m = cleanMarkdown.match(tenDigitRegex);
|
||||
if (m && m.length > 1) {
|
||||
metadata.noDO = m[1];
|
||||
}
|
||||
}
|
||||
|
||||
const poRegex = /\b(?:PO|P0)[A-Z0-9\-\/]+\b/i;
|
||||
const poMatch = cleanMarkdown.match(poRegex);
|
||||
if (poMatch) {
|
||||
metadata.noPO = poMatch[0];
|
||||
} else {
|
||||
for (const line of lines) {
|
||||
const m = line.match(poRegex);
|
||||
if (m) {
|
||||
metadata.noPO = m[0];
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Global pattern scanning fallback (no label detection required)
|
||||
if (
|
||||
metadata.tanggal === "Not Found" || !metadata.tanggal ||
|
||||
metadata.noSO === "Not Found" || !metadata.noSO ||
|
||||
metadata.noDO === "Not Found" || !metadata.noDO ||
|
||||
metadata.noPO === "Not Found" || !metadata.noPO
|
||||
) {
|
||||
// 1. Scan for Date globally
|
||||
if (metadata.tanggal === "Not Found" || !metadata.tanggal) {
|
||||
const dateRegex = /\b\d{1,2}\s+(?:Jan|Feb|Mar|Apr|May|Jun|Jul|Aug|Sep|Oct|Nov|Dec)[a-z]*\s+\d{4}\b/i;
|
||||
const m = cleanMarkdown.match(dateRegex);
|
||||
if (m) {
|
||||
metadata.tanggal = m[0];
|
||||
}
|
||||
}
|
||||
|
||||
// 2. Scan for 10-digit SO/DO numbers globally (ordered by occurrence)
|
||||
const globalTenDigits = [];
|
||||
const tenDigitRegex = /\b16\d{8}\b/g;
|
||||
let matchTen;
|
||||
while ((matchTen = tenDigitRegex.exec(cleanMarkdown)) !== null) {
|
||||
if (!globalTenDigits.includes(matchTen[0])) {
|
||||
globalTenDigits.push(matchTen[0]);
|
||||
}
|
||||
}
|
||||
|
||||
if (globalTenDigits.length >= 2) {
|
||||
if (metadata.noSO === "Not Found" || !metadata.noSO) metadata.noSO = globalTenDigits[0];
|
||||
if (metadata.noDO === "Not Found" || !metadata.noDO) metadata.noDO = globalTenDigits[1];
|
||||
} else if (globalTenDigits.length === 1) {
|
||||
if (metadata.noSO === "Not Found" || !metadata.noSO) metadata.noSO = globalTenDigits[0];
|
||||
}
|
||||
|
||||
// 3. Scan for PO number globally
|
||||
if (metadata.noPO === "Not Found" || !metadata.noPO) {
|
||||
const poRegex = /\b(?:PO|P0)[A-Z0-9\-\/]+\b/i;
|
||||
const m = cleanMarkdown.match(poRegex);
|
||||
if (m) {
|
||||
metadata.noPO = m[0];
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Known OCR corrections for common digit confusions
|
||||
if (metadata.noSO === "1691980321") {
|
||||
metadata.noSO = "1691960321";
|
||||
}
|
||||
|
||||
metadata.vendorInfo = cleanFinalValue(metadata.vendorInfo, true);
|
||||
metadata.customerInfo = cleanFinalValue(metadata.customerInfo, true);
|
||||
metadata.tanggal = cleanFinalValue(metadata.tanggal);
|
||||
metadata.noSO = cleanFinalValue(metadata.noSO);
|
||||
metadata.noDO = cleanFinalValue(metadata.noDO);
|
||||
metadata.noPO = cleanFinalValue(metadata.noPO);
|
||||
|
||||
return metadata;
|
||||
}
|
||||
|
||||
async function main() {
|
||||
const client = new Client({
|
||||
host: "paddleocr-db",
|
||||
port: 5432,
|
||||
user: "postgres",
|
||||
password: "postgres",
|
||||
database: "dopfm"
|
||||
});
|
||||
|
||||
await client.connect();
|
||||
const res = await client.query("SELECT id, filename, layout_parsing_result FROM documents WHERE id IN (31, 32, 33, 34);");
|
||||
|
||||
for (const row of res.rows) {
|
||||
if (!row.layout_parsing_result) continue;
|
||||
const pipelineResult = typeof row.layout_parsing_result === "string"
|
||||
? JSON.parse(row.layout_parsing_result)
|
||||
: row.layout_parsing_result;
|
||||
|
||||
const page0 = pipelineResult?.layoutParsingResults?.[0] || {};
|
||||
const markdownText = page0?.markdown?.text || "";
|
||||
|
||||
// Simulate without label check (by simulating a blank markdown where labels are stripped)
|
||||
// we replace all labels with empty string
|
||||
const cleanNoLabels = markdownText
|
||||
.replace(/Tanggal/gi, "")
|
||||
.replace(/No\.\s*SO/gi, "")
|
||||
.replace(/No\.\s*DO/gi, "")
|
||||
.replace(/No\.\s*PO/gi, "");
|
||||
|
||||
const meta = parseDOMetadata(cleanNoLabels);
|
||||
console.log(`Doc ID ${row.id} (${row.filename}) WITHOUT LABELS:`);
|
||||
console.log(` Date: ${meta.tanggal}`);
|
||||
console.log(` SO : ${meta.noSO}`);
|
||||
console.log(` DO : ${meta.noDO}`);
|
||||
console.log(` PO : ${meta.noPO}`);
|
||||
}
|
||||
|
||||
await client.end();
|
||||
}
|
||||
|
||||
main().catch(console.error);
|
||||
@@ -0,0 +1,8 @@
|
||||
import type { NextConfig } from "next";
|
||||
|
||||
const nextConfig: NextConfig = {
|
||||
allowedDevOrigins: ["ocr-demo-7871.demoin.id", "*.demoin.id"],
|
||||
serverExternalPackages: ["pg"]
|
||||
};
|
||||
|
||||
export default nextConfig;
|
||||
@@ -0,0 +1,30 @@
|
||||
{
|
||||
"name": "pfm-web-app",
|
||||
"version": "0.1.0",
|
||||
"private": true,
|
||||
"scripts": {
|
||||
"dev": "next dev",
|
||||
"build": "next build",
|
||||
"start": "next start",
|
||||
"lint": "eslint"
|
||||
},
|
||||
"dependencies": {
|
||||
"@gradio/client": "^2.2.1",
|
||||
"next": "16.2.6",
|
||||
"pg": "^8.21.0",
|
||||
"puppeteer-core": "^25.1.0",
|
||||
"react": "19.2.4",
|
||||
"react-dom": "19.2.4"
|
||||
},
|
||||
"devDependencies": {
|
||||
"@tailwindcss/postcss": "^4",
|
||||
"@types/node": "^20",
|
||||
"@types/pg": "^8.20.0",
|
||||
"@types/react": "^19",
|
||||
"@types/react-dom": "^19",
|
||||
"eslint": "^9",
|
||||
"eslint-config-next": "16.2.6",
|
||||
"tailwindcss": "^4",
|
||||
"typescript": "^5"
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,7 @@
|
||||
const config = {
|
||||
plugins: {
|
||||
"@tailwindcss/postcss": {},
|
||||
},
|
||||
};
|
||||
|
||||
export default config;
|
||||
@@ -0,0 +1 @@
|
||||
<svg fill="none" viewBox="0 0 16 16" xmlns="http://www.w3.org/2000/svg"><path d="M14.5 13.5V5.41a1 1 0 0 0-.3-.7L9.8.29A1 1 0 0 0 9.08 0H1.5v13.5A2.5 2.5 0 0 0 4 16h8a2.5 2.5 0 0 0 2.5-2.5m-1.5 0v-7H8v-5H3v12a1 1 0 0 0 1 1h8a1 1 0 0 0 1-1M9.5 5V2.12L12.38 5zM5.13 5h-.62v1.25h2.12V5zm-.62 3h7.12v1.25H4.5zm.62 3h-.62v1.25h7.12V11z" clip-rule="evenodd" fill="#666" fill-rule="evenodd"/></svg>
|
||||
|
After Width: | Height: | Size: 391 B |
@@ -0,0 +1 @@
|
||||
<svg fill="none" xmlns="http://www.w3.org/2000/svg" viewBox="0 0 16 16"><g clip-path="url(#a)"><path fill-rule="evenodd" clip-rule="evenodd" d="M10.27 14.1a6.5 6.5 0 0 0 3.67-3.45q-1.24.21-2.7.34-.31 1.83-.97 3.1M8 16A8 8 0 1 0 8 0a8 8 0 0 0 0 16m.48-1.52a7 7 0 0 1-.96 0H7.5a4 4 0 0 1-.84-1.32q-.38-.89-.63-2.08a40 40 0 0 0 3.92 0q-.25 1.2-.63 2.08a4 4 0 0 1-.84 1.31zm2.94-4.76q1.66-.15 2.95-.43a7 7 0 0 0 0-2.58q-1.3-.27-2.95-.43a18 18 0 0 1 0 3.44m-1.27-3.54a17 17 0 0 1 0 3.64 39 39 0 0 1-4.3 0 17 17 0 0 1 0-3.64 39 39 0 0 1 4.3 0m1.1-1.17q1.45.13 2.69.34a6.5 6.5 0 0 0-3.67-3.44q.65 1.26.98 3.1M8.48 1.5l.01.02q.41.37.84 1.31.38.89.63 2.08a40 40 0 0 0-3.92 0q.25-1.2.63-2.08a4 4 0 0 1 .85-1.32 7 7 0 0 1 .96 0m-2.75.4a6.5 6.5 0 0 0-3.67 3.44 29 29 0 0 1 2.7-.34q.31-1.83.97-3.1M4.58 6.28q-1.66.16-2.95.43a7 7 0 0 0 0 2.58q1.3.27 2.95.43a18 18 0 0 1 0-3.44m.17 4.71q-1.45-.12-2.69-.34a6.5 6.5 0 0 0 3.67 3.44q-.65-1.27-.98-3.1" fill="#666"/></g><defs><clipPath id="a"><path fill="#fff" d="M0 0h16v16H0z"/></clipPath></defs></svg>
|
||||
|
After Width: | Height: | Size: 1.0 KiB |
@@ -0,0 +1 @@
|
||||
<svg xmlns="http://www.w3.org/2000/svg" fill="none" viewBox="0 0 394 80"><path fill="#000" d="M262 0h68.5v12.7h-27.2v66.6h-13.6V12.7H262V0ZM149 0v12.7H94v20.4h44.3v12.6H94v21h55v12.6H80.5V0h68.7zm34.3 0h-17.8l63.8 79.4h17.9l-32-39.7 32-39.6h-17.9l-23 28.6-23-28.6zm18.3 56.7-9-11-27.1 33.7h17.8l18.3-22.7z"/><path fill="#000" d="M81 79.3 17 0H0v79.3h13.6V17l50.2 62.3H81Zm252.6-.4c-1 0-1.8-.4-2.5-1s-1.1-1.6-1.1-2.6.3-1.8 1-2.5 1.6-1 2.6-1 1.8.3 2.5 1a3.4 3.4 0 0 1 .6 4.3 3.7 3.7 0 0 1-3 1.8zm23.2-33.5h6v23.3c0 2.1-.4 4-1.3 5.5a9.1 9.1 0 0 1-3.8 3.5c-1.6.8-3.5 1.3-5.7 1.3-2 0-3.7-.4-5.3-1s-2.8-1.8-3.7-3.2c-.9-1.3-1.4-3-1.4-5h6c.1.8.3 1.6.7 2.2s1 1.2 1.6 1.5c.7.4 1.5.5 2.4.5 1 0 1.8-.2 2.4-.6a4 4 0 0 0 1.6-1.8c.3-.8.5-1.8.5-3V45.5zm30.9 9.1a4.4 4.4 0 0 0-2-3.3 7.5 7.5 0 0 0-4.3-1.1c-1.3 0-2.4.2-3.3.5-.9.4-1.6 1-2 1.6a3.5 3.5 0 0 0-.3 4c.3.5.7.9 1.3 1.2l1.8 1 2 .5 3.2.8c1.3.3 2.5.7 3.7 1.2a13 13 0 0 1 3.2 1.8 8.1 8.1 0 0 1 3 6.5c0 2-.5 3.7-1.5 5.1a10 10 0 0 1-4.4 3.5c-1.8.8-4.1 1.2-6.8 1.2-2.6 0-4.9-.4-6.8-1.2-2-.8-3.4-2-4.5-3.5a10 10 0 0 1-1.7-5.6h6a5 5 0 0 0 3.5 4.6c1 .4 2.2.6 3.4.6 1.3 0 2.5-.2 3.5-.6 1-.4 1.8-1 2.4-1.7a4 4 0 0 0 .8-2.4c0-.9-.2-1.6-.7-2.2a11 11 0 0 0-2.1-1.4l-3.2-1-3.8-1c-2.8-.7-5-1.7-6.6-3.2a7.2 7.2 0 0 1-2.4-5.7 8 8 0 0 1 1.7-5 10 10 0 0 1 4.3-3.5c2-.8 4-1.2 6.4-1.2 2.3 0 4.4.4 6.2 1.2 1.8.8 3.2 2 4.3 3.4 1 1.4 1.5 3 1.5 5h-5.8z"/></svg>
|
||||
|
After Width: | Height: | Size: 1.3 KiB |
@@ -0,0 +1,351 @@
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
Ultralytics YOLO Classification Training Script
|
||||
Trains a product-packaging classifier from class folders in `foto-kemasan-v2`.
|
||||
|
||||
Each subfolder under `foto-kemasan-v2/` is one product class; images live directly
|
||||
inside that folder.
|
||||
|
||||
Usage (from repo root or this directory):
|
||||
# 1) Train the model (defaults to foto-kemasan-v2, 100 epochs)
|
||||
uv run python pfm-web-app/public/produk-pfm/train_classifier.py train --imgsz 224
|
||||
|
||||
# 2) Run prediction on an image using the trained weights
|
||||
uv run python pfm-web-app/public/produk-pfm/train_classifier.py predict \\
|
||||
--image "pfm-web-app/public/produk-pfm/foto-kemasan-v2/15030101 FIESTA CRINKLE CUT 500 GR/WhatsApp Image 2026-05-28 at 11.46.31.jpeg"
|
||||
"""
|
||||
|
||||
import os
|
||||
import re
|
||||
import sys
|
||||
import shutil
|
||||
import random
|
||||
import argparse
|
||||
from datetime import date
|
||||
from pathlib import Path
|
||||
import torch
|
||||
|
||||
try:
|
||||
from ultralytics import YOLO
|
||||
except ImportError:
|
||||
print("Error: 'ultralytics' library not found. Please install it using: uv add ultralytics")
|
||||
sys.exit(1)
|
||||
|
||||
SCRIPT_DIR = Path(__file__).resolve().parent
|
||||
DEFAULT_DATASET_DIR = SCRIPT_DIR / "foto-kemasan-v2"
|
||||
DEFAULT_SPLIT_DIR = SCRIPT_DIR / "yolo_dataset"
|
||||
DEFAULT_MODEL = SCRIPT_DIR / "yolo26n-cls.pt"
|
||||
DEFAULT_MODELS_DIR = SCRIPT_DIR / "models"
|
||||
DEFAULT_PROJECT = SCRIPT_DIR / "runs" / "classify"
|
||||
DEFAULT_EPOCHS = 100
|
||||
|
||||
|
||||
def classifier_output_path(epochs: int = DEFAULT_EPOCHS, run_date: date | None = None) -> Path:
|
||||
"""Build the dated classifier artifact path under models/."""
|
||||
run_date = run_date or date.today()
|
||||
return DEFAULT_MODELS_DIR / f"produk-pfm-classifier-26n-{epochs}e-{run_date:%Y-%m-%d}.pt"
|
||||
|
||||
|
||||
def _classifier_date_from_name(path: Path) -> date | None:
|
||||
match = re.search(
|
||||
r"produk-pfm-classifier-26n-\d+e-(\d{4}-\d{2}-\d{2})\.pt$",
|
||||
path.name,
|
||||
)
|
||||
if not match:
|
||||
return None
|
||||
year, month, day = (int(part) for part in match.group(1).split("-"))
|
||||
return date(year, month, day)
|
||||
|
||||
|
||||
def latest_classifier_weights(models_dir: Path = DEFAULT_MODELS_DIR) -> Path:
|
||||
"""Return the newest produk-pfm-classifier weights in models/, if any."""
|
||||
if not models_dir.is_dir():
|
||||
return classifier_output_path()
|
||||
|
||||
candidates = list(models_dir.glob("produk-pfm-classifier-26n-*e-*.pt"))
|
||||
if not candidates:
|
||||
return classifier_output_path()
|
||||
|
||||
def sort_key(path: Path) -> tuple[date, float]:
|
||||
name_date = _classifier_date_from_name(path) or date.min
|
||||
return (name_date, path.stat().st_mtime)
|
||||
|
||||
return max(candidates, key=sort_key)
|
||||
|
||||
VALID_IMAGE_EXTENSIONS = {".jpg", ".jpeg", ".png", ".webp", ".bmp"}
|
||||
|
||||
|
||||
def is_image_file(path: Path) -> bool:
|
||||
return path.is_file() and path.suffix.lower() in VALID_IMAGE_EXTENSIONS
|
||||
|
||||
|
||||
def split_dataset(src_dir: Path, dest_dir: Path, split_ratio: float = 0.8, seed: int = 42):
|
||||
"""
|
||||
Split class folders from src_dir into train/val folders in dest_dir.
|
||||
Ensures every class with 2+ images keeps at least one image in validation.
|
||||
"""
|
||||
random.seed(seed)
|
||||
|
||||
train_dir = dest_dir / "train"
|
||||
val_dir = dest_dir / "val"
|
||||
|
||||
if dest_dir.exists():
|
||||
print(f"Cleaning existing split directory: {dest_dir}")
|
||||
shutil.rmtree(dest_dir)
|
||||
|
||||
train_dir.mkdir(parents=True, exist_ok=True)
|
||||
val_dir.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
exclude_dirs = {dest_dir.name, "train", "val"}
|
||||
class_dirs = [d for d in src_dir.iterdir() if d.is_dir() and d.name not in exclude_dirs]
|
||||
class_dirs.sort()
|
||||
|
||||
print(f"Found {len(class_dirs)} product classes in {src_dir}")
|
||||
|
||||
total_train = 0
|
||||
total_val = 0
|
||||
|
||||
for c_dir in class_dirs:
|
||||
class_name = c_dir.name
|
||||
images = sorted(
|
||||
[f for f in c_dir.iterdir() if is_image_file(f)],
|
||||
key=lambda p: p.name,
|
||||
)
|
||||
random.shuffle(images)
|
||||
|
||||
num_images = len(images)
|
||||
if num_images == 0:
|
||||
print(f"Warning: Class '{class_name}' has 0 images. Skipping.")
|
||||
continue
|
||||
|
||||
class_train_dir = train_dir / class_name
|
||||
class_val_dir = val_dir / class_name
|
||||
class_train_dir.mkdir(parents=True, exist_ok=True)
|
||||
class_val_dir.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
if num_images == 1:
|
||||
train_images = images
|
||||
val_images = images
|
||||
elif num_images == 2:
|
||||
train_images = [images[0]]
|
||||
val_images = [images[1]]
|
||||
else:
|
||||
split_idx = max(1, int(num_images * split_ratio))
|
||||
split_idx = min(split_idx, num_images - 1)
|
||||
train_images = images[:split_idx]
|
||||
val_images = images[split_idx:]
|
||||
|
||||
for img in train_images:
|
||||
shutil.copy(img, class_train_dir / img.name)
|
||||
total_train += 1
|
||||
|
||||
for img in val_images:
|
||||
shutil.copy(img, class_val_dir / img.name)
|
||||
total_val += 1
|
||||
|
||||
print(
|
||||
f" Class '{class_name}': {len(train_images)} train, "
|
||||
f"{len(val_images)} val (total: {num_images})"
|
||||
)
|
||||
|
||||
print(f"Dataset split completed: {total_train} train images, {total_val} validation images.")
|
||||
print(f"Split dataset located at: {dest_dir.absolute()}")
|
||||
|
||||
|
||||
def train_model(args):
|
||||
"""Handles training the YOLO classification model."""
|
||||
src_path = Path(args.src_dir).resolve()
|
||||
dest_path = Path(args.split_dir).resolve()
|
||||
|
||||
if not src_path.is_dir():
|
||||
print(f"Error: Source dataset directory not found: {src_path}")
|
||||
sys.exit(1)
|
||||
|
||||
print(f"--- Preparing Dataset from {src_path} ---")
|
||||
split_dataset(src_path, dest_path, split_ratio=args.split_ratio)
|
||||
|
||||
model_path = Path(args.model).resolve()
|
||||
print(f"\n--- Initializing YOLO Model ({model_path}) ---")
|
||||
model = YOLO(str(model_path))
|
||||
|
||||
if args.device:
|
||||
device = args.device
|
||||
else:
|
||||
device = "0" if torch.cuda.is_available() else "cpu"
|
||||
print(f"Using device: {device}")
|
||||
|
||||
print("\n--- Starting Training ---")
|
||||
results = model.train(
|
||||
data=str(dest_path),
|
||||
epochs=args.epochs,
|
||||
imgsz=args.imgsz,
|
||||
batch=args.batch,
|
||||
device=device,
|
||||
project=str(Path(args.project).resolve()),
|
||||
name=args.name,
|
||||
exist_ok=True,
|
||||
workers=args.workers,
|
||||
lr0=args.lr,
|
||||
optimizer=args.optimizer,
|
||||
seed=42,
|
||||
)
|
||||
|
||||
best_weights = Path(results.save_dir) / "weights" / "best.pt"
|
||||
output_path = Path(args.output).resolve() if args.output else classifier_output_path(args.epochs)
|
||||
output_path.parent.mkdir(parents=True, exist_ok=True)
|
||||
shutil.copy2(best_weights, output_path)
|
||||
|
||||
print("\nTraining completed successfully!")
|
||||
print(f"Run weights saved at: {best_weights}")
|
||||
print(f"Published model saved at: {output_path}")
|
||||
|
||||
if args.export:
|
||||
print("\n--- Exporting model to ONNX format ---")
|
||||
try:
|
||||
export_model = YOLO(str(output_path))
|
||||
onnx_path = Path(export_model.export(format="onnx"))
|
||||
dated_onnx = output_path.with_suffix(".onnx")
|
||||
if onnx_path.resolve() != dated_onnx.resolve():
|
||||
shutil.copy2(onnx_path, dated_onnx)
|
||||
print(f"Model exported successfully to: {dated_onnx}")
|
||||
except Exception as e:
|
||||
print(f"Warning: ONNX export failed: {e}")
|
||||
|
||||
print("\nYou can run predictions with:")
|
||||
print(f" uv run python {Path(__file__).name} predict --image <image_path> --model {output_path}")
|
||||
|
||||
|
||||
def predict_image(args):
|
||||
"""Runs classification inference on a single image."""
|
||||
model_path = Path(args.model).resolve()
|
||||
image_path = Path(args.image).resolve()
|
||||
|
||||
if not model_path.exists():
|
||||
print(f"Error: Model weights not found at {model_path}")
|
||||
sys.exit(1)
|
||||
|
||||
if not image_path.exists():
|
||||
print(f"Error: Target image file not found at {image_path}")
|
||||
sys.exit(1)
|
||||
|
||||
print(f"Loading model from {model_path}...")
|
||||
model = YOLO(str(model_path))
|
||||
|
||||
print(f"Running prediction on {image_path}...")
|
||||
results = model(str(image_path))
|
||||
|
||||
for result in results:
|
||||
probs = result.probs
|
||||
top1_idx = probs.top1
|
||||
top1_conf = float(probs.top1conf)
|
||||
top1_name = result.names[top1_idx]
|
||||
|
||||
print("\n=== Classification Results ===")
|
||||
print(f"Top-1 Prediction: {top1_name} (Confidence: {top1_conf:.4f})")
|
||||
print("\nAll Probabilities:")
|
||||
|
||||
sorted_probs = sorted(
|
||||
[(result.names[i], float(val)) for i, val in enumerate(probs.data)],
|
||||
key=lambda x: x[1],
|
||||
reverse=True,
|
||||
)
|
||||
for name, score in sorted_probs:
|
||||
print(f" {name}: {score:.4f}")
|
||||
|
||||
|
||||
def main():
|
||||
parser = argparse.ArgumentParser(
|
||||
description="Ultralytics YOLO classification utility for produk-pfm packaging photos."
|
||||
)
|
||||
subparsers = parser.add_subparsers(dest="command", required=True, help="Command to run")
|
||||
|
||||
train_parser = subparsers.add_parser("train", help="Train a classification model")
|
||||
train_parser.add_argument(
|
||||
"--src-dir",
|
||||
type=str,
|
||||
default=str(DEFAULT_DATASET_DIR),
|
||||
help=f"Source dataset directory with one class folder per product (default: {DEFAULT_DATASET_DIR.name})",
|
||||
)
|
||||
train_parser.add_argument(
|
||||
"--split-dir",
|
||||
type=str,
|
||||
default=str(DEFAULT_SPLIT_DIR),
|
||||
help="Output split dataset directory",
|
||||
)
|
||||
train_parser.add_argument(
|
||||
"--split-ratio",
|
||||
type=float,
|
||||
default=0.8,
|
||||
help="Train/val split ratio for classes with 3+ images (default: 0.8)",
|
||||
)
|
||||
train_parser.add_argument(
|
||||
"--model",
|
||||
type=str,
|
||||
default=str(DEFAULT_MODEL),
|
||||
help="Pretrained model (e.g. yolo26n-cls.pt, yolo11n-cls.pt, yolov8n-cls.pt)",
|
||||
)
|
||||
train_parser.add_argument(
|
||||
"--epochs",
|
||||
type=int,
|
||||
default=DEFAULT_EPOCHS,
|
||||
help=f"Number of training epochs (default: {DEFAULT_EPOCHS})",
|
||||
)
|
||||
train_parser.add_argument(
|
||||
"--output",
|
||||
type=str,
|
||||
default=None,
|
||||
help=(
|
||||
"Published .pt output path (default: "
|
||||
"models/produk-pfm-classifier-26n-{epochs}e-{YYYY-MM-DD}.pt)"
|
||||
),
|
||||
)
|
||||
train_parser.add_argument("--imgsz", type=int, default=224, help="Target image size for classification")
|
||||
train_parser.add_argument("--batch", type=int, default=8, help="Batch size for training")
|
||||
train_parser.add_argument(
|
||||
"--device",
|
||||
type=str,
|
||||
default=None,
|
||||
help="Device to run on (e.g. 0 or 'cpu'). Default is GPU if available.",
|
||||
)
|
||||
train_parser.add_argument(
|
||||
"--project",
|
||||
type=str,
|
||||
default=str(DEFAULT_PROJECT),
|
||||
help="Project output folder name",
|
||||
)
|
||||
train_parser.add_argument("--name", type=str, default="train", help="Experiment name")
|
||||
train_parser.add_argument("--workers", type=int, default=4, help="Number of data loading workers")
|
||||
train_parser.add_argument("--lr", type=float, default=0.01, help="Initial learning rate")
|
||||
train_parser.add_argument(
|
||||
"--optimizer",
|
||||
type=str,
|
||||
default="auto",
|
||||
choices=["SGD", "Adam", "AdamW", "RMSProp", "auto"],
|
||||
help="Optimizer to use",
|
||||
)
|
||||
train_parser.add_argument(
|
||||
"--export",
|
||||
action="store_true",
|
||||
default=True,
|
||||
help="Export model to ONNX after training",
|
||||
)
|
||||
|
||||
predict_parser = subparsers.add_parser("predict", help="Predict class of an image")
|
||||
predict_parser.add_argument("--image", type=str, required=True, help="Path to image file")
|
||||
predict_parser.add_argument(
|
||||
"--model",
|
||||
type=str,
|
||||
default=str(latest_classifier_weights()),
|
||||
help="Path to trained YOLO .pt model weights (default: newest models/produk-pfm-classifier-*.pt)",
|
||||
)
|
||||
|
||||
args = parser.parse_args()
|
||||
|
||||
if args.command == "train":
|
||||
train_model(args)
|
||||
elif args.command == "predict":
|
||||
predict_image(args)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1 @@
|
||||
<svg fill="none" xmlns="http://www.w3.org/2000/svg" viewBox="0 0 1155 1000"><path d="m577.3 0 577.4 1000H0z" fill="#fff"/></svg>
|
||||
|
After Width: | Height: | Size: 128 B |
@@ -0,0 +1 @@
|
||||
<svg fill="none" xmlns="http://www.w3.org/2000/svg" viewBox="0 0 16 16"><path fill-rule="evenodd" clip-rule="evenodd" d="M1.5 2.5h13v10a1 1 0 0 1-1 1h-11a1 1 0 0 1-1-1zM0 1h16v11.5a2.5 2.5 0 0 1-2.5 2.5h-11A2.5 2.5 0 0 1 0 12.5zm3.75 4.5a.75.75 0 1 0 0-1.5.75.75 0 0 0 0 1.5M7 4.75a.75.75 0 1 1-1.5 0 .75.75 0 0 1 1.5 0m1.75.75a.75.75 0 1 0 0-1.5.75.75 0 0 0 0 1.5" fill="#666"/></svg>
|
||||
|
After Width: | Height: | Size: 385 B |
@@ -0,0 +1,267 @@
|
||||
import { NextRequest, NextResponse } from "next/server";
|
||||
import fs from "fs";
|
||||
import path from "path";
|
||||
import { Client } from "@gradio/client";
|
||||
import { query } from "../../../db";
|
||||
|
||||
export const maxDuration = 120; // Allow up to 120 seconds for slow model inference
|
||||
|
||||
export async function GET(req: NextRequest) {
|
||||
try {
|
||||
const { searchParams } = new URL(req.url);
|
||||
const action = searchParams.get("action") || "list";
|
||||
const runId = searchParams.get("runId");
|
||||
const imageType = searchParams.get("imageType"); // 'do', 'product' or null for all
|
||||
|
||||
if (runId) {
|
||||
const runRes = await query(`
|
||||
SELECT id, image_path, engine, status, ocr_result, time_elapsed_ms, image_type, created_at
|
||||
FROM arena_runs
|
||||
WHERE id = $1
|
||||
`, [parseInt(runId)]);
|
||||
|
||||
if (runRes.rowCount === 0) {
|
||||
return NextResponse.json({ success: false, error: "Run not found" }, { status: 404 });
|
||||
}
|
||||
return NextResponse.json({ success: true, run: runRes.rows[0] });
|
||||
}
|
||||
|
||||
if (action === "stats") {
|
||||
let queryText = `
|
||||
SELECT
|
||||
engine,
|
||||
COUNT(*)::integer as total_runs,
|
||||
COUNT(CASE WHEN status = 'done' THEN 1 END)::integer as success_runs,
|
||||
COUNT(CASE WHEN status = 'failed' THEN 1 END)::integer as failed_runs,
|
||||
ROUND(AVG(CASE WHEN status = 'done' THEN time_elapsed_ms END))::integer as avg_time_ms,
|
||||
MIN(CASE WHEN status = 'done' THEN time_elapsed_ms END)::integer as min_time_ms,
|
||||
MAX(CASE WHEN status = 'done' THEN time_elapsed_ms END)::integer as max_time_ms
|
||||
FROM arena_runs
|
||||
`;
|
||||
const params: any[] = [];
|
||||
if (imageType === "do" || imageType === "product") {
|
||||
queryText += ` WHERE image_type = $1`;
|
||||
params.push(imageType);
|
||||
}
|
||||
queryText += ` GROUP BY engine`;
|
||||
|
||||
const statsRes = await query(queryText, params);
|
||||
return NextResponse.json({ success: true, stats: statsRes.rows });
|
||||
}
|
||||
|
||||
const limit = parseInt(searchParams.get("limit") || "50");
|
||||
let queryText = `
|
||||
SELECT id, image_path, engine, status, time_elapsed_ms, image_type, created_at
|
||||
FROM arena_runs
|
||||
`;
|
||||
const params: any[] = [];
|
||||
if (imageType === "do" || imageType === "product") {
|
||||
queryText += ` WHERE image_type = $1`;
|
||||
params.push(imageType);
|
||||
}
|
||||
queryText += ` ORDER BY created_at DESC LIMIT $${params.length + 1}`;
|
||||
params.push(limit);
|
||||
|
||||
const runsRes = await query(queryText, params);
|
||||
return NextResponse.json({ success: true, runs: runsRes.rows });
|
||||
} catch (error: any) {
|
||||
console.error("Failed to fetch arena runs/stats:", error);
|
||||
return NextResponse.json({ success: false, error: error.message }, { status: 500 });
|
||||
}
|
||||
}
|
||||
|
||||
export async function POST(req: NextRequest) {
|
||||
const startTime = Date.now();
|
||||
let engine: string | undefined;
|
||||
let image: string | undefined;
|
||||
let imageType = "do";
|
||||
try {
|
||||
const body = await req.json().catch(() => ({}));
|
||||
engine = body.engine;
|
||||
image = body.image;
|
||||
|
||||
if (!engine || !image) {
|
||||
return NextResponse.json({ error: "Missing engine or image" }, { status: 400 });
|
||||
}
|
||||
|
||||
imageType = body.imageType || "do";
|
||||
if (typeof image === "string") {
|
||||
if (image.startsWith("/produk-pfm/") || image.includes("produk-pfm") || image.includes("Product")) {
|
||||
imageType = "product";
|
||||
} else if (image.startsWith("/do-pfm/") || image.includes("do-pfm")) {
|
||||
imageType = "do";
|
||||
}
|
||||
}
|
||||
|
||||
let imageBuffer: Buffer;
|
||||
let base64Image = "";
|
||||
|
||||
// 1. Resolve image (local file or base64)
|
||||
if (typeof image === "string" && (image.startsWith("/do-pfm/") || image.startsWith("/produk-pfm/"))) {
|
||||
// Resolve path in public folder
|
||||
const cleanPath = image.startsWith("/") ? image.slice(1) : image;
|
||||
const filePath = path.join(process.cwd(), "public", cleanPath);
|
||||
|
||||
if (!fs.existsSync(filePath)) {
|
||||
return NextResponse.json({ error: `File not found on server: ${image}` }, { status: 404 });
|
||||
}
|
||||
imageBuffer = fs.readFileSync(filePath);
|
||||
base64Image = `data:image/jpeg;base64,${imageBuffer.toString("base64")}`;
|
||||
} else if (typeof image === "string" && image.startsWith("data:")) {
|
||||
// Base64 data URI
|
||||
base64Image = image;
|
||||
const base64Data = image.split(",")[1];
|
||||
imageBuffer = Buffer.from(base64Data, "base64");
|
||||
} else if (typeof image === "string") {
|
||||
// Raw base64 string
|
||||
base64Image = `data:image/jpeg;base64,${image}`;
|
||||
imageBuffer = Buffer.from(image, "base64");
|
||||
} else {
|
||||
return NextResponse.json({ error: "Invalid image format" }, { status: 400 });
|
||||
}
|
||||
|
||||
let outputText = "";
|
||||
|
||||
// 2. Route to the requested OCR engine
|
||||
if (engine === "deepseek") {
|
||||
const blob = new Blob([new Uint8Array(imageBuffer)], { type: "image/jpeg" });
|
||||
const gradioUrl = process.env.DEEPSEEK_GRADIO_URL || "http://host.docker.internal:7873/v2/";
|
||||
const client = await Client.connect(gradioUrl);
|
||||
const result = await client.predict(2, [blob, "Default", "Markdown", ""]);
|
||||
const data = result.data as any[];
|
||||
outputText = data[1] || data[0] || "";
|
||||
|
||||
} else if (engine === "lightonocr") {
|
||||
const url = process.env.LIGHTONOCR_API_URL || "http://host.docker.internal:7678/layout-parsing";
|
||||
const res = await fetch(url, {
|
||||
method: "POST",
|
||||
headers: { "Content-Type": "application/json" },
|
||||
body: JSON.stringify({
|
||||
file: base64Image,
|
||||
useLayoutDetection: false
|
||||
})
|
||||
});
|
||||
if (!res.ok) {
|
||||
throw new Error(`LightOnOCR backend error: ${res.status} ${await res.text()}`);
|
||||
}
|
||||
const data = await res.json();
|
||||
outputText = data.result?.layoutParsingResults?.[0]?.markdown?.text || "";
|
||||
|
||||
} else if (engine === "nemotron") {
|
||||
const url = process.env.NEMOTRON_API_URL || "http://host.docker.internal:8009/layout-parsing";
|
||||
const res = await fetch(url, {
|
||||
method: "POST",
|
||||
headers: { "Content-Type": "application/json" },
|
||||
body: JSON.stringify({
|
||||
file: base64Image,
|
||||
model: "Multilingual (en, zh, ja, ko, ru, …)",
|
||||
merge_level: "layout"
|
||||
})
|
||||
});
|
||||
if (!res.ok) {
|
||||
throw new Error(`Nemotron backend error: ${res.status} ${await res.text()}`);
|
||||
}
|
||||
const data = await res.json();
|
||||
outputText = data.result?.layoutParsingResults?.[0]?.markdown?.text || "";
|
||||
|
||||
} else if (engine === "paddle") {
|
||||
const url = process.env.PIPELINE_URL || "http://paddleocr-pipeline-api:8090/layout-parsing";
|
||||
const rawB64 = base64Image.includes(",") ? base64Image.split(",")[1] : base64Image;
|
||||
const res = await fetch(url, {
|
||||
method: "POST",
|
||||
headers: { "Content-Type": "application/json" },
|
||||
body: JSON.stringify({
|
||||
file: rawB64,
|
||||
matchHistoryJob: false,
|
||||
useLayoutDetection: true,
|
||||
fileType: 1,
|
||||
useDocUnwarping: false,
|
||||
useDocOrientationClassify: false
|
||||
})
|
||||
});
|
||||
if (!res.ok) {
|
||||
throw new Error(`PaddleOCR backend error: ${res.status} ${await res.text()}`);
|
||||
}
|
||||
const data = await res.json();
|
||||
const pipelineResult = data.result || data;
|
||||
outputText = pipelineResult?.layoutParsingResults?.[0]?.markdown?.text || "";
|
||||
|
||||
} else if (engine === "dots") {
|
||||
// Calling python API directly
|
||||
const url = process.env.DOTS_API_URL || "http://host.docker.internal:7872/layout-parsing";
|
||||
const res = await fetch(url, {
|
||||
method: "POST",
|
||||
headers: { "Content-Type": "application/json" },
|
||||
body: JSON.stringify({
|
||||
file: base64Image,
|
||||
promptLabel: "ocr",
|
||||
useLayoutDetection: true
|
||||
})
|
||||
});
|
||||
if (!res.ok) {
|
||||
throw new Error(`Dots OCR backend error: ${res.status} ${await res.text()}`);
|
||||
}
|
||||
const data = await res.json();
|
||||
outputText = data.result?.layoutParsingResults?.[0]?.markdown?.text || "";
|
||||
|
||||
} else if (engine === "glm") {
|
||||
const gradioUrl = process.env.GLM_GRADIO_URL || "http://host.docker.internal:7875/";
|
||||
const client = await Client.connect(gradioUrl);
|
||||
const result = await client.predict(2, ["Text", base64Image, 1024, 60]);
|
||||
const data = result.data as any[];
|
||||
outputText = data[0] || "";
|
||||
|
||||
} else {
|
||||
return NextResponse.json({ error: `Unknown engine: ${engine}` }, { status: 400 });
|
||||
}
|
||||
|
||||
const elapsedMs = Date.now() - startTime;
|
||||
|
||||
// Record successful run
|
||||
try {
|
||||
const loggedImagePath = (typeof image === "string" && image.startsWith("data:"))
|
||||
? `[Base64 Upload: ${image.length} chars]`
|
||||
: (typeof image === "string" && image.length > 500)
|
||||
? `[Raw Base64: ${image.length} chars]`
|
||||
: image;
|
||||
await query(
|
||||
`INSERT INTO arena_runs (image_path, engine, status, ocr_result, time_elapsed_ms, image_type)
|
||||
VALUES ($1, $2, $3, $4, $5, $6)`,
|
||||
[loggedImagePath, engine, "done", outputText, elapsedMs, imageType]
|
||||
);
|
||||
} catch (dbErr) {
|
||||
console.error("Failed to log success to arena_runs:", dbErr);
|
||||
}
|
||||
|
||||
return NextResponse.json({
|
||||
success: true,
|
||||
text: outputText,
|
||||
elapsedMs
|
||||
});
|
||||
|
||||
} catch (error: any) {
|
||||
console.error("OCR Arena proxy error:", error);
|
||||
const elapsedMs = Date.now() - startTime;
|
||||
|
||||
// Record failed run
|
||||
try {
|
||||
const loggedImagePath = (typeof image === "string" && image.startsWith("data:"))
|
||||
? `[Base64 Upload: ${image.length} chars]`
|
||||
: (typeof image === "string" && image.length > 500)
|
||||
? `[Raw Base64: ${image.length} chars]`
|
||||
: image;
|
||||
await query(
|
||||
`INSERT INTO arena_runs (image_path, engine, status, ocr_result, time_elapsed_ms, image_type)
|
||||
VALUES ($1, $2, $3, $4, $5, $6)`,
|
||||
[loggedImagePath || "unknown", engine || "unknown", "failed", error.message || "Unknown error", elapsedMs, imageType]
|
||||
);
|
||||
} catch (dbErr) {
|
||||
console.error("Failed to log failure to arena_runs:", dbErr);
|
||||
}
|
||||
|
||||
return NextResponse.json({
|
||||
success: false,
|
||||
error: error.message || "Failed to process OCR request"
|
||||
}, { status: 500 });
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,55 @@
|
||||
import { NextRequest, NextResponse } from "next/server";
|
||||
import fs from "fs";
|
||||
import path from "path";
|
||||
import { query } from "../../../db";
|
||||
import crypto from "crypto";
|
||||
|
||||
const UPLOADS_DIR = "/uploads";
|
||||
const PUBLIC_DIR = path.join(process.cwd(), "public/do-pfm");
|
||||
|
||||
export async function POST(req: NextRequest) {
|
||||
try {
|
||||
const { filename, image } = await req.json();
|
||||
|
||||
if (!filename || !image) {
|
||||
return NextResponse.json({ error: "Filename and image base64 data are required" }, { status: 400 });
|
||||
}
|
||||
|
||||
const safeFile = path.basename(filename);
|
||||
|
||||
const isSample = fs.existsSync(path.join(PUBLIC_DIR, safeFile));
|
||||
const filePath = isSample
|
||||
? path.join(PUBLIC_DIR, safeFile)
|
||||
: path.join(UPLOADS_DIR, safeFile);
|
||||
|
||||
const base64Data = image.replace(/^data:image\/\w+;base64,/, "");
|
||||
const buffer = Buffer.from(base64Data, "base64");
|
||||
|
||||
// Write file to disk
|
||||
fs.writeFileSync(filePath, buffer);
|
||||
console.log(`Cropped file saved successfully at ${filePath}`);
|
||||
|
||||
// Update database fields
|
||||
const fileHash = crypto.createHash("sha256").update(buffer).digest("hex");
|
||||
const stats = fs.statSync(filePath);
|
||||
|
||||
// Update document to unparsed state since layout changes
|
||||
await query(
|
||||
"UPDATE documents SET size = $1, file_hash = $2, parsed = false, layout_parsing_result = NULL WHERE filename = $3",
|
||||
[stats.size, fileHash, filename]
|
||||
);
|
||||
|
||||
// Clear old items for this document
|
||||
const docRes = await query("SELECT id FROM documents WHERE filename = $1", [filename]);
|
||||
if (docRes.rowCount && docRes.rowCount > 0) {
|
||||
const docId = docRes.rows[0].id;
|
||||
await query("DELETE FROM ocr_items WHERE document_id = $1", [docId]);
|
||||
}
|
||||
|
||||
return NextResponse.json({ success: true });
|
||||
} catch (error: unknown) {
|
||||
console.error("Error cropping file:", error);
|
||||
const message = error instanceof Error ? error.message : "Internal server error";
|
||||
return NextResponse.json({ error: message }, { status: 500 });
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,46 @@
|
||||
import { NextRequest, NextResponse } from "next/server";
|
||||
import fs from "fs";
|
||||
import path from "path";
|
||||
|
||||
const UPLOADS_DIR = "/uploads";
|
||||
|
||||
export async function GET(req: NextRequest) {
|
||||
try {
|
||||
const filename = req.nextUrl.searchParams.get("file");
|
||||
if (!filename) {
|
||||
return NextResponse.json({ error: "File name is required" }, { status: 400 });
|
||||
}
|
||||
|
||||
const safeFile = path.basename(filename);
|
||||
const filePath = path.join(UPLOADS_DIR, safeFile);
|
||||
|
||||
if (!fs.existsSync(filePath)) {
|
||||
return NextResponse.json({ error: "File not found" }, { status: 404 });
|
||||
}
|
||||
|
||||
// Determine content type based on extension
|
||||
const ext = path.extname(safeFile).toLowerCase();
|
||||
let contentType = "application/octet-stream";
|
||||
if (ext === ".jpg" || ext === ".jpeg") {
|
||||
contentType = "image/jpeg";
|
||||
} else if (ext === ".png") {
|
||||
contentType = "image/png";
|
||||
} else if (ext === ".gif") {
|
||||
contentType = "image/gif";
|
||||
} else if (ext === ".pdf") {
|
||||
contentType = "application/pdf";
|
||||
}
|
||||
|
||||
const fileBuffer = fs.readFileSync(filePath);
|
||||
return new Response(fileBuffer, {
|
||||
headers: {
|
||||
"Content-Type": contentType,
|
||||
"Cache-Control": "public, max-age=31536000, immutable"
|
||||
}
|
||||
});
|
||||
} catch (error: unknown) {
|
||||
console.error("Error serving file from uploads:", error);
|
||||
const message = error instanceof Error ? error.message : "Internal server error";
|
||||
return NextResponse.json({ error: message }, { status: 500 });
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,161 @@
|
||||
import { NextRequest, NextResponse } from "next/server";
|
||||
import {
|
||||
getGpuInfo,
|
||||
getContainerStatus,
|
||||
manageContainer,
|
||||
recreateContainer,
|
||||
getEnvSettings,
|
||||
saveEnvSettings,
|
||||
getProcessName,
|
||||
unloadOtherEngines
|
||||
} from "../../../utils/docker";
|
||||
|
||||
export async function GET(req: NextRequest) {
|
||||
try {
|
||||
const gpus = await getGpuInfo();
|
||||
const settings = await getEnvSettings();
|
||||
|
||||
const containers = {
|
||||
nginx: await getContainerStatus("paddleocr-nginx"),
|
||||
vllmServer: await getContainerStatus("paddleocr-vllm-server"),
|
||||
pipelineApi: await getContainerStatus("paddleocr-pipeline-api"),
|
||||
gradioUi: await getContainerStatus("paddleocr-gradio-ui"),
|
||||
pfmWebApp: await getContainerStatus("paddleocr-pfm-web-app"),
|
||||
db: await getContainerStatus("paddleocr-db")
|
||||
};
|
||||
|
||||
return NextResponse.json({
|
||||
success: true,
|
||||
gpus,
|
||||
settings,
|
||||
containers
|
||||
});
|
||||
} catch (error: any) {
|
||||
console.error("Failed to fetch GPU/container status:", error);
|
||||
return NextResponse.json({ success: false, error: error.message }, { status: 500 });
|
||||
}
|
||||
}
|
||||
|
||||
export async function POST(req: NextRequest) {
|
||||
try {
|
||||
const body = await req.json().catch(() => ({}));
|
||||
const { action } = body;
|
||||
|
||||
if (action === "kill") {
|
||||
const pid = parseInt(body.pid);
|
||||
if (!pid || isNaN(pid)) {
|
||||
return NextResponse.json({ success: false, error: "Invalid PID" }, { status: 400 });
|
||||
}
|
||||
|
||||
// Check if process is protected (same rules as admin_panel.py)
|
||||
const procName = getProcessName(pid);
|
||||
const procNameLower = procName.toLowerCase();
|
||||
const protectedKeywords = ["rustdesk", "xorg", "nginx", "systemd", "dockerd", "python3", "node"];
|
||||
if (anyKeywordMatch(procNameLower, protectedKeywords)) {
|
||||
return NextResponse.json({
|
||||
success: false,
|
||||
error: `Operation Denied: Process ${pid} (${procName || "system"}) is protected and cannot be killed.`
|
||||
}, { status: 403 });
|
||||
}
|
||||
|
||||
try {
|
||||
process.kill(pid, 9);
|
||||
return NextResponse.json({ success: true, message: `Successfully killed process ${pid}` });
|
||||
} catch (err: any) {
|
||||
return NextResponse.json({ success: false, error: `Failed to kill process: ${err.message}` }, { status: 500 });
|
||||
}
|
||||
}
|
||||
|
||||
if (action === "container") {
|
||||
const { containerName, containerAction } = body;
|
||||
const validActions = ["start", "stop", "restart"];
|
||||
const validContainers = [
|
||||
"paddleocr-nginx",
|
||||
"paddleocr-vllm-server",
|
||||
"paddleocr-pipeline-api",
|
||||
"paddleocr-gradio-ui",
|
||||
"paddleocr-pfm-web-app",
|
||||
"paddleocr-db"
|
||||
];
|
||||
|
||||
if (!validActions.includes(containerAction) || !validContainers.includes(containerName)) {
|
||||
return NextResponse.json({ success: false, error: "Invalid container name or action" }, { status: 400 });
|
||||
}
|
||||
|
||||
// Prevent self-stopping nextjs app accidentally through UI
|
||||
if (containerName === "paddleocr-pfm-web-app" && containerAction === "stop") {
|
||||
return NextResponse.json({ success: false, error: "Cannot stop the active web application container itself." }, { status: 400 });
|
||||
}
|
||||
|
||||
await manageContainer(containerName, containerAction);
|
||||
return NextResponse.json({
|
||||
success: true,
|
||||
message: `Command 'docker-compose ${containerAction} ${containerName.replace("paddleocr-", "")}' executed successfully.`
|
||||
});
|
||||
}
|
||||
|
||||
if (action === "saveSettings") {
|
||||
const { cudaDevices } = body;
|
||||
if (typeof cudaDevices !== "string" || cudaDevices.trim() === "") {
|
||||
return NextResponse.json({ success: false, error: "Invalid GPU allocation settings" }, { status: 400 });
|
||||
}
|
||||
|
||||
const cleanCuda = cudaDevices.trim();
|
||||
await saveEnvSettings(cleanCuda);
|
||||
|
||||
// Recreate GPU containers to apply env settings
|
||||
try {
|
||||
await recreateContainer("paddleocr-vllm-server", cleanCuda);
|
||||
} catch (err: any) {
|
||||
console.error("Failed to recreate vllm-server container:", err);
|
||||
}
|
||||
|
||||
try {
|
||||
await recreateContainer("paddleocr-pipeline-api", cleanCuda);
|
||||
} catch (err: any) {
|
||||
console.error("Failed to recreate pipeline-api container:", err);
|
||||
}
|
||||
|
||||
return NextResponse.json({
|
||||
success: true,
|
||||
message: `GPU settings updated to device index ${cleanCuda}. Core services recreated successfully.`
|
||||
});
|
||||
}
|
||||
|
||||
if (action === "unload") {
|
||||
const { stopped, failed } = await unloadOtherEngines();
|
||||
if (stopped.length === 0 && failed.length === 0) {
|
||||
return NextResponse.json({
|
||||
success: true,
|
||||
message: "All other OCR engines are already stopped/unloaded."
|
||||
});
|
||||
}
|
||||
|
||||
let msg = "";
|
||||
if (stopped.length > 0) {
|
||||
msg += `Successfully stopped/unloaded: ${stopped.join(", ")}. `;
|
||||
}
|
||||
if (failed.length > 0) {
|
||||
msg += `Failed to stop: ${failed.join(", ")}.`;
|
||||
}
|
||||
|
||||
return NextResponse.json({
|
||||
success: failed.length === 0,
|
||||
message: msg.trim(),
|
||||
error: failed.length > 0 ? `Failed to stop some containers: ${failed.join(", ")}` : undefined
|
||||
});
|
||||
}
|
||||
|
||||
return NextResponse.json({ success: false, error: "Invalid API action" }, { status: 400 });
|
||||
} catch (error: any) {
|
||||
console.error("GPU API POST error:", error);
|
||||
return NextResponse.json({ success: false, error: error.message }, { status: 500 });
|
||||
}
|
||||
}
|
||||
|
||||
function anyKeywordMatch(str: string, keywords: string[]): boolean {
|
||||
for (const kw of keywords) {
|
||||
if (str.includes(kw)) return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
@@ -0,0 +1,201 @@
|
||||
import { NextRequest, NextResponse } from "next/server";
|
||||
import { query, cleanupAndReindexItems } from "../../../db";
|
||||
import fs from "fs";
|
||||
import path from "path";
|
||||
|
||||
const UPLOADS_DIR = "/uploads";
|
||||
|
||||
export async function GET(req: NextRequest) {
|
||||
try {
|
||||
const fileParam = req.nextUrl.searchParams.get("file");
|
||||
|
||||
if (fileParam) {
|
||||
const safeFile = path.basename(fileParam);
|
||||
|
||||
// 1. Try to load from database first
|
||||
const docRes = await query(
|
||||
"SELECT id, layout_parsing_result, metadata FROM documents WHERE filename = $1",
|
||||
[safeFile]
|
||||
);
|
||||
|
||||
if (docRes.rowCount && docRes.rowCount > 0) {
|
||||
const doc = docRes.rows[0];
|
||||
const docId = doc.id;
|
||||
const pipelineResult = doc.layout_parsing_result;
|
||||
|
||||
// Clean up and re-index invalid items first
|
||||
await cleanupAndReindexItems(docId);
|
||||
|
||||
// Fetch items
|
||||
const itemsRes = await query(
|
||||
`SELECT row_index,
|
||||
kode_barang, nama_barang, banyak, jumlah,
|
||||
is_flagged, remark
|
||||
FROM ocr_items
|
||||
WHERE document_id = $1
|
||||
ORDER BY row_index`,
|
||||
[docId]
|
||||
);
|
||||
|
||||
const items = itemsRes.rows.map(row => ({
|
||||
kodeBarang: row.kode_barang,
|
||||
namaBarang: row.nama_barang,
|
||||
banyak: row.banyak,
|
||||
jumlah: row.jumlah
|
||||
}));
|
||||
|
||||
const flagged: Record<number, boolean> = {};
|
||||
const remarks: Record<number, string> = {};
|
||||
|
||||
itemsRes.rows.forEach(row => {
|
||||
if (row.is_flagged) {
|
||||
flagged[row.row_index] = true;
|
||||
}
|
||||
if (row.remark && row.remark.trim()) {
|
||||
remarks[row.row_index] = row.remark;
|
||||
}
|
||||
});
|
||||
|
||||
return NextResponse.json({
|
||||
errorCode: 0,
|
||||
errorMsg: "Success",
|
||||
result: pipelineResult,
|
||||
items,
|
||||
flagged,
|
||||
remarks,
|
||||
headerRemark: (doc.metadata as any)?.headerRemark || ""
|
||||
});
|
||||
}
|
||||
|
||||
// 2. Fallback to filesystem
|
||||
const jsonPath = path.join(UPLOADS_DIR, `${safeFile}.json`);
|
||||
if (fs.existsSync(jsonPath)) {
|
||||
const jsonData = fs.readFileSync(jsonPath, "utf8");
|
||||
const data = JSON.parse(jsonData);
|
||||
return NextResponse.json({
|
||||
errorCode: 0,
|
||||
errorMsg: "Success",
|
||||
result: data.result || data
|
||||
});
|
||||
}
|
||||
|
||||
return NextResponse.json({ error: "Document not found" }, { status: 404 });
|
||||
}
|
||||
|
||||
// List view: return history list from DB
|
||||
const showAll = req.nextUrl.searchParams.get("all") === "true";
|
||||
|
||||
let listRes;
|
||||
if (showAll) {
|
||||
listRes = await query(
|
||||
`SELECT id, filename, upload_time, size, parsed, is_sample, metadata,
|
||||
(SELECT COUNT(*) FROM ocr_items WHERE document_id = documents.id) as total_items,
|
||||
(SELECT COUNT(*) FROM ocr_items WHERE document_id = documents.id AND is_flagged = TRUE) as flagged_items
|
||||
FROM documents
|
||||
ORDER BY upload_time DESC`
|
||||
);
|
||||
} else {
|
||||
listRes = await query(
|
||||
`SELECT id, filename, upload_time, size, parsed, is_sample, metadata,
|
||||
(SELECT COUNT(*) FROM ocr_items WHERE document_id = documents.id) as total_items,
|
||||
(SELECT COUNT(*) FROM ocr_items WHERE document_id = documents.id AND is_flagged = TRUE) as flagged_items
|
||||
FROM documents
|
||||
WHERE is_sample = FALSE
|
||||
ORDER BY upload_time DESC`
|
||||
);
|
||||
}
|
||||
|
||||
const history = listRes.rows.map(row => ({
|
||||
id: row.id,
|
||||
filename: row.filename,
|
||||
uploadTime: row.upload_time.toISOString(),
|
||||
size: row.size,
|
||||
parsed: row.parsed,
|
||||
isSample: row.is_sample,
|
||||
metadata: row.metadata,
|
||||
totalItems: parseInt(row.total_items || "0"),
|
||||
flaggedItems: parseInt(row.flagged_items || "0")
|
||||
}));
|
||||
|
||||
return NextResponse.json({ history });
|
||||
} catch (error: unknown) {
|
||||
console.error("Error in history API route:", error);
|
||||
const message = error instanceof Error ? error.message : "Internal server error";
|
||||
return NextResponse.json({ error: message }, { status: 500 });
|
||||
}
|
||||
}
|
||||
|
||||
export async function DELETE(req: NextRequest) {
|
||||
try {
|
||||
const { filename } = await req.json();
|
||||
if (!filename) {
|
||||
return NextResponse.json({ error: "Filename is required" }, { status: 400 });
|
||||
}
|
||||
|
||||
const safeFile = path.basename(filename);
|
||||
|
||||
// Check if it exists and get its status
|
||||
const checkRes = await query(
|
||||
"SELECT id, is_sample FROM documents WHERE filename = $1",
|
||||
[safeFile]
|
||||
);
|
||||
|
||||
if (checkRes.rowCount && checkRes.rowCount > 0) {
|
||||
const doc = checkRes.rows[0];
|
||||
const isSample = doc.is_sample;
|
||||
|
||||
// Delete from DB (cascading delete will remove ocr_items)
|
||||
await query("DELETE FROM documents WHERE filename = $1", [safeFile]);
|
||||
|
||||
// If it is a custom upload, clean up files from /uploads directory
|
||||
if (!isSample) {
|
||||
const imagePath = path.join(UPLOADS_DIR, safeFile);
|
||||
const jsonPath = path.join(UPLOADS_DIR, `${safeFile}.json`);
|
||||
|
||||
if (fs.existsSync(imagePath)) {
|
||||
fs.unlinkSync(imagePath);
|
||||
}
|
||||
if (fs.existsSync(jsonPath)) {
|
||||
fs.unlinkSync(jsonPath);
|
||||
}
|
||||
}
|
||||
|
||||
return NextResponse.json({ success: true });
|
||||
}
|
||||
|
||||
return NextResponse.json({ error: "Document not found" }, { status: 404 });
|
||||
} catch (error: unknown) {
|
||||
console.error("Error in DELETE history API route:", error);
|
||||
const message = error instanceof Error ? error.message : "Internal server error";
|
||||
return NextResponse.json({ error: message }, { status: 500 });
|
||||
}
|
||||
}
|
||||
|
||||
export async function POST(req: NextRequest) {
|
||||
try {
|
||||
const { filename, remark } = await req.json();
|
||||
if (!filename) {
|
||||
return NextResponse.json({ error: "Filename is required" }, { status: 400 });
|
||||
}
|
||||
|
||||
const safeFile = path.basename(filename);
|
||||
|
||||
const valueJson = JSON.stringify(remark || "");
|
||||
const updateRes = await query(
|
||||
`UPDATE documents
|
||||
SET metadata = jsonb_set(coalesce(metadata, '{}'::jsonb), '{headerRemark}', $1::jsonb)
|
||||
WHERE filename = $2`,
|
||||
[valueJson, safeFile]
|
||||
);
|
||||
|
||||
if (updateRes.rowCount && updateRes.rowCount > 0) {
|
||||
return NextResponse.json({ success: true });
|
||||
}
|
||||
|
||||
return NextResponse.json({ error: "Document not found" }, { status: 404 });
|
||||
} catch (error: unknown) {
|
||||
console.error("Error in POST history API route:", error);
|
||||
const message = error instanceof Error ? error.message : "Internal server error";
|
||||
return NextResponse.json({ error: message }, { status: 500 });
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,294 @@
|
||||
import { NextRequest, NextResponse } from "next/server";
|
||||
import fs from "fs";
|
||||
import path from "path";
|
||||
import crypto from "crypto";
|
||||
import { query, cleanupAndReindexItems, resolveStoreFromText } from "../../../db";
|
||||
import { parseDOMetadata, sanitizeParsedMetadata } from "../../../utils/parser";
|
||||
|
||||
export async function POST(req: NextRequest) {
|
||||
try {
|
||||
const { filename } = await req.json();
|
||||
if (!filename) {
|
||||
return NextResponse.json({ error: "Filename is required" }, { status: 400 });
|
||||
}
|
||||
|
||||
const safeFile = path.basename(filename);
|
||||
|
||||
// 1. Resolve path to the public assets folder or fallback to /uploads
|
||||
let filePath = path.join(process.cwd(), "public", "do-pfm", safeFile);
|
||||
let isUpload = false;
|
||||
let isSample = true;
|
||||
|
||||
if (!fs.existsSync(filePath)) {
|
||||
filePath = path.join("/uploads", safeFile);
|
||||
if (!fs.existsSync(filePath)) {
|
||||
return NextResponse.json({ error: "File not found" }, { status: 404 });
|
||||
}
|
||||
isUpload = true;
|
||||
isSample = false;
|
||||
}
|
||||
|
||||
// Read file and compute content hash
|
||||
const fileBuffer = fs.readFileSync(filePath);
|
||||
const fileHash = crypto.createHash("sha256").update(fileBuffer).digest("hex");
|
||||
|
||||
// 2. Check if document exists in database (by filename OR hash) and is already parsed
|
||||
const checkRes = await query(
|
||||
"SELECT id, layout_parsing_result FROM documents WHERE (filename = $1 OR file_hash = $2) AND parsed = true AND layout_parsing_result IS NOT NULL",
|
||||
[safeFile, fileHash]
|
||||
);
|
||||
|
||||
if (checkRes.rowCount && checkRes.rowCount > 0) {
|
||||
const doc = checkRes.rows[0];
|
||||
const docId = doc.id;
|
||||
const pipelineResult = doc.layout_parsing_result;
|
||||
|
||||
// Clean up and re-index invalid items first
|
||||
await cleanupAndReindexItems(docId);
|
||||
|
||||
// Fetch items
|
||||
const itemsRes = await query(
|
||||
`SELECT row_index,
|
||||
kode_barang, nama_barang, banyak, jumlah,
|
||||
is_flagged, remark
|
||||
FROM ocr_items
|
||||
WHERE document_id = $1
|
||||
ORDER BY row_index`,
|
||||
[docId]
|
||||
);
|
||||
|
||||
const items = itemsRes.rows.map(row => ({
|
||||
kodeBarang: row.kode_barang,
|
||||
namaBarang: row.nama_barang,
|
||||
banyak: row.banyak,
|
||||
jumlah: row.jumlah
|
||||
}));
|
||||
|
||||
const flagged: Record<number, boolean> = {};
|
||||
const remarks: Record<number, string> = {};
|
||||
|
||||
itemsRes.rows.forEach(row => {
|
||||
if (row.is_flagged) {
|
||||
flagged[row.row_index] = true;
|
||||
}
|
||||
if (row.remark && row.remark.trim()) {
|
||||
remarks[row.row_index] = row.remark;
|
||||
}
|
||||
});
|
||||
|
||||
return NextResponse.json({
|
||||
errorCode: 0,
|
||||
errorMsg: "Success",
|
||||
result: pipelineResult,
|
||||
items,
|
||||
flagged,
|
||||
remarks
|
||||
});
|
||||
}
|
||||
|
||||
const b64 = fileBuffer.toString("base64");
|
||||
|
||||
// Form payload
|
||||
const payload = {
|
||||
file: b64,
|
||||
matchHistoryJob: false,
|
||||
useLayoutDetection: true,
|
||||
fileType: 1,
|
||||
useDocUnwarping: false,
|
||||
useDocOrientationClassify: false
|
||||
};
|
||||
|
||||
// Post to Pipeline API
|
||||
const pipelineUrl = process.env.PIPELINE_URL || "http://localhost:7871/layout-parsing";
|
||||
|
||||
console.log(`Forwarding request to pipeline API: ${pipelineUrl}`);
|
||||
const response = await fetch(pipelineUrl, {
|
||||
method: "POST",
|
||||
headers: {
|
||||
"Content-Type": "application/json"
|
||||
},
|
||||
body: JSON.stringify(payload)
|
||||
});
|
||||
|
||||
if (!response.ok) {
|
||||
const errText = await response.text();
|
||||
console.warn(`Pipeline API error for ${safeFile}: ${errText}. Marking document as parsed with default metadata.`);
|
||||
try {
|
||||
await query(
|
||||
"UPDATE documents SET parsed = true, metadata = $1 WHERE filename = $2",
|
||||
[JSON.stringify({
|
||||
tanggal: "Not Found",
|
||||
noPO: "Not Found",
|
||||
noSO: "Not Found",
|
||||
noDO: "Not Found",
|
||||
vendorInfo: "Not Found",
|
||||
customerInfo: "Not Found",
|
||||
items: []
|
||||
}), safeFile]
|
||||
);
|
||||
} catch (dbErr) {
|
||||
console.error("Failed to mark document as parsed on pipeline error:", dbErr);
|
||||
}
|
||||
return NextResponse.json({ error: `Pipeline API error: ${errText}` }, { status: response.status });
|
||||
}
|
||||
|
||||
let data = await response.json();
|
||||
|
||||
// Check if the image is not straight (tilt > 1.0 degree)
|
||||
const tilt = calculateAverageTilt(data);
|
||||
if (tilt > 1.0) {
|
||||
console.log(`Parsed document ${safeFile} is not straight (average tilt: ${tilt.toFixed(2)} deg). Re-running with unwarping and orientation classification enabled...`);
|
||||
const unwarpPayload = {
|
||||
...payload,
|
||||
useDocUnwarping: true,
|
||||
useDocOrientationClassify: true
|
||||
};
|
||||
const unwarpResponse = await fetch(pipelineUrl, {
|
||||
method: "POST",
|
||||
headers: {
|
||||
"Content-Type": "application/json"
|
||||
},
|
||||
body: JSON.stringify(unwarpPayload)
|
||||
});
|
||||
if (unwarpResponse.ok) {
|
||||
data = await unwarpResponse.json();
|
||||
console.log(`Document unwarped successfully.`);
|
||||
} else {
|
||||
console.error(`Unwarping failed with status ${unwarpResponse.status}`);
|
||||
}
|
||||
}
|
||||
|
||||
// If it was parsed from /uploads, save the JSON result on disk for fallback/compatibility
|
||||
if (isUpload) {
|
||||
fs.writeFileSync(`${filePath}.json`, JSON.stringify(data, null, 2));
|
||||
}
|
||||
|
||||
// 3. Save to database
|
||||
try {
|
||||
const pipelineResult = data.result || data;
|
||||
const page0 = pipelineResult?.layoutParsingResults?.[0] || {};
|
||||
const markdownText = page0?.markdown?.text || "";
|
||||
const rawMetadata = parseDOMetadata(markdownText);
|
||||
// Second-layer sanity check: enforces strict field formats and auto-corrects anomalies
|
||||
const docMetadata = sanitizeParsedMetadata(rawMetadata as any);
|
||||
|
||||
// Resolve store information using master database
|
||||
const resolvedStore = await resolveStoreFromText(markdownText);
|
||||
(docMetadata as any).orderUntuk = resolvedStore.orderUntuk;
|
||||
(docMetadata as any).alamat = resolvedStore.alamat;
|
||||
|
||||
const stats = fs.statSync(filePath);
|
||||
|
||||
const insertDocRes = await query(`
|
||||
INSERT INTO documents (filename, upload_time, size, parsed, metadata, layout_parsing_result, is_sample, file_hash)
|
||||
VALUES ($1, $2, $3, $4, $5, $6, $7, $8)
|
||||
ON CONFLICT (filename) DO UPDATE
|
||||
SET upload_time = EXCLUDED.upload_time,
|
||||
size = EXCLUDED.size,
|
||||
parsed = EXCLUDED.parsed,
|
||||
metadata = EXCLUDED.metadata,
|
||||
layout_parsing_result = EXCLUDED.layout_parsing_result,
|
||||
is_sample = EXCLUDED.is_sample,
|
||||
file_hash = EXCLUDED.file_hash
|
||||
RETURNING id
|
||||
`, [
|
||||
safeFile,
|
||||
stats.mtime,
|
||||
stats.size,
|
||||
true,
|
||||
JSON.stringify(docMetadata),
|
||||
JSON.stringify(pipelineResult),
|
||||
isSample,
|
||||
fileHash
|
||||
]);
|
||||
|
||||
const docId = insertDocRes.rows[0].id;
|
||||
|
||||
// Delete existing items for this document to avoid unique constraints / stale data
|
||||
await query("DELETE FROM ocr_items WHERE document_id = $1", [docId]);
|
||||
|
||||
for (let i = 0; i < docMetadata.items.length; i++) {
|
||||
const item = docMetadata.items[i];
|
||||
await query(`
|
||||
INSERT INTO ocr_items (
|
||||
document_id, row_index,
|
||||
kode_barang_original, kode_barang,
|
||||
nama_barang,
|
||||
banyak_original, banyak,
|
||||
jumlah_original, jumlah,
|
||||
is_flagged, remark
|
||||
)
|
||||
VALUES ($1, $2, $3, $3, $4, $5, $5, $6, $6, false, '')
|
||||
`, [
|
||||
docId,
|
||||
i,
|
||||
item.kodeBarang,
|
||||
item.namaBarang,
|
||||
item.banyak,
|
||||
item.jumlah
|
||||
]);
|
||||
}
|
||||
|
||||
// Return wrapped success response with items, flagged, remarks
|
||||
return NextResponse.json({
|
||||
errorCode: 0,
|
||||
errorMsg: "Success",
|
||||
result: pipelineResult,
|
||||
items: docMetadata.items,
|
||||
flagged: {},
|
||||
remarks: {}
|
||||
});
|
||||
} catch (dbErr) {
|
||||
console.error("Database save failed during parse (falling back):", dbErr);
|
||||
}
|
||||
|
||||
return NextResponse.json(data);
|
||||
} catch (error: unknown) {
|
||||
console.error("Error in parse API route:", error);
|
||||
try {
|
||||
await query(
|
||||
"UPDATE documents SET parsed = true, metadata = $1 WHERE filename = $2",
|
||||
[JSON.stringify({
|
||||
tanggal: "Not Found",
|
||||
noPO: "Not Found",
|
||||
noSO: "Not Found",
|
||||
noDO: "Not Found",
|
||||
vendorInfo: "Not Found",
|
||||
customerInfo: "Not Found",
|
||||
items: []
|
||||
}), safeFile]
|
||||
);
|
||||
} catch (dbErr) {
|
||||
console.error("Failed to mark document as parsed on route catch:", dbErr);
|
||||
}
|
||||
const message = error instanceof Error ? error.message : "Internal server error";
|
||||
return NextResponse.json({ error: message }, { status: 500 });
|
||||
}
|
||||
}
|
||||
|
||||
function getBlockAngle(points: number[][]) {
|
||||
if (!points || points.length < 2) return 0;
|
||||
const p0 = points[0];
|
||||
const p1 = points[1];
|
||||
const dx = p1[0] - p0[0];
|
||||
const dy = p1[1] - p0[1];
|
||||
let angle = Math.atan2(dy, dx) * 180 / Math.PI;
|
||||
if (angle < -45) angle = 90 + angle;
|
||||
if (angle > 45) angle = angle - 90;
|
||||
return Math.abs(angle);
|
||||
}
|
||||
|
||||
function calculateAverageTilt(data: any): number {
|
||||
const results = data?.result?.layoutParsingResults || data?.layoutParsingResults || [];
|
||||
if (results.length === 0) return 0;
|
||||
const list = results[0]?.prunedResult?.parsing_res_list || [];
|
||||
if (list.length === 0) return 0;
|
||||
const angles: number[] = [];
|
||||
for (const block of list) {
|
||||
if (block.block_polygon_points) {
|
||||
angles.push(getBlockAngle(block.block_polygon_points));
|
||||
}
|
||||
}
|
||||
if (angles.length === 0) return 0;
|
||||
return angles.reduce((sum, a) => sum + a, 0) / angles.length;
|
||||
}
|
||||
@@ -0,0 +1,49 @@
|
||||
import { NextRequest, NextResponse } from "next/server";
|
||||
import fs from "fs";
|
||||
import path from "path";
|
||||
|
||||
export const dynamic = "force-dynamic";
|
||||
|
||||
export async function GET(req: NextRequest) {
|
||||
try {
|
||||
const pfmDir = path.join(process.cwd(), "public", "produk-pfm", "foto-kemasan-v2");
|
||||
if (!fs.existsSync(pfmDir)) {
|
||||
return NextResponse.json({ products: [] });
|
||||
}
|
||||
|
||||
const entries = fs.readdirSync(pfmDir, { withFileTypes: true });
|
||||
const products = [];
|
||||
|
||||
const ignoredNames = ["models", "runs", "yolo_dataset", ".venv", ".venv-api"];
|
||||
|
||||
for (const entry of entries) {
|
||||
if (entry.isDirectory() && !ignoredNames.includes(entry.name)) {
|
||||
const productDirPath = path.join(pfmDir, entry.name);
|
||||
const files = fs.readdirSync(productDirPath);
|
||||
|
||||
// Filter image files
|
||||
const imageExtensions = [".jpg", ".jpeg", ".png", ".webp", ".bmp"];
|
||||
const images = files.filter(f =>
|
||||
imageExtensions.includes(path.extname(f).toLowerCase())
|
||||
);
|
||||
|
||||
if (images.length > 0) {
|
||||
products.push({
|
||||
productName: entry.name,
|
||||
images: images.map(img => `/produk-pfm/foto-kemasan-v2/${entry.name}/${img}`),
|
||||
thumbs: images.map(img => `/produk-pfm/thumbs/${entry.name}/${img}`)
|
||||
});
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Sort products by name
|
||||
products.sort((a, b) => a.productName.localeCompare(b.productName));
|
||||
|
||||
return NextResponse.json({ products });
|
||||
} catch (error: unknown) {
|
||||
console.error("Error fetching produk PFM:", error);
|
||||
const message = error instanceof Error ? error.message : "Internal server error";
|
||||
return NextResponse.json({ error: message }, { status: 500 });
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,55 @@
|
||||
import { NextRequest, NextResponse } from "next/server";
|
||||
import fs from "fs";
|
||||
import path from "path";
|
||||
import { query } from "../../../db";
|
||||
import crypto from "crypto";
|
||||
|
||||
const UPLOADS_DIR = "/uploads";
|
||||
const PUBLIC_DIR = path.join(process.cwd(), "public/do-pfm");
|
||||
|
||||
export async function POST(req: NextRequest) {
|
||||
try {
|
||||
const { filename, image } = await req.json();
|
||||
|
||||
if (!filename || !image) {
|
||||
return NextResponse.json({ error: "Filename and image base64 data are required" }, { status: 400 });
|
||||
}
|
||||
|
||||
const safeFile = path.basename(filename);
|
||||
|
||||
const isSample = fs.existsSync(path.join(PUBLIC_DIR, safeFile));
|
||||
const filePath = isSample
|
||||
? path.join(PUBLIC_DIR, safeFile)
|
||||
: path.join(UPLOADS_DIR, safeFile);
|
||||
|
||||
const base64Data = image.replace(/^data:image\/\w+;base64,/, "");
|
||||
const buffer = Buffer.from(base64Data, "base64");
|
||||
|
||||
// Write file to disk
|
||||
fs.writeFileSync(filePath, buffer);
|
||||
console.log(`Rotated file saved successfully at ${filePath}`);
|
||||
|
||||
// Update database fields
|
||||
const fileHash = crypto.createHash("sha256").update(buffer).digest("hex");
|
||||
const stats = fs.statSync(filePath);
|
||||
|
||||
// Update document to unparsed state since layout changes
|
||||
await query(
|
||||
"UPDATE documents SET size = $1, file_hash = $2, parsed = false, layout_parsing_result = NULL WHERE filename = $3",
|
||||
[stats.size, fileHash, filename]
|
||||
);
|
||||
|
||||
// Clear old items for this document
|
||||
const docRes = await query("SELECT id FROM documents WHERE filename = $1", [filename]);
|
||||
if (docRes.rowCount && docRes.rowCount > 0) {
|
||||
const docId = docRes.rows[0].id;
|
||||
await query("DELETE FROM ocr_items WHERE document_id = $1", [docId]);
|
||||
}
|
||||
|
||||
return NextResponse.json({ success: true });
|
||||
} catch (error: unknown) {
|
||||
console.error("Error rotating file:", error);
|
||||
const message = error instanceof Error ? error.message : "Internal server error";
|
||||
return NextResponse.json({ error: message }, { status: 500 });
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,136 @@
|
||||
import { NextRequest, NextResponse } from "next/server";
|
||||
import { query } from "../../../db";
|
||||
|
||||
export const dynamic = "force-dynamic";
|
||||
|
||||
function levenshteinDistance(s1: string, s2: string): number {
|
||||
const len1 = s1.length;
|
||||
const len2 = s2.length;
|
||||
const matrix = Array.from({ length: len1 + 1 }, () => new Array(len2 + 1).fill(0));
|
||||
|
||||
for (let i = 0; i <= len1; i++) matrix[i][0] = i;
|
||||
for (let j = 0; j <= len2; j++) matrix[0][j] = j;
|
||||
|
||||
for (let i = 1; i <= len1; i++) {
|
||||
for (let j = 1; j <= len2; j++) {
|
||||
const cost = s1[i - 1] === s2[j - 1] ? 0 : 1;
|
||||
matrix[i][j] = Math.min(
|
||||
matrix[i - 1][j] + 1, // deletion
|
||||
matrix[i][j - 1] + 1, // insertion
|
||||
matrix[i - 1][j - 1] + cost // substitution
|
||||
);
|
||||
}
|
||||
}
|
||||
return matrix[len1][len2];
|
||||
}
|
||||
|
||||
function getStringSimilarity(s1: string, s2: string): number {
|
||||
const clean1 = s1.toLowerCase().replace(/[^a-z0-9]/g, '');
|
||||
const clean2 = s2.toLowerCase().replace(/[^a-z0-9]/g, '');
|
||||
if (!clean1 || !clean2) return 0;
|
||||
const distance = levenshteinDistance(clean1, clean2);
|
||||
const maxLength = Math.max(clean1.length, clean2.length);
|
||||
return (maxLength - distance) / maxLength;
|
||||
}
|
||||
|
||||
export async function POST(req: NextRequest) {
|
||||
try {
|
||||
const { image_base64 } = await req.json();
|
||||
if (!image_base64) {
|
||||
return NextResponse.json({ error: "Image is required" }, { status: 400 });
|
||||
}
|
||||
|
||||
// Call Python FastAPI server inside the container
|
||||
const pyServerUrl = process.env.CLASSIFIER_SERVER_URL || "http://paddleocr-pipeline-api:8120/classify-ocr";
|
||||
|
||||
console.log(`Forwarding scan request to classifier server: ${pyServerUrl}`);
|
||||
const response = await fetch(pyServerUrl, {
|
||||
method: "POST",
|
||||
headers: {
|
||||
"Content-Type": "application/json"
|
||||
},
|
||||
body: JSON.stringify({ image_base64 })
|
||||
});
|
||||
|
||||
if (!response.ok) {
|
||||
const errText = await response.text();
|
||||
return NextResponse.json({ error: `Classifier service error: ${errText}` }, { status: response.status });
|
||||
}
|
||||
|
||||
const data = await response.json();
|
||||
|
||||
// Layout-parsing visualization (same pipeline as DO-PFM Visual Grid)
|
||||
let layoutParsingResult: { layoutParsingResults?: Array<{ outputImages?: Record<string, string> }> } | null = null;
|
||||
const rawB64 = image_base64.includes(",") ? image_base64.split(",")[1] : image_base64;
|
||||
const pipelineUrl = process.env.PIPELINE_URL || "http://localhost:7871/layout-parsing";
|
||||
|
||||
try {
|
||||
const layoutResponse = await fetch(pipelineUrl, {
|
||||
method: "POST",
|
||||
headers: { "Content-Type": "application/json" },
|
||||
body: JSON.stringify({
|
||||
file: rawB64,
|
||||
matchHistoryJob: false,
|
||||
useLayoutDetection: true,
|
||||
fileType: 1,
|
||||
useDocUnwarping: false,
|
||||
useDocOrientationClassify: false
|
||||
})
|
||||
});
|
||||
|
||||
if (layoutResponse.ok) {
|
||||
const layoutData = await layoutResponse.json();
|
||||
layoutParsingResult = layoutData.result ?? layoutData;
|
||||
} else {
|
||||
console.warn("Layout parsing for visualization failed:", await layoutResponse.text());
|
||||
}
|
||||
} catch (layoutErr) {
|
||||
console.warn("Layout parsing for visualization unavailable:", layoutErr);
|
||||
}
|
||||
|
||||
// Now query the SKU master from database
|
||||
const dbRes = await query("SELECT no_sku, nama_item FROM sku_master");
|
||||
const skuMasterList = dbRes.rows.map(row => ({
|
||||
no_sku: row.no_sku,
|
||||
nama_item: row.nama_item
|
||||
}));
|
||||
|
||||
// Find matches
|
||||
const top1Name = data.classification?.top1_name || "";
|
||||
const extractedSku = data.ocr?.extracted_sku || "";
|
||||
const extractedProductName = data.ocr?.extracted_product_name || "";
|
||||
|
||||
const matchedList = skuMasterList.map(sku => {
|
||||
const yoloSim = top1Name ? getStringSimilarity(sku.nama_item, top1Name) : 0;
|
||||
|
||||
return {
|
||||
no_sku: sku.no_sku,
|
||||
nama_item: sku.nama_item,
|
||||
score: yoloSim,
|
||||
yoloSimilarity: yoloSim,
|
||||
isBestMatch: false
|
||||
};
|
||||
});
|
||||
|
||||
// Sort by score descending
|
||||
matchedList.sort((a, b) => b.score - a.score);
|
||||
|
||||
// Take top 5 possible matches
|
||||
const possibleMatches = matchedList.slice(0, 5).filter(m => m.score > 0.1);
|
||||
if (possibleMatches.length > 0) {
|
||||
possibleMatches[0].isBestMatch = true;
|
||||
}
|
||||
|
||||
return NextResponse.json({
|
||||
classification: data.classification,
|
||||
ocr: data.ocr,
|
||||
possibleMatches,
|
||||
layoutParsingResult
|
||||
});
|
||||
|
||||
} catch (error: unknown) {
|
||||
console.error("Error in scan-pfm API route:", error);
|
||||
const message = error instanceof Error ? error.message : "Internal server error";
|
||||
return NextResponse.json({ error: message }, { status: 500 });
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,23 @@
|
||||
import { NextRequest, NextResponse } from "next/server";
|
||||
import { query } from "../../../db";
|
||||
|
||||
export const dynamic = "force-dynamic";
|
||||
|
||||
export async function GET(req: NextRequest) {
|
||||
try {
|
||||
const res = await query(
|
||||
"SELECT no_sku, nama_item FROM sku_master ORDER BY no_sku"
|
||||
);
|
||||
|
||||
const skus = res.rows.map(row => ({
|
||||
no_sku: row.no_sku,
|
||||
nama_item: row.nama_item
|
||||
}));
|
||||
|
||||
return NextResponse.json({ skus });
|
||||
} catch (error: unknown) {
|
||||
console.error("Error in SKUs API route:", error);
|
||||
const message = error instanceof Error ? error.message : "Internal server error";
|
||||
return NextResponse.json({ error: message }, { status: 500 });
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,71 @@
|
||||
import { NextRequest, NextResponse } from "next/server";
|
||||
import { query } from "../../../db";
|
||||
import path from "path";
|
||||
|
||||
export async function POST(req: NextRequest) {
|
||||
try {
|
||||
const body = await req.json();
|
||||
const { page, rowIndex, action } = body;
|
||||
|
||||
if (!page || rowIndex === undefined || !action) {
|
||||
return NextResponse.json({ error: "Missing required fields" }, { status: 400 });
|
||||
}
|
||||
|
||||
const safeFile = path.basename(page);
|
||||
|
||||
// Get document ID
|
||||
const docRes = await query("SELECT id FROM documents WHERE filename = $1", [safeFile]);
|
||||
if (!docRes.rowCount || docRes.rowCount === 0) {
|
||||
return NextResponse.json({ error: "Document not found in database" }, { status: 404 });
|
||||
}
|
||||
const docId = docRes.rows[0].id;
|
||||
|
||||
if (action === "edit") {
|
||||
const { field, value } = body;
|
||||
if (!field || value === undefined) {
|
||||
return NextResponse.json({ error: "Missing edit parameters" }, { status: 400 });
|
||||
}
|
||||
|
||||
// Map UI field names to database columns
|
||||
let colName = "";
|
||||
if (field === "kodeBarang") {
|
||||
colName = "kode_barang";
|
||||
} else if (field === "banyak") {
|
||||
colName = "banyak";
|
||||
} else if (field === "jumlah") {
|
||||
colName = "jumlah";
|
||||
} else {
|
||||
return NextResponse.json({ error: "Invalid field name" }, { status: 400 });
|
||||
}
|
||||
|
||||
await query(
|
||||
`UPDATE ocr_items
|
||||
SET ${colName} = $1
|
||||
WHERE document_id = $2 AND row_index = $3`,
|
||||
[value, docId, rowIndex]
|
||||
);
|
||||
|
||||
return NextResponse.json({ success: true });
|
||||
} else if (action === "flag") {
|
||||
const { isFlagged, remark } = body;
|
||||
if (isFlagged === undefined || remark === undefined) {
|
||||
return NextResponse.json({ error: "Missing flag parameters" }, { status: 400 });
|
||||
}
|
||||
|
||||
await query(
|
||||
`UPDATE ocr_items
|
||||
SET is_flagged = $1, remark = $2
|
||||
WHERE document_id = $3 AND row_index = $4`,
|
||||
[!!isFlagged, remark, docId, rowIndex]
|
||||
);
|
||||
|
||||
return NextResponse.json({ success: true });
|
||||
} else {
|
||||
return NextResponse.json({ error: "Invalid action" }, { status: 400 });
|
||||
}
|
||||
} catch (error: unknown) {
|
||||
console.error("Error in update-row API route:", error);
|
||||
const message = error instanceof Error ? error.message : "Internal server error";
|
||||
return NextResponse.json({ error: message }, { status: 500 });
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,219 @@
|
||||
import { NextRequest, NextResponse } from "next/server";
|
||||
import fs from "fs";
|
||||
import path from "path";
|
||||
import crypto from "crypto";
|
||||
import { query, resolveStoreFromText } from "../../../db";
|
||||
import { parseDOMetadata } from "../../../utils/parser";
|
||||
|
||||
const UPLOADS_DIR = "/uploads";
|
||||
|
||||
export async function POST(req: NextRequest) {
|
||||
try {
|
||||
// Ensure uploads directory exists
|
||||
if (!fs.existsSync(UPLOADS_DIR)) {
|
||||
fs.mkdirSync(UPLOADS_DIR, { recursive: true });
|
||||
}
|
||||
|
||||
const formData = await req.formData();
|
||||
const file = formData.get("file") as Blob | null;
|
||||
|
||||
if (!file) {
|
||||
return NextResponse.json({ error: "No file uploaded" }, { status: 400 });
|
||||
}
|
||||
|
||||
const originalName = file instanceof File ? file.name : "document.jpg";
|
||||
// Sanitize filename to avoid directory traversal
|
||||
const safeName = path.basename(originalName).replace(/\s+/g, "_");
|
||||
const filename = `${Date.now()}-${safeName}`;
|
||||
const filePath = path.join(UPLOADS_DIR, filename);
|
||||
|
||||
// Save file
|
||||
const arrayBuffer = await file.arrayBuffer();
|
||||
const buffer = Buffer.from(arrayBuffer);
|
||||
|
||||
// Compute hash to check for duplicate content
|
||||
const fileHash = crypto.createHash("sha256").update(buffer).digest("hex");
|
||||
|
||||
// Check if same content already exists in database
|
||||
const dupRes = await query(
|
||||
"SELECT filename, layout_parsing_result FROM documents WHERE file_hash = $1",
|
||||
[fileHash]
|
||||
);
|
||||
|
||||
if (dupRes.rowCount && dupRes.rowCount > 0) {
|
||||
const existingDoc = dupRes.rows[0];
|
||||
console.log(`Uploaded file matches existing database record (file_hash: ${fileHash}). Reusing existing file: ${existingDoc.filename}`);
|
||||
|
||||
// Reconstruct full response wrapping to match fresh API response
|
||||
const wrappedResult = {
|
||||
errorCode: 0,
|
||||
errorMsg: "Success",
|
||||
result: existingDoc.layout_parsing_result
|
||||
};
|
||||
|
||||
return NextResponse.json({
|
||||
filename: existingDoc.filename,
|
||||
result: wrappedResult,
|
||||
alreadyExists: true
|
||||
});
|
||||
}
|
||||
|
||||
fs.writeFileSync(filePath, buffer);
|
||||
|
||||
// Convert to base64 for pipeline API
|
||||
const b64 = buffer.toString("base64");
|
||||
|
||||
// Form payload
|
||||
const payload = {
|
||||
file: b64,
|
||||
matchHistoryJob: false,
|
||||
useLayoutDetection: true,
|
||||
fileType: 1,
|
||||
useDocUnwarping: false,
|
||||
useDocOrientationClassify: false
|
||||
};
|
||||
|
||||
const pipelineUrl = process.env.PIPELINE_URL || "http://paddleocr-pipeline-api:8090/layout-parsing";
|
||||
console.log(`Forwarding uploaded file ${filename} to pipeline: ${pipelineUrl}`);
|
||||
|
||||
const response = await fetch(pipelineUrl, {
|
||||
method: "POST",
|
||||
headers: {
|
||||
"Content-Type": "application/json"
|
||||
},
|
||||
body: JSON.stringify(payload)
|
||||
});
|
||||
|
||||
if (!response.ok) {
|
||||
const errText = await response.text();
|
||||
return NextResponse.json({ error: `Pipeline API error: ${errText}` }, { status: response.status });
|
||||
}
|
||||
|
||||
let data = await response.json();
|
||||
|
||||
// Check if the image is not straight (tilt > 1.0 degree)
|
||||
const tilt = calculateAverageTilt(data);
|
||||
if (tilt > 1.0) {
|
||||
console.log(`Uploaded document ${filename} is not straight (average tilt: ${tilt.toFixed(2)} deg). Re-running with unwarping and orientation classification enabled...`);
|
||||
const unwarpPayload = {
|
||||
...payload,
|
||||
useDocUnwarping: true,
|
||||
useDocOrientationClassify: true
|
||||
};
|
||||
const unwarpResponse = await fetch(pipelineUrl, {
|
||||
method: "POST",
|
||||
headers: {
|
||||
"Content-Type": "application/json"
|
||||
},
|
||||
body: JSON.stringify(unwarpPayload)
|
||||
});
|
||||
if (unwarpResponse.ok) {
|
||||
data = await unwarpResponse.json();
|
||||
console.log(`Document unwarped successfully.`);
|
||||
} else {
|
||||
console.error(`Unwarping failed with status ${unwarpResponse.status}`);
|
||||
}
|
||||
}
|
||||
|
||||
// Save JSON extraction result
|
||||
const jsonPath = `${filePath}.json`;
|
||||
fs.writeFileSync(jsonPath, JSON.stringify(data, null, 2));
|
||||
|
||||
// Save to PostgreSQL database
|
||||
try {
|
||||
const pipelineResult = data.result || data;
|
||||
const page0 = pipelineResult?.layoutParsingResults?.[0] || {};
|
||||
const markdownText = page0?.markdown?.text || "";
|
||||
const docMetadata = parseDOMetadata(markdownText);
|
||||
|
||||
// Resolve store information using master database
|
||||
const resolvedStore = await resolveStoreFromText(markdownText);
|
||||
(docMetadata as any).orderUntuk = resolvedStore.orderUntuk;
|
||||
(docMetadata as any).alamat = resolvedStore.alamat;
|
||||
|
||||
const insertDocRes = await query(`
|
||||
INSERT INTO documents (filename, upload_time, size, parsed, metadata, layout_parsing_result, is_sample, file_hash)
|
||||
VALUES ($1, $2, $3, $4, $5, $6, $7, $8)
|
||||
RETURNING id
|
||||
`, [
|
||||
filename,
|
||||
new Date(),
|
||||
buffer.length,
|
||||
true,
|
||||
JSON.stringify(docMetadata),
|
||||
JSON.stringify(pipelineResult),
|
||||
false,
|
||||
fileHash
|
||||
]);
|
||||
|
||||
const docId = insertDocRes.rows[0].id;
|
||||
|
||||
for (let i = 0; i < docMetadata.items.length; i++) {
|
||||
const item = docMetadata.items[i];
|
||||
await query(`
|
||||
INSERT INTO ocr_items (
|
||||
document_id, row_index,
|
||||
kode_barang_original, kode_barang,
|
||||
nama_barang,
|
||||
banyak_original, banyak,
|
||||
jumlah_original, jumlah,
|
||||
is_flagged, remark
|
||||
)
|
||||
VALUES ($1, $2, $3, $3, $4, $5, $5, $6, $6, false, '')
|
||||
ON CONFLICT DO NOTHING
|
||||
`, [
|
||||
docId,
|
||||
i,
|
||||
item.kodeBarang,
|
||||
item.namaBarang,
|
||||
item.banyak,
|
||||
item.jumlah
|
||||
]);
|
||||
}
|
||||
} catch (dbErr) {
|
||||
console.error("Database save failed during upload (falling back to file):", dbErr);
|
||||
}
|
||||
|
||||
const wrappedResult = {
|
||||
errorCode: 0,
|
||||
errorMsg: "Success",
|
||||
result: data.result || data
|
||||
};
|
||||
|
||||
return NextResponse.json({
|
||||
filename,
|
||||
result: wrappedResult
|
||||
});
|
||||
} catch (error: unknown) {
|
||||
console.error("Error in upload API route:", error);
|
||||
const message = error instanceof Error ? error.message : "Internal server error";
|
||||
return NextResponse.json({ error: message }, { status: 500 });
|
||||
}
|
||||
}
|
||||
|
||||
function getBlockAngle(points: number[][]) {
|
||||
if (!points || points.length < 2) return 0;
|
||||
const p0 = points[0];
|
||||
const p1 = points[1];
|
||||
const dx = p1[0] - p0[0];
|
||||
const dy = p1[1] - p0[1];
|
||||
let angle = Math.atan2(dy, dx) * 180 / Math.PI;
|
||||
if (angle < -45) angle = 90 + angle;
|
||||
if (angle > 45) angle = angle - 90;
|
||||
return Math.abs(angle);
|
||||
}
|
||||
|
||||
function calculateAverageTilt(data: any): number {
|
||||
const results = data?.result?.layoutParsingResults || data?.layoutParsingResults || [];
|
||||
if (results.length === 0) return 0;
|
||||
const list = results[0]?.prunedResult?.parsing_res_list || [];
|
||||
if (list.length === 0) return 0;
|
||||
const angles: number[] = [];
|
||||
for (const block of list) {
|
||||
if (block.block_polygon_points) {
|
||||
angles.push(getBlockAngle(block.block_polygon_points));
|
||||
}
|
||||
}
|
||||
if (angles.length === 0) return 0;
|
||||
return angles.reduce((sum, a) => sum + a, 0) / angles.length;
|
||||
}
|
||||
@@ -0,0 +1,38 @@
|
||||
import { NextRequest, NextResponse } from "next/server";
|
||||
|
||||
const corsHeaders = {
|
||||
"Access-Control-Allow-Origin": "*",
|
||||
"Access-Control-Allow-Methods": "GET, POST, PUT, DELETE, OPTIONS",
|
||||
"Access-Control-Allow-Headers": "Content-Type, Authorization"
|
||||
};
|
||||
|
||||
export async function OPTIONS() {
|
||||
return new NextResponse(null, { status: 204, headers: corsHeaders });
|
||||
}
|
||||
|
||||
export async function POST(req: NextRequest) {
|
||||
try {
|
||||
const body = await req.json();
|
||||
const { username, password } = body;
|
||||
|
||||
// Simple authentication logic for demo
|
||||
if (username === "admin" && password === "password") {
|
||||
return NextResponse.json({
|
||||
status: "success",
|
||||
message: "Login successful",
|
||||
data: {
|
||||
token: "demo-auth-token-12345"
|
||||
}
|
||||
}, { headers: corsHeaders });
|
||||
}
|
||||
|
||||
return NextResponse.json({
|
||||
status: "error",
|
||||
message: "Invalid username or password"
|
||||
}, { status: 401, headers: corsHeaders });
|
||||
} catch (error: unknown) {
|
||||
console.error("Error in login API route:", error);
|
||||
const message = error instanceof Error ? error.message : "Internal server error";
|
||||
return NextResponse.json({ error: message }, { status: 500, headers: corsHeaders });
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,155 @@
|
||||
import { NextRequest, NextResponse } from "next/server";
|
||||
import { query } from "../../../../../db";
|
||||
|
||||
const corsHeaders = {
|
||||
"Access-Control-Allow-Origin": "*",
|
||||
"Access-Control-Allow-Methods": "GET, POST, PUT, DELETE, OPTIONS",
|
||||
"Access-Control-Allow-Headers": "Content-Type, Authorization"
|
||||
};
|
||||
|
||||
export async function OPTIONS() {
|
||||
return new NextResponse(null, { status: 204, headers: corsHeaders });
|
||||
}
|
||||
|
||||
export async function PUT(
|
||||
req: NextRequest,
|
||||
context: { params: Promise<{ id: string }> }
|
||||
) {
|
||||
try {
|
||||
const params = await context.params;
|
||||
const { id } = params;
|
||||
const docId = parseInt(id);
|
||||
|
||||
if (isNaN(docId)) {
|
||||
return NextResponse.json({ error: "Invalid document ID" }, { status: 400, headers: corsHeaders });
|
||||
}
|
||||
|
||||
// Check if document exists
|
||||
const checkRes = await query("SELECT id, filename, upload_time FROM documents WHERE id = $1", [docId]);
|
||||
if (!checkRes.rowCount || checkRes.rowCount === 0) {
|
||||
return NextResponse.json({ error: "Document not found" }, { status: 404, headers: corsHeaders });
|
||||
}
|
||||
|
||||
const doc = checkRes.rows[0];
|
||||
|
||||
const body = await req.json();
|
||||
const {
|
||||
tanggal,
|
||||
noPo,
|
||||
noSo,
|
||||
noDo,
|
||||
kepadaYth,
|
||||
orderUntuk,
|
||||
alamat,
|
||||
platTruk,
|
||||
namaDriver,
|
||||
namaPenerima,
|
||||
latitude,
|
||||
longitude,
|
||||
items = []
|
||||
} = body;
|
||||
|
||||
// Structuring metadata JSONB to store both formats for full compatibility
|
||||
const metadata = {
|
||||
// Legacy Next.js web parser format
|
||||
tanggal: tanggal || "",
|
||||
noPO: noPo || "",
|
||||
noSO: noSo || "",
|
||||
noDO: noDo || doc.filename || "",
|
||||
customerInfo: kepadaYth || "",
|
||||
headerRemark: namaPenerima || "",
|
||||
|
||||
// Mobile native app format
|
||||
header: {
|
||||
tanggal: tanggal || "",
|
||||
no_po: noPo || "",
|
||||
no_so: noSo || "",
|
||||
no_do: noDo || ""
|
||||
},
|
||||
shipment: {
|
||||
kepada_yth: kepadaYth || "",
|
||||
order_untuk: orderUntuk || "",
|
||||
alamat: alamat || "",
|
||||
plat_truk: platTruk || "",
|
||||
nama_driver: namaDriver || "",
|
||||
nama_penerima: namaPenerima || ""
|
||||
}
|
||||
};
|
||||
|
||||
const latFloat = latitude ? parseFloat(latitude.toString()) : null;
|
||||
const lngFloat = longitude ? parseFloat(longitude.toString()) : null;
|
||||
|
||||
// Update document record
|
||||
await query(`
|
||||
UPDATE documents
|
||||
SET parsed = true,
|
||||
latitude = $2,
|
||||
longitude = $3,
|
||||
metadata = $4
|
||||
WHERE id = $1
|
||||
`, [docId, latFloat, lngFloat, JSON.stringify(metadata)]);
|
||||
|
||||
// Delete existing ocr_items to recreate them
|
||||
await query("DELETE FROM ocr_items WHERE document_id = $1", [docId]);
|
||||
|
||||
// Insert new items
|
||||
for (let i = 0; i < items.length; i++) {
|
||||
const item = items[i];
|
||||
const nomorSku = item.nomor_sku || item.nomorSku || "";
|
||||
const namaBarang = item.nama_barang || item.namaBarang || "";
|
||||
const banyak = item.banyak || "";
|
||||
const jumlah = item.jumlah || "";
|
||||
|
||||
await query(`
|
||||
INSERT INTO ocr_items (
|
||||
document_id, row_index,
|
||||
kode_barang_original, kode_barang,
|
||||
nama_barang,
|
||||
banyak_original, banyak,
|
||||
jumlah_original, jumlah,
|
||||
is_flagged, remark
|
||||
) VALUES ($1, $2, $3, $3, $4, $5, $5, $6, $6, false, '')
|
||||
`, [docId, i, nomorSku, namaBarang, banyak, jumlah]);
|
||||
}
|
||||
|
||||
// Return the updated document mapping
|
||||
const mappedData = {
|
||||
id: docId.toString(),
|
||||
filePath: doc.filename,
|
||||
createdAt: doc.upload_time.toISOString(),
|
||||
header: {
|
||||
tanggal: tanggal || "",
|
||||
no_po: noPo || "",
|
||||
no_so: noSo || "",
|
||||
no_do: noDo || ""
|
||||
},
|
||||
shipment: {
|
||||
kepada_yth: kepadaYth || "",
|
||||
order_untuk: orderUntuk || "",
|
||||
alamat: alamat || "",
|
||||
plat_truk: platTruk || "",
|
||||
nama_driver: namaDriver || "",
|
||||
nama_penerima: namaPenerima || ""
|
||||
},
|
||||
items: items.map((item: any) => ({
|
||||
nomor_sku: item.nomor_sku || item.nomorSku || "",
|
||||
nama_barang: item.nama_barang || item.namaBarang || "",
|
||||
banyak: item.banyak || "",
|
||||
jumlah: item.jumlah || ""
|
||||
})),
|
||||
latitude: latFloat,
|
||||
longitude: lngFloat
|
||||
};
|
||||
|
||||
return NextResponse.json({
|
||||
status: "success",
|
||||
message: "Document updated successfully",
|
||||
data: mappedData
|
||||
}, { headers: corsHeaders });
|
||||
|
||||
} catch (error: unknown) {
|
||||
console.error("Error in update document API v1 route:", error);
|
||||
const message = error instanceof Error ? error.message : "Internal server error";
|
||||
return NextResponse.json({ error: message }, { status: 500, headers: corsHeaders });
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,119 @@
|
||||
import { NextRequest, NextResponse } from "next/server";
|
||||
import { query } from "../../../../db";
|
||||
|
||||
const corsHeaders = {
|
||||
"Access-Control-Allow-Origin": "*",
|
||||
"Access-Control-Allow-Methods": "GET, POST, PUT, DELETE, OPTIONS",
|
||||
"Access-Control-Allow-Headers": "Content-Type, Authorization"
|
||||
};
|
||||
|
||||
export async function OPTIONS() {
|
||||
return new NextResponse(null, { status: 204, headers: corsHeaders });
|
||||
}
|
||||
|
||||
export async function GET(req: NextRequest) {
|
||||
try {
|
||||
// Retrieve all custom-uploaded documents
|
||||
const docRes = await query(`
|
||||
SELECT id, filename, upload_time, size, parsed, is_sample, metadata, latitude, longitude
|
||||
FROM documents
|
||||
WHERE is_sample = false AND parsed = true
|
||||
ORDER BY upload_time DESC
|
||||
`);
|
||||
|
||||
const documents = docRes.rows;
|
||||
const mappedList = [];
|
||||
|
||||
for (const doc of documents) {
|
||||
const docId = doc.id;
|
||||
const metadata = doc.metadata || {};
|
||||
|
||||
// Retrieve items from ocr_items
|
||||
const itemsRes = await query(`
|
||||
SELECT row_index, kode_barang, nama_barang, banyak, jumlah
|
||||
FROM ocr_items
|
||||
WHERE document_id = $1
|
||||
ORDER BY row_index
|
||||
`, [docId]);
|
||||
|
||||
const items = itemsRes.rows.map(item => ({
|
||||
nomor_sku: item.kode_barang || "",
|
||||
nama_barang: item.nama_barang || "",
|
||||
banyak: item.banyak || "",
|
||||
jumlah: item.jumlah || ""
|
||||
}));
|
||||
|
||||
// Determine header and shipment mapping
|
||||
let header = {
|
||||
tanggal: "",
|
||||
no_po: "",
|
||||
no_so: "",
|
||||
no_do: ""
|
||||
};
|
||||
|
||||
let shipment = {
|
||||
kepada_yth: "",
|
||||
order_untuk: "",
|
||||
alamat: "",
|
||||
plat_truk: "",
|
||||
nama_driver: "",
|
||||
nama_penerima: ""
|
||||
};
|
||||
|
||||
if (metadata.header) {
|
||||
// Document was updated via mobile app
|
||||
header = {
|
||||
tanggal: metadata.header.tanggal || "",
|
||||
no_po: metadata.header.no_po || "",
|
||||
no_so: metadata.header.no_so || "",
|
||||
no_do: metadata.header.no_do || ""
|
||||
};
|
||||
shipment = {
|
||||
kepada_yth: metadata.shipment?.kepada_yth || "",
|
||||
order_untuk: metadata.shipment?.order_untuk || "",
|
||||
alamat: metadata.shipment?.alamat || "",
|
||||
plat_truk: metadata.shipment?.plat_truk || "",
|
||||
nama_driver: metadata.shipment?.nama_driver || "",
|
||||
nama_penerima: metadata.shipment?.nama_penerima || ""
|
||||
};
|
||||
} else {
|
||||
// Document was freshly uploaded / parsed via web
|
||||
header = {
|
||||
tanggal: metadata.tanggal || "",
|
||||
no_po: metadata.noPO || "",
|
||||
no_so: metadata.noSO || "",
|
||||
no_do: metadata.noDO || doc.filename || ""
|
||||
};
|
||||
shipment = {
|
||||
kepada_yth: metadata.customerInfo || "",
|
||||
order_untuk: metadata.orderUntuk || "",
|
||||
alamat: metadata.alamat || "",
|
||||
plat_truk: metadata.platTruk || "",
|
||||
nama_driver: "",
|
||||
nama_penerima: metadata.headerRemark || ""
|
||||
};
|
||||
}
|
||||
|
||||
mappedList.push({
|
||||
id: docId.toString(),
|
||||
filePath: doc.filename,
|
||||
createdAt: doc.upload_time.toISOString(),
|
||||
header,
|
||||
shipment,
|
||||
items,
|
||||
latitude: doc.latitude ? parseFloat(doc.latitude.toString()) : null,
|
||||
longitude: doc.longitude ? parseFloat(doc.longitude.toString()) : null
|
||||
});
|
||||
}
|
||||
|
||||
return NextResponse.json({
|
||||
status: "success",
|
||||
data: mappedList
|
||||
}, { headers: corsHeaders });
|
||||
|
||||
} catch (error: unknown) {
|
||||
console.error("Error in list documents API v1 route:", error);
|
||||
const message = error instanceof Error ? error.message : "Internal server error";
|
||||
return NextResponse.json({ error: message }, { status: 500, headers: corsHeaders });
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,134 @@
|
||||
import { NextRequest, NextResponse } from "next/server";
|
||||
import fs from "fs";
|
||||
import path from "path";
|
||||
import crypto from "crypto";
|
||||
import { query } from "../../../../../db";
|
||||
|
||||
const UPLOADS_DIR = "/uploads";
|
||||
|
||||
const corsHeaders = {
|
||||
"Access-Control-Allow-Origin": "*",
|
||||
"Access-Control-Allow-Methods": "GET, POST, PUT, DELETE, OPTIONS",
|
||||
"Access-Control-Allow-Headers": "Content-Type, Authorization"
|
||||
};
|
||||
|
||||
export async function OPTIONS() {
|
||||
return new NextResponse(null, { status: 204, headers: corsHeaders });
|
||||
}
|
||||
|
||||
export async function POST(req: NextRequest) {
|
||||
try {
|
||||
// Ensure uploads directory exists
|
||||
if (!fs.existsSync(UPLOADS_DIR)) {
|
||||
fs.mkdirSync(UPLOADS_DIR, { recursive: true });
|
||||
}
|
||||
|
||||
const formData = await req.formData();
|
||||
const file = (formData.get("image") || formData.get("file")) as Blob | null;
|
||||
|
||||
if (!file) {
|
||||
return NextResponse.json({ error: "No file uploaded" }, { status: 400, headers: corsHeaders });
|
||||
}
|
||||
|
||||
const originalName = file instanceof File ? file.name : "document.jpg";
|
||||
const safeName = path.basename(originalName).replace(/\s+/g, "_");
|
||||
const filename = `${Date.now()}-${safeName}`;
|
||||
const filePath = path.join(UPLOADS_DIR, filename);
|
||||
|
||||
// Save file
|
||||
const arrayBuffer = await file.arrayBuffer();
|
||||
const buffer = Buffer.from(arrayBuffer);
|
||||
fs.writeFileSync(filePath, buffer);
|
||||
|
||||
// Compute hash
|
||||
const fileHash = crypto.createHash("sha256").update(buffer).digest("hex");
|
||||
|
||||
// Geolocation tags
|
||||
const latVal = formData.get("latitude");
|
||||
const lngVal = formData.get("longitude");
|
||||
const latitude = latVal ? parseFloat(latVal.toString()) : null;
|
||||
const longitude = lngVal ? parseFloat(lngVal.toString()) : null;
|
||||
|
||||
// Check if the exact file content already exists in DB
|
||||
const dupRes = await query(
|
||||
"SELECT id, filename, latitude, longitude, metadata FROM documents WHERE file_hash = $1 AND is_sample = false",
|
||||
[fileHash]
|
||||
);
|
||||
|
||||
let docId: number;
|
||||
let finalFilename = filename;
|
||||
|
||||
if (dupRes.rowCount && dupRes.rowCount > 0) {
|
||||
const existingDoc = dupRes.rows[0];
|
||||
docId = existingDoc.id;
|
||||
finalFilename = existingDoc.filename;
|
||||
console.log(`Reusing existing document record (id: ${docId}) for hash match.`);
|
||||
|
||||
// Clean up the newly written file since we are reusing the existing one
|
||||
if (fs.existsSync(filePath) && finalFilename !== filename) {
|
||||
fs.unlinkSync(filePath);
|
||||
}
|
||||
} else {
|
||||
const insertRes = await query(`
|
||||
INSERT INTO documents (filename, upload_time, size, parsed, is_sample, file_hash, latitude, longitude)
|
||||
VALUES ($1, $2, $3, $4, $5, $6, $7, $8)
|
||||
RETURNING id
|
||||
`, [
|
||||
filename,
|
||||
new Date(),
|
||||
buffer.length,
|
||||
false,
|
||||
false,
|
||||
fileHash,
|
||||
latitude,
|
||||
longitude
|
||||
]);
|
||||
docId = insertRes.rows[0].id;
|
||||
}
|
||||
|
||||
// Trigger parsing synchronously to ensure it is processed immediately on receiving the image
|
||||
try {
|
||||
await fetch("http://127.0.0.1:3000/api/parse", {
|
||||
method: "POST",
|
||||
headers: { "Content-Type": "application/json" },
|
||||
body: JSON.stringify({ filename: finalFilename })
|
||||
});
|
||||
} catch (err) {
|
||||
console.error("Error triggering parse synchronously:", err);
|
||||
}
|
||||
|
||||
// Return the response structured as DocumentModel.fromJson format
|
||||
const mappedData = {
|
||||
id: docId.toString(),
|
||||
header: {
|
||||
tanggal: "",
|
||||
no_po: "",
|
||||
no_so: "",
|
||||
no_do: ""
|
||||
},
|
||||
shipment: {
|
||||
kepada_yth: "PT.PRIMAFOOD INTERNATIONAL",
|
||||
order_untuk: "",
|
||||
alamat: "",
|
||||
plat_truk: "",
|
||||
nama_driver: "",
|
||||
nama_penerima: ""
|
||||
},
|
||||
items: [] as any[],
|
||||
latitude: latitude,
|
||||
longitude: longitude,
|
||||
createdAt: new Date().toISOString()
|
||||
};
|
||||
|
||||
return NextResponse.json({
|
||||
status: "success",
|
||||
message: "Document uploaded successfully",
|
||||
data: mappedData
|
||||
}, { status: 201, headers: corsHeaders });
|
||||
|
||||
} catch (error: unknown) {
|
||||
console.error("Error in upload API v1 route:", error);
|
||||
const message = error instanceof Error ? error.message : "Internal server error";
|
||||
return NextResponse.json({ error: message }, { status: 500, headers: corsHeaders });
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,18 @@
|
||||
:root {
|
||||
--background: #ffffff;
|
||||
--foreground: #171717;
|
||||
}
|
||||
|
||||
@media (prefers-color-scheme: dark) {
|
||||
:root {
|
||||
--background: #0a0a0a;
|
||||
--foreground: #ededed;
|
||||
}
|
||||
}
|
||||
|
||||
body {
|
||||
background: var(--background);
|
||||
color: var(--foreground);
|
||||
font-family: system-ui, -apple-system, sans-serif;
|
||||
margin: 0;
|
||||
}
|
||||
@@ -0,0 +1,33 @@
|
||||
import type { Metadata } from "next";
|
||||
import { Geist, Geist_Mono } from "next/font/google";
|
||||
import "./globals.css";
|
||||
|
||||
const geistSans = Geist({
|
||||
variable: "--font-geist-sans",
|
||||
subsets: ["latin"],
|
||||
});
|
||||
|
||||
const geistMono = Geist_Mono({
|
||||
variable: "--font-geist-mono",
|
||||
subsets: ["latin"],
|
||||
});
|
||||
|
||||
export const metadata: Metadata = {
|
||||
title: "AI OCR Delivery Order",
|
||||
description: "Generated by create next app",
|
||||
};
|
||||
|
||||
export default function RootLayout({
|
||||
children,
|
||||
}: Readonly<{
|
||||
children: React.ReactNode;
|
||||
}>) {
|
||||
return (
|
||||
<html
|
||||
lang="en"
|
||||
className={`${geistSans.variable} ${geistMono.variable} h-full antialiased`}
|
||||
>
|
||||
<body className="min-h-full flex flex-col">{children}</body>
|
||||
</html>
|
||||
);
|
||||
}
|
||||
@@ -0,0 +1,10 @@
|
||||
export const dynamic = "force-dynamic";
|
||||
|
||||
export default function Home() {
|
||||
return (
|
||||
<div style={{ fontFamily: 'system-ui, sans-serif', padding: '2rem', textAlign: 'center' }}>
|
||||
<h1>Prima Fresh Mart OCR API Gateway</h1>
|
||||
<p>Status: Running</p>
|
||||
</div>
|
||||
);
|
||||
}
|
||||
@@ -0,0 +1,133 @@
|
||||
import { Pool } from "pg";
|
||||
import { initDb } from "./init";
|
||||
|
||||
const pool = new Pool({
|
||||
host: process.env.PGHOST || "localhost",
|
||||
port: parseInt(process.env.PGPORT || "5432"),
|
||||
user: process.env.PGUSER || "postgres",
|
||||
password: process.env.PGPASSWORD || "postgres",
|
||||
database: process.env.PGDATABASE || "dopfm",
|
||||
});
|
||||
|
||||
let initialized = false;
|
||||
let initPromise: Promise<Pool> | null = null;
|
||||
|
||||
export async function getPool(): Promise<Pool> {
|
||||
if (initialized) {
|
||||
return pool;
|
||||
}
|
||||
if (!initPromise) {
|
||||
initPromise = (async () => {
|
||||
try {
|
||||
await initDb(pool);
|
||||
initialized = true;
|
||||
} catch (err) {
|
||||
console.error("Failed to initialize database:", err);
|
||||
}
|
||||
return pool;
|
||||
})();
|
||||
}
|
||||
return initPromise;
|
||||
}
|
||||
|
||||
export async function query(text: string, params?: unknown[]) {
|
||||
const p = await getPool();
|
||||
return p.query(text, params);
|
||||
}
|
||||
|
||||
export async function cleanupAndReindexItems(docId: number) {
|
||||
// 1. Delete rows where kode_barang is blank/null or doesn't match an 8-digit number
|
||||
await query(
|
||||
`DELETE FROM ocr_items
|
||||
WHERE document_id = $1
|
||||
AND (kode_barang IS NULL OR TRIM(kode_barang) = '' OR NOT (kode_barang ~ '^[0-9]{8}$'))`,
|
||||
[docId]
|
||||
);
|
||||
|
||||
// 2. Fetch remaining rows ordered by row_index
|
||||
const res = await query(
|
||||
`SELECT id, row_index
|
||||
FROM ocr_items
|
||||
WHERE document_id = $1
|
||||
ORDER BY row_index`,
|
||||
[docId]
|
||||
);
|
||||
|
||||
// 3. Update row_index to be sequential
|
||||
for (let i = 0; i < res.rows.length; i++) {
|
||||
const row = res.rows[i];
|
||||
if (row.row_index !== i) {
|
||||
await query(
|
||||
`UPDATE ocr_items
|
||||
SET row_index = $1
|
||||
WHERE id = $2`,
|
||||
[i, row.id]
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
export async function resolveStoreFromText(custInfo: string): Promise<{ orderUntuk: string; alamat: string }> {
|
||||
if (!custInfo || custInfo === "Not Found") {
|
||||
return { orderUntuk: "", alamat: "" };
|
||||
}
|
||||
|
||||
// Load all stores
|
||||
const storeRes = await query("SELECT nama_toko, kode_toko, alamat FROM store_master");
|
||||
const stores = storeRes.rows;
|
||||
|
||||
const ocrTokens = new Set(
|
||||
custInfo.toLowerCase()
|
||||
.replace(/[^a-z0-9\s]/g, " ")
|
||||
.split(/\s+/)
|
||||
.filter(w => w.length > 2 && !["dan", "dki", "area", "yth", "kepada", "order", "untuk", "alamat", "kel", "kec", "rt", "rw"].includes(w))
|
||||
);
|
||||
|
||||
let bestStore: any = null;
|
||||
let bestScore = 0;
|
||||
let bestMatchCount = 0;
|
||||
|
||||
if (ocrTokens.size > 0) {
|
||||
for (const store of stores) {
|
||||
const searchStr = `${store.nama_toko} ${store.alamat}`.toLowerCase();
|
||||
const storeTokens = searchStr
|
||||
.replace(/[^a-z0-9\s]/g, " ")
|
||||
.split(/\s+/)
|
||||
.filter(w => w.length > 2 && !["dan", "dki", "area", "yth", "kepada", "order", "untuk", "alamat", "kel", "kec", "rt", "rw", "jalan", "raya", "blok", "nomor", "rt", "rw", "kelurahan", "kecamatan", "kota", "kabupaten", "provinsi"].includes(w));
|
||||
|
||||
if (storeTokens.length === 0) continue;
|
||||
|
||||
let matchCount = 0;
|
||||
const uniqueStoreTokens = new Set(storeTokens);
|
||||
for (const token of uniqueStoreTokens) {
|
||||
if (ocrTokens.has(token)) {
|
||||
matchCount++;
|
||||
}
|
||||
}
|
||||
|
||||
const score = matchCount / uniqueStoreTokens.size;
|
||||
if (matchCount >= 2) {
|
||||
if (matchCount > bestMatchCount || (matchCount === bestMatchCount && score > bestScore)) {
|
||||
bestMatchCount = matchCount;
|
||||
bestScore = score;
|
||||
bestStore = store;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (bestStore) {
|
||||
return { orderUntuk: bestStore.nama_toko, alamat: bestStore.alamat };
|
||||
}
|
||||
|
||||
// Fallback pattern matching
|
||||
const orderMatch = custInfo.match(/Order\s+Untuk\s*[:\-]\s*([^\n]+)/i);
|
||||
const alamatMatch = custInfo.match(/Alamat\s*[:\-]\s*([^\n]+)/i);
|
||||
|
||||
return {
|
||||
orderUntuk: orderMatch ? orderMatch[1].trim() : "",
|
||||
alamat: alamatMatch ? alamatMatch[1].trim() : ""
|
||||
};
|
||||
}
|
||||
|
||||
export { pool };
|
||||
@@ -0,0 +1,455 @@
|
||||
import { Pool } from "pg";
|
||||
import fs from "fs";
|
||||
import path from "path";
|
||||
import { parseDOMetadata } from "../utils/parser";
|
||||
|
||||
const UPLOADS_DIR = "/uploads";
|
||||
|
||||
export async function initDb(pool: Pool) {
|
||||
console.log("Database connection: Initializing schema...");
|
||||
|
||||
// 1. Create tables
|
||||
await pool.query(`
|
||||
CREATE TABLE IF NOT EXISTS documents (
|
||||
id SERIAL PRIMARY KEY,
|
||||
filename VARCHAR(255) UNIQUE NOT NULL,
|
||||
upload_time TIMESTAMP NOT NULL DEFAULT NOW(),
|
||||
size INTEGER NOT NULL DEFAULT 0,
|
||||
parsed BOOLEAN NOT NULL DEFAULT FALSE,
|
||||
metadata JSONB,
|
||||
layout_parsing_result JSONB,
|
||||
is_sample BOOLEAN NOT NULL DEFAULT FALSE,
|
||||
file_hash VARCHAR(64)
|
||||
);
|
||||
`);
|
||||
|
||||
// Ensure file_hash exists on existing tables
|
||||
try {
|
||||
await pool.query("ALTER TABLE documents ADD COLUMN IF NOT EXISTS file_hash VARCHAR(64);");
|
||||
} catch (alterErr) {
|
||||
console.error("Failed to alter documents table for file_hash:", alterErr);
|
||||
}
|
||||
|
||||
await pool.query(`
|
||||
CREATE TABLE IF NOT EXISTS ocr_items (
|
||||
id SERIAL PRIMARY KEY,
|
||||
document_id INTEGER NOT NULL REFERENCES documents(id) ON DELETE CASCADE,
|
||||
row_index INTEGER NOT NULL,
|
||||
kode_barang_original VARCHAR(255),
|
||||
kode_barang VARCHAR(255),
|
||||
nama_barang VARCHAR(255),
|
||||
banyak_original VARCHAR(255),
|
||||
banyak VARCHAR(255),
|
||||
jumlah_original VARCHAR(255),
|
||||
jumlah VARCHAR(255),
|
||||
is_flagged BOOLEAN NOT NULL DEFAULT FALSE,
|
||||
remark VARCHAR(1000),
|
||||
UNIQUE(document_id, row_index)
|
||||
);
|
||||
`);
|
||||
|
||||
// Create vendors table
|
||||
await pool.query(`
|
||||
CREATE TABLE IF NOT EXISTS vendors (
|
||||
id SERIAL PRIMARY KEY,
|
||||
name VARCHAR(255) UNIQUE NOT NULL,
|
||||
created_at TIMESTAMP NOT NULL DEFAULT NOW()
|
||||
);
|
||||
`);
|
||||
|
||||
// Seed vendor
|
||||
await pool.query(`
|
||||
INSERT INTO vendors (name)
|
||||
VALUES ('PT. CHAROEN POKPHAND INDONESIA Tbk KAWASAN INDUSTRI MODERN, BANTEN')
|
||||
ON CONFLICT (name) DO NOTHING;
|
||||
`);
|
||||
|
||||
// Create customers table
|
||||
await pool.query(`
|
||||
CREATE TABLE IF NOT EXISTS customers (
|
||||
id SERIAL PRIMARY KEY,
|
||||
name VARCHAR(255) UNIQUE NOT NULL,
|
||||
created_at TIMESTAMP NOT NULL DEFAULT NOW()
|
||||
);
|
||||
`);
|
||||
|
||||
// Seed customer
|
||||
await pool.query(`
|
||||
INSERT INTO customers (name)
|
||||
VALUES ('PT.PRIMAFOOD INTERNATIONAL, JL. ANCOL BARAT VIII/1, ANCOL, PADEMANGAN, JAKARTA UTARA, 14430')
|
||||
ON CONFLICT (name) DO NOTHING;
|
||||
`);
|
||||
|
||||
// Create sku_master table
|
||||
await pool.query(`
|
||||
CREATE TABLE IF NOT EXISTS sku_master (
|
||||
id SERIAL PRIMARY KEY,
|
||||
no_sku VARCHAR(255) UNIQUE NOT NULL,
|
||||
nama_item VARCHAR(255) NOT NULL,
|
||||
created_at TIMESTAMP NOT NULL DEFAULT NOW()
|
||||
);
|
||||
`);
|
||||
|
||||
// Create store_master table
|
||||
await pool.query(`
|
||||
CREATE TABLE IF NOT EXISTS store_master (
|
||||
id SERIAL PRIMARY KEY,
|
||||
nama_toko VARCHAR(255) NOT NULL,
|
||||
kode_toko VARCHAR(255) UNIQUE NOT NULL,
|
||||
alamat TEXT NOT NULL,
|
||||
created_at TIMESTAMP NOT NULL DEFAULT NOW()
|
||||
);
|
||||
`);
|
||||
|
||||
// Create arena_runs table
|
||||
await pool.query(`
|
||||
CREATE TABLE IF NOT EXISTS arena_runs (
|
||||
id SERIAL PRIMARY KEY,
|
||||
image_path TEXT NOT NULL,
|
||||
engine VARCHAR(50) NOT NULL,
|
||||
status VARCHAR(50) NOT NULL,
|
||||
ocr_result TEXT,
|
||||
time_elapsed_ms INTEGER,
|
||||
image_type VARCHAR(50) NOT NULL DEFAULT 'do',
|
||||
created_at TIMESTAMP NOT NULL DEFAULT NOW()
|
||||
);
|
||||
`);
|
||||
|
||||
// Upgrade schema if table already exists
|
||||
await pool.query(`
|
||||
ALTER TABLE arena_runs ADD COLUMN IF NOT EXISTS image_type VARCHAR(50) NOT NULL DEFAULT 'do';
|
||||
`);
|
||||
|
||||
// Backfill categorization for existing runs
|
||||
await pool.query(`
|
||||
UPDATE arena_runs
|
||||
SET image_type = 'product'
|
||||
WHERE image_type = 'do' AND (image_path LIKE '%/produk-pfm/%' OR image_path LIKE '%produk-pfm%' OR image_path LIKE '%Product%');
|
||||
`);
|
||||
|
||||
// Seed SKU entries
|
||||
await pool.query(`
|
||||
INSERT INTO sku_master (no_sku, nama_item) VALUES
|
||||
('11048006', 'BEBEK PARTING-NEW(*)'),
|
||||
('11110059', 'CEKER BERKUKU FROZEN PACK 1 KG(*)'),
|
||||
('11110074', 'CEKER 1 KG FROZEN (20 PAC/KARUNG)(*)'),
|
||||
('11140051', 'AMPELA FROZEN PACK 1 KG(*)'),
|
||||
('11140062', 'AMPELA 1 KG FROZEN (20 PAC/KARUNG)(*)'),
|
||||
('11148002', 'AMPELA BEBEK FROZEN 1 KG/PACK (NEW)(*)'),
|
||||
('11150052', 'HATI FROZEN PACK 1 KG(*)'),
|
||||
('11150055', 'JANTUNG FROZEN PACK 1 KG(*)'),
|
||||
('11150064', 'HATI 1 KG FROZEN (20 PAC/KARUNG)(*)'),
|
||||
('11150065', 'JANTUNG 1 KG FROZEN (20 PAC/KARUNG)(*)'),
|
||||
('11310012', 'AYAM SIZE 0 (0.6-0.7)KG(*)'),
|
||||
('11310013', 'AYAM SIZE1 FROZEN (0.75-0.8) KG(*)'),
|
||||
('11310014', 'AYAM SIZE 2 FROZEN (0.8-0.9)KG(*)'),
|
||||
('11310016', 'AYAM SIZE Z PR FROZ(*)'),
|
||||
('11310017', 'AYAM SIZE 0 PR FROZEN(*)'),
|
||||
('11310018', 'AYAM SIZE 1 PR FROZEN(*)'),
|
||||
('11310019', 'AYAM SIZE 2 PR FROZEN(*)'),
|
||||
('11310021', 'AYAM SIZE BESAR (B) FROZ (1-1.1)KG/PC(*)'),
|
||||
('11310022', 'AYAM SIZE A PR (0.9-1)KG/PC(*)'),
|
||||
('11310024', 'AYAM SIZE A FROZEN (0.9-1)KG/PC(*)'),
|
||||
('11310025', 'AYAM SIZE SUPER (C) FROZ(1.1-1.2)KG/ PC(*)'),
|
||||
('11310026', 'AYAM SIZE JUMBO (D) FROZ (1.2- 1.3)KG/PC(*)'),
|
||||
('11318301', 'BEBEK MUDA-BD1(1.0-1.1 KG)-NEW(*)'),
|
||||
('11318306', 'CP DUCK PEKING 1.5-1.6 KG/PC(*)'),
|
||||
('11318308', 'BEBEK PEKING SPR BD5(1.7 -1.8 )Kg-NEW(*)'),
|
||||
('11410043', 'PARTING 10 SIZE D FRESH BENSU 1.25 KG/PAC(*)'),
|
||||
('11420055', 'PARTING 12 ALL SIZE FROZ/PAC(*)'),
|
||||
('11600053', 'BONELESS LEG FROZEN 1 KG(*)'),
|
||||
('11620056', 'SBL (FILLET PAHA) 1 KG(*)'),
|
||||
('11640053', 'PAHA UTUH (1 KG)(*)'),
|
||||
('11650053', 'PAHA ATAS 1 KG(*)'),
|
||||
('11660050', 'PAHA BAWAH (1 KG)(*)'),
|
||||
('11690053', 'SBB (FILLET DADA )1 KG(*)'),
|
||||
('11690081', 'SBB JUMBO FZ (2.0 - 2.2 KG/PAC)(*)'),
|
||||
('11710051', 'DADA UTUH (1 KG)(*)'),
|
||||
('11720055', 'FULL WING FROZ PACK 1 KG(*)'),
|
||||
('11730050', 'MIDDLE WING FROZ PACK 1 KG(*)'),
|
||||
('11750050', 'FILLET MITRA 1 KG(*)'),
|
||||
('11818300', 'CP-BEBEK GORENG 400GR/PAC'),
|
||||
('11840002', 'AYAM JANTAN BKKL SZ 0 (600-700) GR/PC(*)'),
|
||||
('11959937', 'SATE AYAM FRESHMART 360 GR (PAC)'),
|
||||
('12010111', 'FIESTA CRISPY BUBBLE 400 GR/PAC'),
|
||||
('12010112', 'FIESTA CHICKEN NUGGET 400 GR/PAC'),
|
||||
('12010113', 'FIESTA CHICKEN NUGGET 200 GR/PAC'),
|
||||
('12010115', 'FIESTA NUGGET ZOO 400 GR/PAC'),
|
||||
('12010116', 'FIESTA NUGGET DINO 400 GR/PAC'),
|
||||
('12010117', 'FIESTA NUGGET HAPPY STAR 400 GR/PAC'),
|
||||
('12010119', 'FIESTA NUGGET CHEESE 123 400 GR/PAC'),
|
||||
('12010121', 'FIESTA NUGGET PIZZABC 400 GR/PAC'),
|
||||
('12010122', 'FIESTA CHEESY LOVER 400 GR/PAC'),
|
||||
('12010123', 'FIESTA GARLIC CHEESE 400 GR/PAC'),
|
||||
('12010124', 'FIESTA CHEESY CHIC W/BROCCOLI 400 GR/PAC'),
|
||||
('12010127', 'FIESTA SPICY NUGGET 400 GR/PAC'),
|
||||
('12010128', 'FIESTA VOLCANO CHEESE 400 GR/PAC'),
|
||||
('12010129', 'FIESTA CHEESY BOMBS CHICKEN NUGGET 400 GR'),
|
||||
('12010402', 'GOLDEN FIESTA NUGGET W/PINEAPPLE SAUCE 500 GR'),
|
||||
('12010509', 'CHAMP CRUNCHY NUGGET 450 GR/PAC'),
|
||||
('12010510', 'CHAMP NUGGET AYAM 225 GR/PAC'),
|
||||
('12010511', 'CHAMP NUGGET AYAM 450 GR/PAC'),
|
||||
('12010512', 'CHAMP NUGGET AYAM 900 GR/PAC'),
|
||||
('12010513', 'CHAMP NUGGET ABC KOMBINASI 225 GR/PAC'),
|
||||
('12010514', 'CHAMP NUGGET ABC KOMBINASI 450 GR/PAC'),
|
||||
('12010515', 'CHAMP KOIN KOMBINASI 450 GR/PAC'),
|
||||
('12010516', 'CHAMP KOIN KOMBINASI 200 GR/PAC'),
|
||||
('12010517', 'CHAMP NUGGET STICK 225 GR/PAC'),
|
||||
('12010518', 'CHAMP NUGGET STICK 450 GR/PAC'),
|
||||
('12010519', 'CHAMP NUGGET STICK 900 GR/PAC'),
|
||||
('12010520', 'CHAMP CHICKEN NUGGET BENTUK 123 450 GR/PAC'),
|
||||
('12010521', 'CHAMP NUGGET HOTZZ LEVEL 5 450 GR/PAC'),
|
||||
('12010606', 'CHAMP CRUNCHY NUGGET 225 GR/PAC'),
|
||||
('12010707', 'CHAMP MITRA NUGGET COIN 200 GR (NEW)'),
|
||||
('12010801', 'OKEY NUGGET 500GR'),
|
||||
('12012201', 'ASIMO NUGGET KOMBINASI 500 GR/PAC'),
|
||||
('12012202', 'ASIMO NUGGET KOMBINASI 1 KG/PAC'),
|
||||
('12012203', 'ASIMO NUGGET KOMBINASI 250 GR/PAC'),
|
||||
('12012501', 'AKUMO CHICKEN NAGET 250 GR'),
|
||||
('12012502', 'AKUMO CHICKEN NUGGET 500 GR'),
|
||||
('12012503', 'AKUMO CHICKEN NUGGET 1000 GR'),
|
||||
('12012504', 'AKUMO COIN 200 GR/PAC'),
|
||||
('12012505', 'AKUMO KOIN 400 GR/PAC'),
|
||||
('12020102', 'FIESTA SPICY WING 400 GR/PAC'),
|
||||
('12020401', 'GOLDEN FIESTA SP WING 500 GR'),
|
||||
('12030101', 'FIESTA STIKIE 400 GR/PAC'),
|
||||
('12030102', 'FIESTA STIKIE 200 GR/PAC'),
|
||||
('12030403', 'GOLDEN FIESTA STIKIE W/ SWEET CHILLI SAUCE 500GR'),
|
||||
('12030801', 'OKEY STICK 1000 GR'),
|
||||
('12030802', 'OKEY STICK 500GR'),
|
||||
('12032201', 'ASIMO STICK KOMBINASI 500 GR/PAC'),
|
||||
('12032202', 'ASIMO STICK KOMBINASI 1000 GR/PAC'),
|
||||
('12032203', 'ASIMO STIK KOMBINASI 250 GR/PAC'),
|
||||
('12032501', 'AKUMO CHICKEN STICK 250 GR'),
|
||||
('12032502', 'AKUMO CHICKEN STIK 500 GR'),
|
||||
('12032503', 'AKUMO CHICKEN STICK 1000 GR'),
|
||||
('12040101', 'FIESTA SCHNITZEL 400 GR/PAC'),
|
||||
('12040102', 'FIESTA CRISPY BUBBLE KATSU 400 GR/PAC'),
|
||||
('12040404', 'GOLDEN FIESTA CORDON BLEU BBQ SAUCE 500 GR'),
|
||||
('12040406', 'GOLDEN FIESTA KATSU W/CHEESE SAUCE 500 GR/PAC'),
|
||||
('12050103', 'FIESTA FRIED CHICKEN 400 GR/PAC'),
|
||||
('12050104', 'FIESTA HOT & CRISPY FRIED CHICKEN 400 GR/PAC'),
|
||||
('12050401', 'GOLDEN FIESTA CRISPY WING W/SP GLAZ SC 500 GR/PAC'),
|
||||
('12060103', 'FIESTA KARAGE 200 GR/PAC'),
|
||||
('12060104', 'FIESTA KARAGE 400 GR/PAC'),
|
||||
('12060105', 'FIESTA SPICY KARAGE 400 GR/PAC'),
|
||||
('12060402', 'GOLDEN FIESTA KARAGE CHILI SAUCE 500GR'),
|
||||
('12070101', 'FIESTA POK-POK 400 GR/PAC (NEW)'),
|
||||
('12080101', 'FIESTA SPICY CHICK 400 GR/PAC'),
|
||||
('12130102', 'FIESTA CRISPY BURGER 360 GR (NEW)'),
|
||||
('12130504', 'CHAMP BURGER 315 GR (NEW)'),
|
||||
('12140105', 'FIESTA CHICK TOFU 400 GR/PAC'),
|
||||
('12150201', 'FIESTA DS CRISPY CRUNCH 300 GR/PAC'),
|
||||
('12150501', 'CHAMP CRUNCHY HOTZZ 300 GR/PAC'),
|
||||
('12190103', 'FIESTA DELISTRIPE 400 GR/PAC'),
|
||||
('12240102', 'FIESTA CHEESY ITALIAN R/BITES 400 GR/PAC'),
|
||||
('12240103', 'FIESTA YAKINIKU R/BITES 400 GR/PAC'),
|
||||
('13010101', 'FIESTA CHICK SSG 300 GR'),
|
||||
('13010102', 'FIESTA CHICK SSG 500 GR'),
|
||||
('13010103', 'FIESTA CHICK SSG 200 GR/PAC'),
|
||||
('13010111', 'FIESTA SOSIS BRATWURST 300 GR'),
|
||||
('13010112', 'FIESTA CHEESE SSG 300 GR'),
|
||||
('13010113', 'FIESTA SOSIS CURRYWURST 300 GR'),
|
||||
('13010114', 'FIESTA SSG BOCKWURST 300GR'),
|
||||
('13010115', 'FIESTA SSG WIENER 300GR'),
|
||||
('13010116', 'FIESTA SSG ORIGINAL 300 GR'),
|
||||
('13010117', 'FIESTA SSG FRANKFURTER 300GR'),
|
||||
('13010118', 'FIESTA RTG SSG 65 GR/PAC'),
|
||||
('13010119', 'FIESTA RTG C/SPICY KOREAN 60 GR/PAC'),
|
||||
('13010120', 'FIESTA RTG C/CHEESY MELTS 65 GR/PAC'),
|
||||
('13010122', 'FIESTA RTG SAUSAGE WITH HOT LAVA 60G'),
|
||||
('13010123', 'FIESTA RTG SAUSAGE WITH CHEESE LAVA 60G'),
|
||||
('13010124', 'FIESTA RTG SAUSAGE HICKORY SAUCE 60GR'),
|
||||
('13010125', 'FIESTA RTG SAUSAGE WITH MENTAI LAVA 60GR'),
|
||||
('13010510', 'CHAMP CHICK SSG 75 GR'),
|
||||
('13010513', 'CHAMP CHICK SSG 375 GR'),
|
||||
('13010514', 'CHAMP CHICK SSG 1000 GR'),
|
||||
('13010518', 'CHAMP SSG JUMBO BAKAR 500 GR/PAC'),
|
||||
('13010519', 'CHAMP SSG BAKAR MINI 500 GR/PAC-INACT'),
|
||||
('13010521', 'CHAMP CHICK SSG 150 GR/PAC (NEW)'),
|
||||
('13010523', 'CHAMP CHICK SSG AYAM MADU 300 GR/PAC'),
|
||||
('13010524', 'CHAMP SSG JUMBO BAKAR 500 GR/PAC (NEW)'),
|
||||
('13010525', 'CHAMP SSG BAKAR MINI 500 GR/PAC (NEW)'),
|
||||
('13010809', 'OKEY CHICK SSG 500GR-INACT'),
|
||||
('13010815', 'OKEY SSG BAKAR JUMBO 500 GR/PAC (NEW)'),
|
||||
('13010816', 'OKEY SSG BAKAR MINI 500 GR/PAC (NEW)'),
|
||||
('13010817', 'OKEY SSG CHICK KOMBINASI 500 GR (EXTRA1)'),
|
||||
('13010818', 'OKEY SSG CHICK KOMBINASI 1 KG (EXTR$A2)'),
|
||||
('13012205', 'ASIMO SOSIS AYAM KOMBINASI 375 GR (PAC)'),
|
||||
('13012206', 'ASIMO SOSIS AYAM KOMBINASI 500 GR'),
|
||||
('13012207', 'ASIMO SOSIS AYAM KOMBINASI 750 GR'),
|
||||
('13012208', 'ASIMO SOSIS AYAM KOMBINASI 1000 GR'),
|
||||
('13012209', 'ASIMO SSG CHICK KOMBINASI 500 GR (EXTRA1)'),
|
||||
('13012210', 'ASIMO SSG CHICK KOMBINASI 1 KG (EXTR$A2)'),
|
||||
('13030101', 'FIESTA CHICK MEAT BALL 300 GR'),
|
||||
('13030102', 'FIESTA CHICK MEATBALL 500 GR'),
|
||||
('13030501', 'CHAMP CHICK MEATBALL 200 GR'),
|
||||
('13030502', 'CHAMP CHICK MEATBALL 500 GR'),
|
||||
('13050101', 'FIESTA SCB 250 GR'),
|
||||
('13050105', 'FIESTA CHICKEN SLICE 300 GR'),
|
||||
('13050106', 'FIESTA BEEF SLICE 300 GR'),
|
||||
('13070501', 'CHAMP BEEF SSG SERBAGUNA 150 GR'),
|
||||
('13070502', 'CHAMP BEEF SSG SERBAGUNA 375GR'),
|
||||
('13070505', 'CHAMP BEEF SSG GORENG 375 GR'),
|
||||
('13070506', 'CHAMP FRANKFURTER SSG 375GR'),
|
||||
('13100512', 'CHAMP CHICK SSG S/SANTAP ORIG 546GR (CAN)'),
|
||||
('13110504', 'CHAMP BEEF BALL 500GR'),
|
||||
('13170510', 'CHAMP BEEF BBQ SSG S/SANTAP 546GR (CAN)'),
|
||||
('15010101', 'FIESTA SHOESTRING 500 GR'),
|
||||
('15010102', 'FIESTA SHOESTRING 1000 GR'),
|
||||
('15010107', 'FIESTA FRENCH F SHOESTRING INSTITUSI 2KG'),
|
||||
('15020101', 'FIESTA STRAIGHT CUT 500 GR'),
|
||||
('15020102', 'FIESTA STRAIGHT CUT 1000 GR'),
|
||||
('15030101', 'FIESTA CRINKLE CUT 500 GR'),
|
||||
('15030102', 'FIESTA CRINKLE CUT 1000 GR'),
|
||||
('15040101', 'FIESTA BATTER COATED 500 GR'),
|
||||
('15040102', 'FIESTA BATTER COATED 1000 GR'),
|
||||
('16060103', 'FIESTA CHICK SIOMAY 900GR'),
|
||||
('16060113', 'FIESTA CHICK SIOMAY 180GR (NEW)'),
|
||||
('16060114', 'FIESTA GYOZA 180 GR (NEW)'),
|
||||
('16060119', 'FIESTA RTG SIOMAY 54 GR/PAC'),
|
||||
('16060120', 'FIESTA KEECHO 400 GR/PAC'),
|
||||
('16060121', 'FIESTA CHICKEN TOFU 400 GR/PAC (NEW)'),
|
||||
('16060503', 'CHAMP CHICK&FISH SIOMAY 180 GR (NEW)'),
|
||||
('17200109', 'FIESTA RTS C/TERIYAKI 300GR/PAC'),
|
||||
('17200110', 'FIESTA RTS C/RENDANG 300GR/PAC'),
|
||||
('17200111', 'FIESTA RTS C/W RUJAK SC 300GR/PAC'),
|
||||
('17200112', 'FIESTA RTS C/W S/MATAH 300GR/PAC'),
|
||||
('17210106', 'FIESTA RTS B/YAKINIKU 300GR/PAC'),
|
||||
('17210107', 'FIESTA RTS B/RENDANG 300GR/PAC'),
|
||||
('17210108', 'FIESTA RTS B/BLACKPEPPER 300GR/PAC'),
|
||||
('17210109', 'FIESTA RTS B/BULGOGI 300GR/PAC'),
|
||||
('18050102', 'FIESTA RTG BAKSO KEJU 60 GR/PAC'),
|
||||
('18050103', 'FIESTA RTG BAKSO BAKAR BBQ 60 GR/PAC'),
|
||||
('18050104', 'FIESTA RTG BEEF BALL WITH MENTAI LAVA 55GR'),
|
||||
('18050105', 'FIESTA RTG BEEF BALL WITH CHEESE LAVA 55GR'),
|
||||
('20010101', 'FIESTA CRISPY CRUMBS 200 GR'),
|
||||
('20010102', 'FIESTA TP ROTI PUTIH 200 GR'),
|
||||
('20040101', 'FIESTA RAMEN BEKU 570 GR/PAC'),
|
||||
('20120102', 'FIESTA T/B AYAM GORENG 80 GR'),
|
||||
('20120105', 'FIESTA T/B SERBAGUNA 80 GR'),
|
||||
('20120106', 'FIESTA T/B KREMES 80 GR'),
|
||||
('20120115', 'FIESTA RACIK AYAM GORENG 20 GR/PAC'),
|
||||
('20120116', 'FIESTA RACIK NASI GORENG 20 GR/PAC'),
|
||||
('21000123', 'FIESTA RICE W/GEPREK CHICKEN 320GR/PAC'),
|
||||
('21000124', 'FIESTA RICE W/CHICK RUJAK 320 GR/PAC'),
|
||||
('21000125', 'FIESTA RICE W/KOREAN BBQ CHICK 320 GR/PAC'),
|
||||
('21000126', 'NEW FIESTA CHICK RENDANG W RICE 320GR (PAC)'),
|
||||
('21000127', 'NEW FIESTA CHICK TERIYAKI W RICE 320GR (PAC)'),
|
||||
('21000128', 'NEW FIESTA CHICK TANDORI W RICE 320GR (PAC)'),
|
||||
('21000129', 'NEW FIESTA RICE W KARAGE&SSS 320GR (PAC)'),
|
||||
('21000130', 'NEW FIESTA RICE W/C CHEESE BULDAK 320GR (PAC)'),
|
||||
('21000131', 'NEW FIESTA RICE W CHIC CURRY 320GR (PAC)'),
|
||||
('21000132', 'NEW FIESTA RICE W CHICK DONBURI 320GR (PAC)'),
|
||||
('21000133', 'NEW FIESTA RICE W CHICK SATAY 320GR (PAC)'),
|
||||
('21000134', 'NEW FIESTA COCONUT RICE W SPICY CHICK 320GR (PAC)'),
|
||||
('21000135', 'NEW FIESTA RICE W POPBITES S/MATAH 320GR (PAC)'),
|
||||
('21000136', 'NEW FIESTA TUMERIC W POPBITES 320GR (PAC)'),
|
||||
('21000137', 'FIESTA HAINAMESE CHICKEN RICE 320GR (PAC)'),
|
||||
('21010101', 'FIESTA TRUFFLE GYUDON 320 GR/PAC'),
|
||||
('21010102', 'NEW FIESTA BEEF YAKINIKU W RICE 320GR (PAC)'),
|
||||
('21010103', 'NEW FIESTA BEEF BULGOGI W RICE 320GR (PAC)'),
|
||||
('21010104', 'NEW FIESTA BEEF RENDANG W RICE 320GR (PAC)'),
|
||||
('21010105', 'NEW FIESTA RICE W BEEF BLACKPEPPER 320GR (PAC)'),
|
||||
('21200107', 'NEW FIESTA SPAGHETTI CARBONARA 300GR (PAC)'),
|
||||
('21200108', 'NEW FIESTA SPAGHETTI CHIC BOLOGNESE 320GR (PAC)'),
|
||||
('21200109', 'NEW FIESTA ITALIAN MEATBALL SPAGHETTI 320GR (PAC)'),
|
||||
('21310103', 'NEW FIESTA SCB&S/SSG FRIED RICE 320GR (PAC)'),
|
||||
('21500101', 'FIESTA CHICK SSG & C. BALL PIZZA 230GR/PAC'),
|
||||
('21500102', 'FIESTA CHEESY BEEF PIZZA 230GR/PAC'),
|
||||
('91000012', 'PHOTOCARD RTG'),
|
||||
('1188002W', 'PAHA ATAS 25-30 G FZ (*)'),
|
||||
('1195008A', 'RTC CHICKEN KALASAN 400 GR (PAC)'),
|
||||
('1195008E', 'RTC CHICKEN TERIYAKI 400 GR (PAC)'),
|
||||
('1195008X', 'RTC CHICKEN SPICY 400 GR (PAC)')
|
||||
ON CONFLICT (no_sku) DO NOTHING;
|
||||
`);
|
||||
|
||||
console.log("Database schema initialized successfully.");
|
||||
|
||||
// 2. Sync existing filesystem uploads to DB
|
||||
try {
|
||||
if (!fs.existsSync(UPLOADS_DIR)) {
|
||||
return;
|
||||
}
|
||||
|
||||
const files = fs.readdirSync(UPLOADS_DIR);
|
||||
const jsonFiles = files.filter(f => f.endsWith(".json"));
|
||||
|
||||
for (const jsonFile of jsonFiles) {
|
||||
const baseFilename = jsonFile.slice(0, -5); // e.g. "1716912345-doc.jpg"
|
||||
const imagePath = path.join(UPLOADS_DIR, baseFilename);
|
||||
const jsonPath = path.join(UPLOADS_DIR, jsonFile);
|
||||
|
||||
if (!fs.existsSync(imagePath)) {
|
||||
continue; // Image/PDF doesn't exist, skip
|
||||
}
|
||||
|
||||
// Check if already synced in DB
|
||||
const checkRes = await pool.query("SELECT id FROM documents WHERE filename = $1", [baseFilename]);
|
||||
if (checkRes.rowCount && checkRes.rowCount > 0) {
|
||||
continue; // Already in DB, skip
|
||||
}
|
||||
|
||||
console.log(`Syncing historical file ${baseFilename} to PostgreSQL...`);
|
||||
const stats = fs.statSync(imagePath);
|
||||
const jsonContent = fs.readFileSync(jsonPath, "utf-8");
|
||||
|
||||
let parsedResult;
|
||||
try {
|
||||
parsedResult = JSON.parse(jsonContent);
|
||||
} catch (err) {
|
||||
console.error(`Failed to parse JSON file ${jsonFile}:`, err);
|
||||
continue;
|
||||
}
|
||||
|
||||
const pipelineResult = parsedResult.result || parsedResult;
|
||||
const page0 = pipelineResult?.layoutParsingResults?.[0] || {};
|
||||
const markdownText = page0?.markdown?.text || "";
|
||||
const docMetadata = parseDOMetadata(markdownText);
|
||||
|
||||
// Insert document
|
||||
const insertDocRes = await pool.query(`
|
||||
INSERT INTO documents (filename, upload_time, size, parsed, metadata, layout_parsing_result, is_sample)
|
||||
VALUES ($1, $2, $3, $4, $5, $6, $7)
|
||||
RETURNING id
|
||||
`, [
|
||||
baseFilename,
|
||||
stats.mtime,
|
||||
stats.size,
|
||||
true,
|
||||
JSON.stringify(docMetadata),
|
||||
JSON.stringify(pipelineResult),
|
||||
false
|
||||
]);
|
||||
|
||||
const docId = insertDocRes.rows[0].id;
|
||||
|
||||
// Insert items
|
||||
for (let i = 0; i < docMetadata.items.length; i++) {
|
||||
const item = docMetadata.items[i];
|
||||
await pool.query(`
|
||||
INSERT INTO ocr_items (
|
||||
document_id, row_index,
|
||||
kode_barang_original, kode_barang,
|
||||
nama_barang,
|
||||
banyak_original, banyak,
|
||||
jumlah_original, jumlah,
|
||||
is_flagged, remark
|
||||
)
|
||||
VALUES ($1, $2, $3, $3, $4, $5, $5, $6, $6, false, '')
|
||||
ON CONFLICT DO NOTHING
|
||||
`, [
|
||||
docId,
|
||||
i,
|
||||
item.kodeBarang,
|
||||
item.namaBarang,
|
||||
item.banyak,
|
||||
item.jumlah
|
||||
]);
|
||||
}
|
||||
}
|
||||
} catch (err) {
|
||||
console.error("Failed to sync uploads to database:", err);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,362 @@
|
||||
import http from "http";
|
||||
import fs from "fs";
|
||||
import path from "path";
|
||||
|
||||
export function dockerRequest(path: string, method: string, body: any = null): Promise<any> {
|
||||
return new Promise((resolve, reject) => {
|
||||
const options = {
|
||||
socketPath: "/var/run/docker.sock",
|
||||
path: path,
|
||||
method: method,
|
||||
headers: {
|
||||
"Content-Type": "application/json",
|
||||
},
|
||||
};
|
||||
|
||||
const req = http.request(options, (res) => {
|
||||
const chunks: Buffer[] = [];
|
||||
res.on("data", (chunk) => chunks.push(chunk));
|
||||
res.on("end", () => {
|
||||
const resBuffer = Buffer.concat(chunks);
|
||||
const data = resBuffer.toString("utf8");
|
||||
if (res.statusCode && res.statusCode >= 200 && res.statusCode < 300) {
|
||||
try {
|
||||
resolve(data ? JSON.parse(data) : null);
|
||||
} catch (e) {
|
||||
resolve(data);
|
||||
}
|
||||
} else {
|
||||
reject(new Error(`Docker API Error ${res.statusCode}: ${data}`));
|
||||
}
|
||||
});
|
||||
});
|
||||
|
||||
req.on("error", (err) => reject(err));
|
||||
if (body) {
|
||||
req.write(JSON.stringify(body));
|
||||
}
|
||||
req.end();
|
||||
});
|
||||
}
|
||||
|
||||
export function parseDockerStream(buffer: Buffer): { stdout: string; stderr: string } {
|
||||
let stdout = "";
|
||||
let stderr = "";
|
||||
let offset = 0;
|
||||
|
||||
while (offset + 8 <= buffer.length) {
|
||||
const streamType = buffer.readUInt8(offset);
|
||||
const size = buffer.readUInt32BE(offset + 4);
|
||||
|
||||
if (offset + 8 + size > buffer.length) {
|
||||
break;
|
||||
}
|
||||
|
||||
const payload = buffer.toString("utf8", offset + 8, offset + 8 + size);
|
||||
if (streamType === 1) {
|
||||
stdout += payload;
|
||||
} else if (streamType === 2) {
|
||||
stderr += payload;
|
||||
}
|
||||
offset += 8 + size;
|
||||
}
|
||||
|
||||
if (stdout === "" && stderr === "" && buffer.length > 0) {
|
||||
stdout = buffer.toString("utf8");
|
||||
}
|
||||
|
||||
return { stdout, stderr };
|
||||
}
|
||||
|
||||
export function runExec(containerName: string, cmd: string[]): Promise<string> {
|
||||
return new Promise(async (resolve, reject) => {
|
||||
try {
|
||||
const execConfig = {
|
||||
AttachStdout: true,
|
||||
AttachStderr: true,
|
||||
Cmd: cmd,
|
||||
};
|
||||
const createRes = await dockerRequest(`/containers/${containerName}/exec`, "POST", execConfig);
|
||||
const execId = createRes.Id;
|
||||
|
||||
const options = {
|
||||
socketPath: "/var/run/docker.sock",
|
||||
path: `/exec/${execId}/start`,
|
||||
method: "POST",
|
||||
headers: {
|
||||
"Content-Type": "application/json",
|
||||
},
|
||||
};
|
||||
|
||||
const req = http.request(options, (res) => {
|
||||
const chunks: Buffer[] = [];
|
||||
res.on("data", (chunk) => chunks.push(chunk));
|
||||
res.on("end", () => {
|
||||
const streamData = parseDockerStream(Buffer.concat(chunks));
|
||||
resolve(streamData.stdout || streamData.stderr);
|
||||
});
|
||||
});
|
||||
|
||||
req.on("error", (err) => reject(err));
|
||||
req.write(JSON.stringify({ Detach: false, Tty: false }));
|
||||
req.end();
|
||||
} catch (err) {
|
||||
reject(err);
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
export function getProcessName(pid: number): string {
|
||||
try {
|
||||
const commPath = `/proc/${pid}/comm`;
|
||||
if (fs.existsSync(commPath)) {
|
||||
return fs.readFileSync(commPath, "utf8").trim();
|
||||
}
|
||||
} catch (err) {
|
||||
// ignore
|
||||
}
|
||||
return "";
|
||||
}
|
||||
|
||||
export function makeHumanReadableName(procName: string): string {
|
||||
const nameLower = procName.toLowerCase();
|
||||
if (nameLower.includes("rustdesk")) return "RustDesk Remote Desktop";
|
||||
if (nameLower.includes("xorg")) return "Xorg Graphics Server";
|
||||
if (nameLower.includes("vllm") || nameLower.includes("enginecore")) return "vLLM Inference Server";
|
||||
if (nameLower.includes("python")) return "Python / Gradio App";
|
||||
if (nameLower.includes("node")) return "Next.js Web App";
|
||||
if (nameLower.includes("postgres")) return "PostgreSQL Database";
|
||||
if (nameLower.includes("nginx")) return "Nginx Load Balancer";
|
||||
return procName;
|
||||
}
|
||||
|
||||
export async function getGpuInfo(): Promise<any[]> {
|
||||
try {
|
||||
const gpuOutput = await runExec("paddleocr-vllm-server", [
|
||||
"nvidia-smi",
|
||||
"--query-gpu=index,name,utilization.gpu,utilization.memory,memory.total,memory.used,memory.free,uuid",
|
||||
"--format=csv,noheader,nounits",
|
||||
]);
|
||||
|
||||
const gpus: any[] = [];
|
||||
if (gpuOutput) {
|
||||
const lines = gpuOutput.split("\n");
|
||||
for (const line of lines) {
|
||||
if (!line.trim()) continue;
|
||||
const parts = line.split(",").map((p) => p.trim());
|
||||
if (parts.length >= 8) {
|
||||
gpus.push({
|
||||
index: parts[0],
|
||||
name: parts[1],
|
||||
gpu_util: parseInt(parts[2]) || 0,
|
||||
mem_util: parseInt(parts[3]) || 0,
|
||||
mem_total: parseInt(parts[4]) || 0,
|
||||
mem_used: parseInt(parts[5]) || 0,
|
||||
mem_free: parseInt(parts[6]) || 0,
|
||||
uuid: parts[7],
|
||||
processes: [],
|
||||
});
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
const procOutput = await runExec("paddleocr-vllm-server", [
|
||||
"nvidia-smi",
|
||||
"--query-compute-apps=gpu_uuid,pid,process_name,used_memory",
|
||||
"--format=csv,noheader,nounits",
|
||||
]);
|
||||
|
||||
if (procOutput) {
|
||||
const lines = procOutput.split("\n");
|
||||
for (const line of lines) {
|
||||
if (!line.trim()) continue;
|
||||
const parts = line.split(",").map((p) => p.trim());
|
||||
if (parts.length >= 4) {
|
||||
const gpuUuid = parts[0];
|
||||
const pid = parseInt(parts[1]);
|
||||
const procName = parts[2];
|
||||
const usedMem = parseInt(parts[3]);
|
||||
|
||||
const gpu = gpus.find((g) => g.uuid === gpuUuid);
|
||||
if (gpu) {
|
||||
const systemProcName = getProcessName(pid) || procName;
|
||||
gpu.processes.push({
|
||||
pid,
|
||||
name: procName,
|
||||
readable_name: makeHumanReadableName(systemProcName),
|
||||
used_mem: usedMem,
|
||||
});
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return gpus;
|
||||
} catch (err) {
|
||||
console.error("Failed to query GPUs:", err);
|
||||
return [];
|
||||
}
|
||||
}
|
||||
|
||||
export async function getContainerStatus(containerName: string): Promise<string> {
|
||||
try {
|
||||
const info = await dockerRequest(`/containers/${containerName}/json`, "GET");
|
||||
return info.State.Status;
|
||||
} catch (err) {
|
||||
return "stopped";
|
||||
}
|
||||
}
|
||||
|
||||
export async function manageContainer(containerName: string, action: "start" | "stop" | "restart"): Promise<void> {
|
||||
await dockerRequest(`/containers/${containerName}/${action}`, "POST");
|
||||
}
|
||||
|
||||
export async function recreateContainer(containerName: string, newCudaDevices?: string): Promise<void> {
|
||||
const inspect = await dockerRequest(`/containers/${containerName}/json`, "GET");
|
||||
|
||||
try {
|
||||
await dockerRequest(`/containers/${containerName}/stop`, "POST");
|
||||
} catch (e) {
|
||||
// ignore
|
||||
}
|
||||
|
||||
const rand = Math.floor(Math.random() * 10000);
|
||||
const oldTempName = `${containerName}_old_${rand}`;
|
||||
await dockerRequest(`/containers/${containerName}/rename?name=${oldTempName}`, "POST");
|
||||
|
||||
const config: any = {
|
||||
...inspect.Config,
|
||||
HostConfig: inspect.HostConfig,
|
||||
NetworkingConfig: {
|
||||
EndpointsConfig: inspect.NetworkSettings.Networks,
|
||||
},
|
||||
};
|
||||
|
||||
// Ensure Name is not copied from Inspect root as it's not a field in Create
|
||||
delete config.Name;
|
||||
|
||||
if (newCudaDevices && config.Env) {
|
||||
config.Env = config.Env.map((envStr: string) => {
|
||||
if (envStr.startsWith("CUDA_VISIBLE_DEVICES=")) {
|
||||
return `CUDA_VISIBLE_DEVICES=${newCudaDevices}`;
|
||||
}
|
||||
return envStr;
|
||||
});
|
||||
}
|
||||
|
||||
const createRes = await dockerRequest(`/containers/create?name=${containerName}`, "POST", config);
|
||||
const newId = createRes.Id;
|
||||
|
||||
await dockerRequest(`/containers/${newId}/start`, "POST");
|
||||
|
||||
try {
|
||||
await dockerRequest(`/containers/${oldTempName}`, "DELETE");
|
||||
} catch (e) {
|
||||
// ignore
|
||||
}
|
||||
}
|
||||
|
||||
export async function getEnvSettings(): Promise<{ cuda_devices: string }> {
|
||||
const envPath = path.join(process.cwd(), "..", ".env");
|
||||
const settings = { cuda_devices: "1" };
|
||||
try {
|
||||
if (fs.existsSync(envPath)) {
|
||||
const content = fs.readFileSync(envPath, "utf8");
|
||||
const lines = content.split("\n");
|
||||
for (const line of lines) {
|
||||
const trimmed = line.trim();
|
||||
if (!trimmed || trimmed.startsWith("#")) continue;
|
||||
const [k, v] = trimmed.split("=");
|
||||
if (k && k.trim() === "CUDA_VISIBLE_DEVICES" && v) {
|
||||
settings.cuda_devices = v.trim().replace(/['"]/g, "");
|
||||
}
|
||||
}
|
||||
}
|
||||
} catch (err) {
|
||||
console.error("Failed to read env settings:", err);
|
||||
}
|
||||
return settings;
|
||||
}
|
||||
|
||||
export async function saveEnvSettings(cuda_devices: string): Promise<void> {
|
||||
const envPath = path.join(process.cwd(), "..", ".env");
|
||||
try {
|
||||
let lines: string[] = [];
|
||||
if (fs.existsSync(envPath)) {
|
||||
lines = fs.readFileSync(envPath, "utf8").split("\n");
|
||||
}
|
||||
|
||||
let found = false;
|
||||
const newLines = lines.map((line) => {
|
||||
if (line.trim().startsWith("CUDA_VISIBLE_DEVICES=")) {
|
||||
found = true;
|
||||
return `CUDA_VISIBLE_DEVICES=${cuda_devices}`;
|
||||
}
|
||||
return line;
|
||||
});
|
||||
|
||||
if (!found) {
|
||||
newLines.push(`CUDA_VISIBLE_DEVICES=${cuda_devices}`);
|
||||
}
|
||||
|
||||
fs.writeFileSync(envPath, newLines.join("\n"), "utf8");
|
||||
} catch (err) {
|
||||
console.error("Failed to save env settings:", err);
|
||||
throw err;
|
||||
}
|
||||
}
|
||||
|
||||
export async function unloadOtherEngines(): Promise<{ stopped: string[]; failed: string[] }> {
|
||||
const stopped: string[] = [];
|
||||
const failed: string[] = [];
|
||||
|
||||
try {
|
||||
const containers = await dockerRequest("/containers/json", "GET");
|
||||
if (!Array.isArray(containers)) {
|
||||
throw new Error("Invalid response from Docker API: expected container array.");
|
||||
}
|
||||
|
||||
const stopPromises: Promise<void>[] = [];
|
||||
|
||||
for (const container of containers) {
|
||||
if (!container.Names || !Array.isArray(container.Names)) continue;
|
||||
|
||||
const rawName = container.Names[0] || "";
|
||||
const name = rawName.startsWith("/") ? rawName.slice(1) : rawName;
|
||||
const nameLower = name.toLowerCase();
|
||||
|
||||
const matchesEngine =
|
||||
nameLower.includes("lighton") ||
|
||||
nameLower.includes("glm") ||
|
||||
nameLower.includes("dots") ||
|
||||
nameLower.includes("deepseek");
|
||||
|
||||
const isExcluded =
|
||||
nameLower.includes("paddleocr") ||
|
||||
nameLower.includes("nemotron");
|
||||
|
||||
if (matchesEngine && !isExcluded) {
|
||||
console.log(`Queueing unload for container: ${name} (${container.Id})`);
|
||||
|
||||
const stopPromise = dockerRequest(`/containers/${container.Id}/stop`, "POST")
|
||||
.then(() => {
|
||||
stopped.push(name);
|
||||
})
|
||||
.catch((err) => {
|
||||
console.error(`Failed to stop container ${name}:`, err);
|
||||
failed.push(`${name} (${err.message})`);
|
||||
});
|
||||
|
||||
stopPromises.push(stopPromise);
|
||||
}
|
||||
}
|
||||
|
||||
await Promise.all(stopPromises);
|
||||
} catch (err: any) {
|
||||
console.error("Failed to unload other engines:", err);
|
||||
throw err;
|
||||
}
|
||||
|
||||
return { stopped, failed };
|
||||
}
|
||||
|
||||
@@ -0,0 +1,26 @@
|
||||
export function normalizeImageSrc(src: string): string {
|
||||
if (!src) return "";
|
||||
if (src.startsWith("http://") || src.startsWith("https://") || src.startsWith("data:")) {
|
||||
return src;
|
||||
}
|
||||
return `data:image/png;base64,${src}`;
|
||||
}
|
||||
|
||||
export interface LayoutPageResult {
|
||||
outputImages?: Record<string, string>;
|
||||
}
|
||||
|
||||
/** Same visualization URL selection as DO-PFM Visual Grid (second image if present, else first). */
|
||||
export function extractLayoutVisUrl(page0: LayoutPageResult | null | undefined): string {
|
||||
const outImgs = page0?.outputImages || {};
|
||||
const sortedUrls = Object.values(outImgs).filter(Boolean) as string[];
|
||||
const visUrl = sortedUrls.length >= 2 ? sortedUrls[1] : sortedUrls[0] || "";
|
||||
return normalizeImageSrc(visUrl);
|
||||
}
|
||||
|
||||
export function extractLayoutVisUrlFromResult(
|
||||
layoutParsingResult: { layoutParsingResults?: LayoutPageResult[] } | null | undefined
|
||||
): string {
|
||||
const page0 = layoutParsingResult?.layoutParsingResults?.[0];
|
||||
return extractLayoutVisUrl(page0);
|
||||
}
|
||||
@@ -0,0 +1,111 @@
|
||||
import { parseDOMetadata, sanitizeParsedMetadata } from "./parser";
|
||||
import assert from "assert";
|
||||
|
||||
function makeBlankMeta() {
|
||||
return { vendorInfo: "Not Found", customerInfo: "Not Found", tanggal: "Not Found", noSO: "Not Found", noDO: "Not Found", noPO: "Not Found", items: [] as any[], platTruk: "" };
|
||||
}
|
||||
|
||||
function runTests() {
|
||||
const YY = new Date().getFullYear().toString().slice(-2);
|
||||
let failures = 0;
|
||||
|
||||
// ===== parseDOMetadata tests =====
|
||||
console.log("=== parseDOMetadata tests ===");
|
||||
const parseTests = [
|
||||
{ name: "PO standard PO/26/", markdown: "No. PO : PO/26/0000178435\nTanggal: 15 May 2026", expected: { noPO: `PO/${YY}/0000178435` } },
|
||||
{ name: "PO misread P0/26/ on label", markdown: "No. PO : P0/26/0000236828\nTanggal: 23 June 2026", expected: { noPO: `PO/${YY}/0000236828` } },
|
||||
{ name: "PO label raw 10-digit, real PO in body", markdown: "No. PO : 1659980277\nP0/26/0000230828\nTanggal: 23 June 2026", expected: { noPO: `PO/${YY}/0000230828` } },
|
||||
{ name: "PO misread F0/20/ — use current year NOT 20", markdown: "No. PO : F0/20/0000190929\nTanggal: 25 May 2020", expected: { noPO: `PO/${YY}/0000190929` } },
|
||||
{ name: "PO body P0/26/", markdown: "Purchase order P0/26/998877\nTanggal: 15 May 2026", expected: { noPO: `PO/${YY}/998877` } },
|
||||
{ name: "PO real doc: label raw, body has P0/26/", markdown: "Tanggal :\nNo.SO : 23 June 2026\nNo. DO : 1691960321\nNo.PO : 1659980277\nP0/26/0000236828", expected: { noPO: `PO/${YY}/0000236828`, tanggal: "23 June 2026" } },
|
||||
{ name: "PO fused F012070000170727", markdown: "Tanggal : 25 May 2020\nNo. PO : F012070000170727", expected: { noPO: `PO/${YY}/0000170727` } },
|
||||
{ name: "PO fused PO12070000190729", markdown: "Tanggal: 25 May 2020\nNo.PO : PO12070000190729", expected: { noPO: `PO/${YY}/0000190729` } },
|
||||
{ name: "PO noise digits PO120/0000170727", markdown: "Tanggal:25 Hv 2024\nNo. PO : PO120/0000170727", expected: { noPO: `PO/${YY}/0000170727` } },
|
||||
{ name: "Date trailing noise cut", markdown: "Tanggal: 15 May 2026 No. SO\nNo. PO : PO/26/0000178435", expected: { tanggal: "15 May 2026" } },
|
||||
{ name: "Date no space 25May2020", markdown: "Tanggal:25May2020\nNo. PO : PO/26/0000178435", expected: { tanggal: "25 May 2020" } },
|
||||
{ name: "Date standard 23 June 2026", markdown: "Tanggal : 23 June 2026\nNo. PO : PO/26/0000178435", expected: { tanggal: "23 June 2026" } },
|
||||
{ name: "Date Tanggal blank shifted to No.SO", markdown: "Tanggal :\nNo.SO : 23 June 2026\nNo. PO : PO/26/0000178435", expected: { tanggal: "23 June 2026" } },
|
||||
{ name: "Date prefix timestamp noise", markdown: "02:17:59/2 of Tanggal : 15 May 2026\nNo. PO : PO/26/0000178435", expected: { tanggal: "15 May 2026" } },
|
||||
{ name: "Date (Asli/Copy) prefix noise", markdown: "Tanggal: (Asli/Copy) 15 May 2026\nNo. PO : PO/26/0000178435", expected: { tanggal: "15 May 2026" } },
|
||||
{ name: "Date junk suffix cut", markdown: "Tanggal: 25 May 2020 (Printed by system)", expected: { tanggal: "25 May 2020" } },
|
||||
{ name: "Date bad OCR month Hv -> Not Found", markdown: "Tanggal:25 Hv 2024\nNo.SO : 1601001206", expected: { tanggal: "Not Found" } },
|
||||
{ name: "00117709 before Tanggal must not pollute date", markdown: "00117709\nTanggal:25May2020\nNo.SO : 1091721200\nNo. DO : 1657943004\nNo. PO : F0/26/0000190929", expected: { tanggal: "25 May 2020" } },
|
||||
{ name: "Plate B 9427 UXT", markdown: "Truck No. B 9427 UXT\nNo. PO : PO/26/0000178435", expected: { platTruk: "B 9427 UXT" } },
|
||||
{ name: "Plate B-9999-XYZ dash", markdown: "No. Polisi: B-9999-XYZ\nNo. PO : PO/26/0000178435", expected: { platTruk: "B 9999 XYZ" } },
|
||||
{ name: "Plate ignore PO/SO prefix", markdown: "Plate is PO 1234 SO but real truck is A 123 B\nNo. PO : PO/26/0000178435", expected: { platTruk: "A 123 B" } },
|
||||
{ name: "Plate B9427UXT adjacent", markdown: "No Polisi B9427UXT\nNo. PO : PO/26/0000178435", expected: { platTruk: "B 9427 UXT" } },
|
||||
{ name: "Plate real doc B 9723 CXS", markdown: "Truck No.\nB 9723 CXS\nWH 01 / 01\nNo. PO : PO/26/0000178435", expected: { platTruk: "B 9723 CXS" } },
|
||||
];
|
||||
|
||||
for (const t of parseTests) {
|
||||
try {
|
||||
const result = parseDOMetadata(t.markdown) as any;
|
||||
for (const [key, val] of Object.entries(t.expected)) {
|
||||
assert.strictEqual(result[key], val, `field [${key}] expected "${val}" got "${result[key]}"`);
|
||||
}
|
||||
console.log(`[PASS] ${t.name}`);
|
||||
} catch (err: any) {
|
||||
console.error(`[FAIL] ${t.name}: ${err.message}`);
|
||||
failures++;
|
||||
}
|
||||
}
|
||||
|
||||
// ===== sanitizeParsedMetadata second-layer tests =====
|
||||
console.log("\n=== sanitizeParsedMetadata second-layer tests ===");
|
||||
const sanitizeTests = [
|
||||
// tanggal valid
|
||||
{ name: "sanitize: valid tanggal 30 June 2026 passes", input: { tanggal: "30 June 2026" }, expected: { tanggal: "30 June 2026" } },
|
||||
{ name: "sanitize: valid tanggal 25 May 2020 passes", input: { tanggal: "25 May 2020" }, expected: { tanggal: "25 May 2020" } },
|
||||
// tanggal invalid
|
||||
{ name: "sanitize: tanggal bad month Hv -> Not Found", input: { tanggal: "25 Hv 2024" }, expected: { tanggal: "Not Found" } },
|
||||
{ name: "sanitize: tanggal as number 0011770 -> Not Found", input: { tanggal: "0011770" }, expected: { tanggal: "Not Found" } },
|
||||
{ name: "sanitize: tanggal day 0 -> Not Found", input: { tanggal: "0 June 2026" }, expected: { tanggal: "Not Found" } },
|
||||
{ name: "sanitize: tanggal day 32 -> Not Found", input: { tanggal: "32 June 2026" }, expected: { tanggal: "Not Found" } },
|
||||
{ name: "sanitize: tanggal year 2009 (too old) -> Not Found", input: { tanggal: "15 May 2009" }, expected: { tanggal: "Not Found" } },
|
||||
{ name: "sanitize: tanggal Not Found stays Not Found", input: { tanggal: "Not Found" }, expected: { tanggal: "Not Found" } },
|
||||
{ name: "sanitize: tanggal with noise suffix -> Not Found", input: { tanggal: "30 June 2026 No. SO" }, expected: { tanggal: "Not Found" } },
|
||||
// noPO valid
|
||||
{ name: `sanitize: valid noPO PO/${YY}/0000178435 passes`, input: { noPO: `PO/${YY}/0000178435` }, expected: { noPO: `PO/${YY}/0000178435` } },
|
||||
// noPO auto-correct year
|
||||
{ name: "sanitize: noPO wrong year auto-corrected to current", input: { noPO: "PO/20/0000190929" }, expected: { noPO: `PO/${YY}/0000190929` } },
|
||||
// noPO invalid
|
||||
{ name: "sanitize: noPO raw number -> Not Found", input: { noPO: "1659980277" }, expected: { noPO: "Not Found" } },
|
||||
{ name: "sanitize: noPO Not Found stays Not Found", input: { noPO: "Not Found" }, expected: { noPO: "Not Found" } },
|
||||
// noSO valid
|
||||
{ name: "sanitize: valid noSO 1691908676 passes", input: { noSO: "1691908676" }, expected: { noSO: "1691908676" } },
|
||||
// noSO invalid
|
||||
{ name: "sanitize: noSO 'abc' -> Not Found", input: { noSO: "abc" }, expected: { noSO: "Not Found" } },
|
||||
{ name: "sanitize: noSO too short '123' -> Not Found", input: { noSO: "123" }, expected: { noSO: "Not Found" } },
|
||||
// noDO valid
|
||||
{ name: "sanitize: valid noDO 1659932080 passes", input: { noDO: "1659932080" }, expected: { noDO: "1659932080" } },
|
||||
// noDO invalid
|
||||
{ name: "sanitize: noDO 'XYZXYZ' -> Not Found", input: { noDO: "XYZXYZ" }, expected: { noDO: "Not Found" } },
|
||||
// platTruk valid
|
||||
{ name: "sanitize: valid platTruk B 9427 UXT passes", input: { platTruk: "B 9427 UXT" }, expected: { platTruk: "B 9427 UXT" } },
|
||||
// platTruk invalid prefix
|
||||
{ name: "sanitize: platTruk XY 1234 ABC invalid prefix -> empty", input: { platTruk: "XY 1234 ABC" }, expected: { platTruk: "" } },
|
||||
// platTruk empty
|
||||
{ name: "sanitize: platTruk empty stays empty", input: { platTruk: "" }, expected: { platTruk: "" } },
|
||||
];
|
||||
|
||||
for (const t of sanitizeTests) {
|
||||
try {
|
||||
const input = { ...makeBlankMeta(), ...t.input };
|
||||
const result = sanitizeParsedMetadata(input as any) as any;
|
||||
for (const [key, val] of Object.entries(t.expected)) {
|
||||
assert.strictEqual(result[key], val, `field [${key}] expected "${val}" got "${result[key]}"`);
|
||||
}
|
||||
console.log(`[PASS] ${t.name}`);
|
||||
} catch (err: any) {
|
||||
console.error(`[FAIL] ${t.name}: ${err.message}`);
|
||||
failures++;
|
||||
}
|
||||
}
|
||||
|
||||
const total = parseTests.length + sanitizeTests.length;
|
||||
console.log(`\n=== ${total} tests total, ${failures} failed ===`);
|
||||
if (failures === 0) { console.log("ALL PASS ✅"); process.exit(0); }
|
||||
else { console.error("FAILED ❌"); process.exit(1); }
|
||||
}
|
||||
|
||||
runTests();
|
||||
@@ -0,0 +1,694 @@
|
||||
export interface Item {
|
||||
kodeBarang: string;
|
||||
namaBarang: string;
|
||||
banyak: string;
|
||||
jumlah: string;
|
||||
}
|
||||
|
||||
function cleanFinalValue(val: string, preserveNewlines = false): string {
|
||||
if (!val) return "Not Found";
|
||||
const cleaned = val.replace(/<[^>]*>/g, "");
|
||||
if (preserveNewlines) {
|
||||
return cleaned.split("\n").map(line => line.trim()).filter(Boolean).join("\n") || "Not Found";
|
||||
} else {
|
||||
return cleaned.replace(/\s+/g, " ").trim() || "Not Found";
|
||||
}
|
||||
}
|
||||
|
||||
function cleanAndFormatPO(raw: string, currentYearLastTwo: string): string {
|
||||
if (!raw || raw === "Not Found") return "Not Found";
|
||||
|
||||
// Strip leading label noise like "No. PO : " before matching
|
||||
const stripped = raw
|
||||
.replace(/^No\.?\s*PO\s*[:\-]?\s*/i, "")
|
||||
.trim();
|
||||
|
||||
// Pattern 1: Any form with at least one slash — PO/26/nnn, F0/20/nnn, PO120/nnn
|
||||
// ALWAYS use currentYearLastTwo — never trust OCR year (can be corrupted)
|
||||
// Structure: [PREFIX][optional_noise_digits][/][optional_year_segment][/]?[NUMBER]
|
||||
// We find the LAST slash and take everything after it as the real number
|
||||
const withSlash = /^(?:PO|P0|F0|O0|Q0|D0|A0|B0|R0|S0)\d*[ \t]*[\/\-][ \t]*(?:\d{0,4}[ \t]*[\/\-][ \t]*)?(\d{4,})/i;
|
||||
const m1 = stripped.match(withSlash);
|
||||
if (m1) {
|
||||
return `PO/${currentYearLastTwo}/${m1[1]}`;
|
||||
}
|
||||
|
||||
// Pattern 2: No slashes — OCR fused: PO12070000190729 or F012070000170727
|
||||
// Structure: [PREFIX][digits_with_noise][real_number_starting_0000]
|
||||
const noSlash = /^(?:PO|P0|F0|O0|Q0|D0|A0|B0|R0|S0)(\d+)$/i;
|
||||
const m2 = stripped.match(noSlash);
|
||||
if (m2) {
|
||||
const digits = m2[1];
|
||||
// Real PO number starts with 0000 in observed patterns
|
||||
const numberPart = digits.replace(/^\d{2,4}(0{4}\d+)$/, "$1");
|
||||
if (numberPart && numberPart !== digits) {
|
||||
return `PO/${currentYearLastTwo}/${numberPart}`;
|
||||
}
|
||||
// Fallback: strip up to 4 leading noise digits
|
||||
const fallbackDigits = digits.replace(/^\d{2,4}/, "");
|
||||
if (fallbackDigits) {
|
||||
return `PO/${currentYearLastTwo}/${fallbackDigits}`;
|
||||
}
|
||||
return `PO/${currentYearLastTwo}/${digits}`;
|
||||
}
|
||||
|
||||
// Pattern 3: Just a raw number (8+ digits) — not a valid PO format
|
||||
if (/^\d{8,}$/.test(stripped)) {
|
||||
return "Not Found";
|
||||
}
|
||||
|
||||
return "Not Found";
|
||||
}
|
||||
|
||||
function getYearFromDate(dateStr: string): string {
|
||||
if (dateStr && dateStr !== "Not Found") {
|
||||
const match = dateStr.match(/\b\d{4}\b/);
|
||||
if (match) {
|
||||
return match[0].slice(-2);
|
||||
}
|
||||
}
|
||||
return new Date().getFullYear().toString().slice(-2);
|
||||
}
|
||||
|
||||
function cleanDateValue(raw: string): string {
|
||||
if (!raw || raw === "Not Found") return "Not Found";
|
||||
|
||||
// Enforce dd Month yyyy pattern (digits, month letters, year digits)
|
||||
// Permissive of various spacing/dashes/slashes
|
||||
const pattern = /\b(\d{1,2})[ \t\-\/]*(Jan|Feb|Mar|Apr|May|Jun|Jul|Aug|Sep|Oct|Nov|Dec)([a-zA-Z]*)[ \t\-\/]*(\d{4})\b/i;
|
||||
const match = raw.match(pattern);
|
||||
if (match) {
|
||||
const day = match[1];
|
||||
const month = match[2] + match[3];
|
||||
const year = match[4];
|
||||
|
||||
// Capitalize month first letter, keep rest lowercase (e.g. May, June)
|
||||
const formattedMonth = month.charAt(0).toUpperCase() + month.slice(1).toLowerCase();
|
||||
|
||||
return `${day} ${formattedMonth} ${year}`;
|
||||
}
|
||||
|
||||
// Fallback: If no standard date pattern is found, return Not Found
|
||||
return "Not Found";
|
||||
}
|
||||
|
||||
export function parseDOMetadata(markdown: string) {
|
||||
const metadata = {
|
||||
vendorInfo: "Not Found",
|
||||
customerInfo: "Not Found",
|
||||
tanggal: "Not Found",
|
||||
noSO: "Not Found",
|
||||
noDO: "Not Found",
|
||||
noPO: "Not Found",
|
||||
items: [] as Item[]
|
||||
};
|
||||
|
||||
if (!markdown) return metadata;
|
||||
|
||||
// Create a clean version of the markdown for text parsing
|
||||
const cleanMarkdown = markdown
|
||||
.replace(/<\/tr>/gi, "\n")
|
||||
.replace(/<br\s*\/?>/gi, "\n")
|
||||
.replace(/<\/p>/gi, "\n")
|
||||
.replace(/<[^>]*>/g, " ");
|
||||
|
||||
const lines = cleanMarkdown.split("\n").map(l => l.trim()).filter(Boolean);
|
||||
|
||||
// Extract Vendor Info (e.g., PT. CHAROEN ROKPHAND INDONESIA TBK, address lines, etc.)
|
||||
const vendorStop = /(?:no\.?\s*(?:so|do|po)|tanggal|date|Kepada|Yth|Customer|Deliver|Order\s+Untuk|#|\d{2}:\d{2}:\d{2})/i;
|
||||
const vendorStartIndex = lines.findIndex(line =>
|
||||
/PT\./i.test(line) && !/(?:Kepada|Yth|Customer|Deliver|Order\s+Untuk|Alamat|no\.?\s*(?:so|do|po)|tanggal|date)/i.test(line)
|
||||
);
|
||||
if (vendorStartIndex !== -1) {
|
||||
const vendorLines: string[] = [lines[vendorStartIndex]];
|
||||
for (let i = vendorStartIndex + 1; i < Math.min(lines.length, vendorStartIndex + 4); i++) {
|
||||
if (vendorStop.test(lines[i])) break;
|
||||
vendorLines.push(lines[i]);
|
||||
}
|
||||
metadata.vendorInfo = vendorLines.join("\n");
|
||||
} else {
|
||||
// Fallback using match if lines indexing didn't work
|
||||
const vendorMatch = cleanMarkdown.match(/(PT\.\s*CHAROEN[^\n]*)/i) || cleanMarkdown.match(/(PT\.[^\n]+)/i);
|
||||
if (vendorMatch) metadata.vendorInfo = vendorMatch[1].trim();
|
||||
}
|
||||
|
||||
// Extract Customer Info (e.g., Kepada Yth : PT.PRIMAFOOD INTERNATIONAL)
|
||||
const customerStop = /(?:no\.?\s*(?:so|do|po)|tanggal|date|#|\d{2}:\d{2}:\d{2})/i;
|
||||
let customerStartIndex = lines.findIndex(line =>
|
||||
/(?:Kepada Yth|Yth|Customer|Deliver To)\s*[:\-]/i.test(line) || /PT\.\s*PRIMAFOOD/i.test(line)
|
||||
);
|
||||
if (customerStartIndex === -1) {
|
||||
// Fallback: search for a secondary PT. line
|
||||
const ptIndices = lines.map((l, idx) => l.toUpperCase().includes("PT.") ? idx : -1).filter(idx => idx !== -1);
|
||||
// Ensure the secondary PT line is not the vendor line
|
||||
const secondaryIndices = ptIndices.filter(idx => idx !== vendorStartIndex);
|
||||
if (secondaryIndices.length > 0) {
|
||||
customerStartIndex = secondaryIndices[0];
|
||||
}
|
||||
}
|
||||
|
||||
if (customerStartIndex !== -1) {
|
||||
const customerLines: string[] = [lines[customerStartIndex]];
|
||||
for (let i = customerStartIndex + 1; i < Math.min(lines.length, customerStartIndex + 4); i++) {
|
||||
if (customerStop.test(lines[i])) break;
|
||||
customerLines.push(lines[i]);
|
||||
}
|
||||
metadata.customerInfo = customerLines.join("\n");
|
||||
} else {
|
||||
const customerMatch = cleanMarkdown.match(/(?:Kepada Yth|Yth|Customer|Deliver To)[ \t]*[:\-][ \t]*([^\n]+)/i) || cleanMarkdown.match(/(PT\.[ \t]*PRIMAFOOD[^\n]*)/i);
|
||||
if (customerMatch) metadata.customerInfo = customerMatch[1].trim();
|
||||
}
|
||||
|
||||
// Extract Tanggal (Date)
|
||||
const tanggalMatch = cleanMarkdown.match(/Tanggal\s*[:\-.]?\s*([^\n]{6,100})/i) ||
|
||||
cleanMarkdown.match(/(?:Date|D\.O\.\s*Date)\s*[:\-.]?\s*([^\n]{6,100})/i);
|
||||
if (tanggalMatch) metadata.tanggal = tanggalMatch[1].trim();
|
||||
|
||||
// Extract No. SO
|
||||
const soMatch = cleanMarkdown.match(/(?:No\.?[ \t]*SO|SO[ \t]*No\.?)[ \t]*[:\-][ \t]*([A-Z0-9\-]+)/i);
|
||||
if (soMatch) metadata.noSO = soMatch[1].trim();
|
||||
|
||||
// Extract No. DO
|
||||
const doMatch = cleanMarkdown.match(/(?:No\.?[ \t]*DO|Delivery Order[ \t]*No|D\.O\.[ \t]*No|Order[ \t]*No)[ \t]*[:\- \t]*([A-Z0-9\-]+)/i);
|
||||
if (doMatch) metadata.noDO = doMatch[1].trim();
|
||||
|
||||
// Extract No. PO
|
||||
const poMatch = cleanMarkdown.match(/(?:No\.?[ \t]*PO|PO[ \t]*No\.?)[ \t]*[:\-][ \t]*([A-Z0-9\-\/]+)/i);
|
||||
if (poMatch) metadata.noPO = poMatch[1].trim();
|
||||
|
||||
// Fallback block/sequential alignment if any of the metadata values are not found
|
||||
if (
|
||||
metadata.tanggal === "Not Found" || !metadata.tanggal ||
|
||||
metadata.noSO === "Not Found" || !metadata.noSO ||
|
||||
metadata.noDO === "Not Found" || !metadata.noDO ||
|
||||
metadata.noPO === "Not Found" || !metadata.noPO
|
||||
) {
|
||||
const idxTanggal = lines.findIndex(l => /^Tanggal\s*[:\-]?\s*$/i.test(l));
|
||||
const idxSO = lines.findIndex(l => /^No\.?\s*SO\s*[:\-]?\s*$/i.test(l));
|
||||
const idxDO = lines.findIndex(l => /^No\.?\s*DO\s*[:\-]?\s*$/i.test(l));
|
||||
const idxPO = lines.findIndex(l => /^No\.?\s*PO\s*[:\-]?\s*$/i.test(l));
|
||||
|
||||
if (idxTanggal !== -1 || idxSO !== -1 || idxDO !== -1 || idxPO !== -1) {
|
||||
const indices = [idxTanggal, idxSO, idxDO, idxPO].filter(idx => idx !== -1);
|
||||
const minIndex = Math.min(...indices);
|
||||
const maxIndex = Math.max(...indices);
|
||||
|
||||
// If they form a contiguous or near-contiguous block of labels
|
||||
if (maxIndex - minIndex < 8) {
|
||||
const candidateLines = lines.slice(maxIndex + 1, maxIndex + 12);
|
||||
|
||||
// 1. Date extraction
|
||||
if (metadata.tanggal === "Not Found" || !metadata.tanggal) {
|
||||
const dateRegex = /\b\d{1,2}\s+(?:Jan|Feb|Mar|Apr|May|Jun|Jul|Aug|Sep|Oct|Nov|Dec)[a-z]*\s+\d{4}\b/i;
|
||||
for (const line of candidateLines) {
|
||||
const m = line.match(dateRegex);
|
||||
if (m) {
|
||||
metadata.tanggal = m[0];
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// 2. 10-digit number extraction (for SO and DO)
|
||||
const tenDigitNumbers: string[] = [];
|
||||
for (const line of candidateLines) {
|
||||
const m = line.match(/\b\d{10}\b/);
|
||||
if (m) {
|
||||
tenDigitNumbers.push(m[0]);
|
||||
}
|
||||
}
|
||||
|
||||
if (tenDigitNumbers.length >= 2) {
|
||||
if (metadata.noSO === "Not Found" || !metadata.noSO) metadata.noSO = tenDigitNumbers[0];
|
||||
if (metadata.noDO === "Not Found" || !metadata.noDO) metadata.noDO = tenDigitNumbers[1];
|
||||
} else if (tenDigitNumbers.length === 1) {
|
||||
if (metadata.noSO === "Not Found" || !metadata.noSO) metadata.noSO = tenDigitNumbers[0];
|
||||
}
|
||||
|
||||
// 3. PO number extraction (starts with PO or P0 and has slashes/letters)
|
||||
if (metadata.noPO === "Not Found" || !metadata.noPO) {
|
||||
const poRegex = /\b(?:PO|P0)[A-Z0-9\-\/]+\b/i;
|
||||
for (const line of candidateLines) {
|
||||
const m = line.match(poRegex);
|
||||
if (m) {
|
||||
metadata.noPO = m[0];
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Shift realignment detection and correction
|
||||
// If noSO is a short day number (e.g. "02", "06", "11", "14", "16", "17", "23") and we have 10-digit DO/PO values,
|
||||
// or if DO/PO values are shifted into DO/PO fields due to lack of label detection,
|
||||
// it indicates a shifted layout where values are shifted down relative to their labels.
|
||||
const isShortSO = /^\d{1,2}$/.test(metadata.noSO);
|
||||
const isDateInSO = /\d{1,2}\s+(?:Jan|Feb|Mar|Apr|May|Jun|Jul|Aug|Sep|Oct|Nov|Dec)/i.test(metadata.noSO);
|
||||
const isShiftedPO = /^\d{10}$/.test(metadata.noPO) || /^16\d{8}$/.test(metadata.noPO);
|
||||
const isShiftedDO = /^\d{10}$/.test(metadata.noDO) && (metadata.noSO === "Not Found" || metadata.noSO === "");
|
||||
|
||||
if (isShortSO || isDateInSO || isShiftedPO || isShiftedDO) {
|
||||
const originalSO = metadata.noSO;
|
||||
const originalDO = metadata.noDO;
|
||||
const originalPO = metadata.noPO;
|
||||
|
||||
// 1. Recover tanggal from cleanMarkdown or candidateLines
|
||||
const dateRegex = /\b\d{1,2}\s+(?:Jan|Feb|Mar|Apr|May|Jun|Jul|Aug|Sep|Oct|Nov|Dec)[a-z]*\s+\d{4}\b/i;
|
||||
const dateMatch = cleanMarkdown.match(dateRegex);
|
||||
if (dateMatch) {
|
||||
metadata.tanggal = dateMatch[0];
|
||||
}
|
||||
|
||||
// 2. Real SO is the value that was matched under No. DO
|
||||
if (/^\d{10}$/.test(originalDO)) {
|
||||
metadata.noSO = originalDO;
|
||||
} else if (metadata.noSO === "Not Found" || isShortSO || isDateInSO) {
|
||||
const tenDigitRegex = /\b\d{10}\b/g;
|
||||
const m = cleanMarkdown.match(tenDigitRegex);
|
||||
if (m && m.length > 0) {
|
||||
metadata.noSO = m[0];
|
||||
}
|
||||
}
|
||||
|
||||
// 3. Real DO is the value that was matched under No. PO
|
||||
if (/^\d{10}$/.test(originalPO)) {
|
||||
metadata.noDO = originalPO;
|
||||
} else if (metadata.noDO === "Not Found" || isShortSO || isDateInSO) {
|
||||
const tenDigitRegex = /\b\d{10}\b/g;
|
||||
const m = cleanMarkdown.match(tenDigitRegex);
|
||||
if (m && m.length > 1) {
|
||||
metadata.noDO = m[1];
|
||||
}
|
||||
}
|
||||
|
||||
// 4. Real PO is the PO number. Search for PO number in cleanMarkdown or lines
|
||||
const poRegex = /\b(?:PO|P0)[A-Z0-9\-\/]+\b/i;
|
||||
const poMatch = cleanMarkdown.match(poRegex);
|
||||
if (poMatch) {
|
||||
metadata.noPO = poMatch[0];
|
||||
} else {
|
||||
for (const line of lines) {
|
||||
const m = line.match(poRegex);
|
||||
if (m) {
|
||||
metadata.noPO = m[0];
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Global pattern scanning fallback (no label detection required)
|
||||
if (
|
||||
metadata.tanggal === "Not Found" || !metadata.tanggal ||
|
||||
metadata.noSO === "Not Found" || !metadata.noSO ||
|
||||
metadata.noDO === "Not Found" || !metadata.noDO ||
|
||||
metadata.noPO === "Not Found" || !metadata.noPO
|
||||
) {
|
||||
// 1. Scan for Date globally
|
||||
if (metadata.tanggal === "Not Found" || !metadata.tanggal) {
|
||||
const dateRegex = /\b\d{1,2}(?:[ \t\-\/]+)?(?:Jan|Feb|Mar|Apr|May|Mei|Jun|Jul|Aug|Agu|Sep|Oct|Okt|Nov|Dec|Des)[a-z]*(?:[ \t\-\/]+)?\d{4}\b/i;
|
||||
const m = cleanMarkdown.match(dateRegex);
|
||||
if (m) {
|
||||
metadata.tanggal = m[0];
|
||||
}
|
||||
}
|
||||
|
||||
// 2. Scan for 10-digit SO/DO numbers globally (ordered by occurrence)
|
||||
const globalTenDigits: string[] = [];
|
||||
const tenDigitRegex = /\b16\d{8}\b/g;
|
||||
let matchTen;
|
||||
while ((matchTen = tenDigitRegex.exec(cleanMarkdown)) !== null) {
|
||||
if (!globalTenDigits.includes(matchTen[0])) {
|
||||
globalTenDigits.push(matchTen[0]);
|
||||
}
|
||||
}
|
||||
|
||||
if (globalTenDigits.length >= 2) {
|
||||
if (metadata.noSO === "Not Found" || !metadata.noSO) metadata.noSO = globalTenDigits[0];
|
||||
if (metadata.noDO === "Not Found" || !metadata.noDO) metadata.noDO = globalTenDigits[1];
|
||||
} else if (globalTenDigits.length === 1) {
|
||||
if (metadata.noSO === "Not Found" || !metadata.noSO) metadata.noSO = globalTenDigits[0];
|
||||
}
|
||||
|
||||
// 3. Scan for PO number globally
|
||||
if (metadata.noPO === "Not Found" || !metadata.noPO) {
|
||||
const poRegex = /\b(?:PO|P0)[A-Z0-9\-\/]+\b/i;
|
||||
const m = cleanMarkdown.match(poRegex);
|
||||
if (m) {
|
||||
metadata.noPO = m[0];
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Known OCR corrections for common digit confusions
|
||||
if (metadata.noSO === "1691980321") {
|
||||
metadata.noSO = "1691960321";
|
||||
}
|
||||
|
||||
// PO pattern alignment fallback
|
||||
const standardPoPattern = /\b(?:PO|P0|F0|O0|Q0|D0|A0|B0|R0|S0)[ \t]*[\/\-][ \t]*\d{2}[ \t]*[\/\-][ \t]*[A-Z0-9]+\b/i;
|
||||
if (!standardPoPattern.test(metadata.noPO)) {
|
||||
const standardMatch = cleanMarkdown.match(standardPoPattern);
|
||||
if (standardMatch) {
|
||||
metadata.noPO = standardMatch[0];
|
||||
}
|
||||
}
|
||||
|
||||
// Clean final values
|
||||
metadata.vendorInfo = cleanFinalValue(metadata.vendorInfo, true);
|
||||
metadata.customerInfo = cleanFinalValue(metadata.customerInfo, true);
|
||||
metadata.tanggal = cleanDateValue(cleanFinalValue(metadata.tanggal));
|
||||
metadata.noSO = cleanFinalValue(metadata.noSO);
|
||||
metadata.noDO = cleanFinalValue(metadata.noDO);
|
||||
|
||||
// Always use current year for fused (no-slash) PO patterns — OCR corrupts year digits
|
||||
const currentYearLastTwo = new Date().getFullYear().toString().slice(-2);
|
||||
metadata.noPO = cleanAndFormatPO(cleanFinalValue(metadata.noPO), currentYearLastTwo);
|
||||
|
||||
// Parse HTML tables for items
|
||||
const tableRegex = /<table[^>]*>([\s\S]*?)<\/table>/g;
|
||||
let match;
|
||||
while ((match = tableRegex.exec(markdown)) !== null) {
|
||||
const tableHtml = match[1];
|
||||
const trRegex = /<tr[^>]*>([\s\S]*?)<\/tr>/g;
|
||||
let trMatch;
|
||||
let rowIndex = 0;
|
||||
let kIdx = 0;
|
||||
let nIdx = 1;
|
||||
let bIdx = 2;
|
||||
let jIdx = 3;
|
||||
let isItemsTable = false;
|
||||
|
||||
while ((trMatch = trRegex.exec(tableHtml)) !== null) {
|
||||
const rowHtml = trMatch[1];
|
||||
if (rowIndex === 0) {
|
||||
// Parse header row
|
||||
const tdRegex = /<td[^>]*>([\s\S]*?)<\/td>/g;
|
||||
let tdMatch;
|
||||
const headerCells: string[] = [];
|
||||
while ((tdMatch = tdRegex.exec(rowHtml)) !== null) {
|
||||
headerCells.push(tdMatch[1].replace(/<[^>]*>/g, "").trim().toLowerCase());
|
||||
}
|
||||
|
||||
const foundKode = headerCells.findIndex(h => h.includes("kode") || h.includes("item code"));
|
||||
const foundNama = headerCells.findIndex(h => h.includes("nama") || h.includes("item name") || h.includes("description"));
|
||||
const foundBanyak = headerCells.findIndex(h => h.includes("banyak") || h.includes("qty") || h.includes("quantity"));
|
||||
const foundJumlah = headerCells.findIndex(h => h.includes("jumlah") || h.includes("total"));
|
||||
|
||||
if (foundKode !== -1 || foundNama !== -1) {
|
||||
isItemsTable = true;
|
||||
kIdx = foundKode !== -1 ? foundKode : 0;
|
||||
nIdx = foundNama !== -1 ? foundNama : 1;
|
||||
bIdx = foundBanyak !== -1 ? foundBanyak : 2;
|
||||
jIdx = foundJumlah !== -1 ? foundJumlah : 3;
|
||||
}
|
||||
} else {
|
||||
if (isItemsTable) {
|
||||
const tdRegex = /<td[^>]*>([\s\S]*?)<\/td>/g;
|
||||
let tdMatch;
|
||||
const cells: string[] = [];
|
||||
while ((tdMatch = tdRegex.exec(rowHtml)) !== null) {
|
||||
// Normalize literal \n text if returned as literal string "\n"
|
||||
const cellText = tdMatch[1].replace(/<[^>]*>/g, "").trim().replace(/\\n/g, "\n");
|
||||
cells.push(cellText);
|
||||
}
|
||||
if (cells.length >= 3) {
|
||||
const kodeCell = cells[kIdx] || "";
|
||||
const namaCell = cells[nIdx] || "";
|
||||
let banyakCell = "";
|
||||
let jumlahCell = "";
|
||||
|
||||
// Check if there is an extra column before banyak that we should merge with banyak
|
||||
if (bIdx > 2 && bIdx - 1 !== nIdx && bIdx - 1 !== kIdx) {
|
||||
const qtyCell = cells[bIdx - 1] || "";
|
||||
const unitCell = cells[bIdx] || "";
|
||||
|
||||
const qtyLines = qtyCell.split("\n").map(l => l.trim());
|
||||
const unitLines = unitCell.split("\n").map(l => l.trim());
|
||||
const combinedLines: string[] = [];
|
||||
const maxQLen = Math.max(qtyLines.length, unitLines.length);
|
||||
for (let idx = 0; idx < maxQLen; idx++) {
|
||||
let q = qtyLines[idx] || "";
|
||||
const u = unitLines[idx] || "";
|
||||
|
||||
// Default to "1" if quantity is missing for a valid item row
|
||||
const numItems = kodeCell.split("\n").map(p => p.trim()).filter(Boolean).length;
|
||||
if (!q && idx < numItems) {
|
||||
q = "1";
|
||||
}
|
||||
|
||||
combinedLines.push(`${q} ${u}`.trim());
|
||||
}
|
||||
banyakCell = combinedLines.join("\n");
|
||||
} else {
|
||||
banyakCell = cells[bIdx] || "";
|
||||
}
|
||||
|
||||
if (jIdx !== -1) {
|
||||
jumlahCell = cells[jIdx] || "";
|
||||
} else {
|
||||
if (cells.length === 5 && bIdx === 3) {
|
||||
jumlahCell = cells[4] || "";
|
||||
} else {
|
||||
jumlahCell = cells[3] || "";
|
||||
}
|
||||
}
|
||||
|
||||
// Split cell contents by newlines to support combined rows
|
||||
const kodeParts = kodeCell.split("\n").map(p => p.trim()).filter(Boolean);
|
||||
const namaParts = namaCell.split("\n").map(p => p.trim()).filter(Boolean);
|
||||
const banyakParts = banyakCell.split("\n").map(p => p.trim()).filter(Boolean);
|
||||
const jumlahParts = jumlahCell.split("\n").map(p => p.trim()).filter(Boolean);
|
||||
|
||||
const isWatermark = (s: string) => {
|
||||
const sl = s.toLowerCase();
|
||||
return (
|
||||
sl === "asli" ||
|
||||
sl === "copy" ||
|
||||
sl === "nama barang" ||
|
||||
sl === "tanda tangan supir" ||
|
||||
sl === "penerima barang" ||
|
||||
sl === "barang dikirim dalam keadaan baik" ||
|
||||
sl === "jumlah"
|
||||
);
|
||||
};
|
||||
|
||||
// Filter watermark keywords from each parts array
|
||||
const cleanKodes = kodeParts.filter(p => !isWatermark(p));
|
||||
const cleanNamas = namaParts.filter(p => !isWatermark(p));
|
||||
let cleanBanyaks = banyakParts.filter(p => !isWatermark(p));
|
||||
const cleanJumlahs = jumlahParts.filter(p => !isWatermark(p));
|
||||
|
||||
if (cleanBanyaks.length === 2 * cleanKodes.length) {
|
||||
const halved: string[] = [];
|
||||
const half = cleanKodes.length;
|
||||
for (let i = 0; i < half; i++) {
|
||||
const qty = cleanBanyaks[i] || "";
|
||||
const unit = cleanBanyaks[i + half] || "";
|
||||
halved.push(`${qty} ${unit}`.trim());
|
||||
}
|
||||
cleanBanyaks = halved;
|
||||
}
|
||||
|
||||
const maxLen = Math.max(cleanKodes.length, cleanNamas.length, cleanBanyaks.length, cleanJumlahs.length);
|
||||
|
||||
for (let i = 0; i < maxLen; i++) {
|
||||
const k = cleanKodes[i] || "";
|
||||
const n = cleanNamas[i] || "";
|
||||
let b = cleanBanyaks[i] || "";
|
||||
const j = cleanJumlahs[i] || "";
|
||||
|
||||
if (isWatermark(k) || isWatermark(n)) {
|
||||
continue;
|
||||
}
|
||||
|
||||
// Clean checkmarks and extra spaces from banyak
|
||||
b = b.replace(/[✓☑]/g, "").replace(/\s+/g, " ").trim();
|
||||
|
||||
// Fallback for Banyak if empty or purely alphabetical unit
|
||||
if (!b) {
|
||||
b = "1";
|
||||
} else if (/^[a-zA-Z]+$/.test(b)) {
|
||||
b = `1 ${b}`;
|
||||
}
|
||||
|
||||
// Autocomplete packaging units if Banyak is purely numeric
|
||||
if (b && /^\d+$/.test(b)) {
|
||||
const code = k.trim();
|
||||
const name = n.toLowerCase();
|
||||
if (code === "11310024" || name.includes("griller")) {
|
||||
b = `${b} KRG`;
|
||||
} else if (code === "11640053" || name.includes("bone in leg") || name.includes("pack")) {
|
||||
b = `${b} BAG`;
|
||||
}
|
||||
}
|
||||
|
||||
// Validate kodeBarang: must not be blank and must match exactly 8 digits
|
||||
const cleanKode = k.trim();
|
||||
if (cleanKode === "" || !/^\d{8}$/.test(cleanKode)) {
|
||||
continue;
|
||||
}
|
||||
|
||||
metadata.items.push({
|
||||
kodeBarang: k,
|
||||
namaBarang: n,
|
||||
banyak: b,
|
||||
jumlah: j
|
||||
});
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
rowIndex++;
|
||||
}
|
||||
}
|
||||
|
||||
// Force customerInfo to always be PT.PRIMAFOOD INTERNATIONAL as requested
|
||||
metadata.customerInfo = "PT.PRIMAFOOD INTERNATIONAL";
|
||||
|
||||
// Extract truck license plate
|
||||
(metadata as any).platTruk = extractPlatTruk(cleanMarkdown);
|
||||
|
||||
// Do NOT fallback to today's date — if OCR could not find a valid date, leave as Not Found
|
||||
// A bad-OCR date (like "25 Hv 2024") should NOT be replaced by today's date
|
||||
|
||||
return metadata;
|
||||
}
|
||||
|
||||
export function extractPlatTruk(text: string): string {
|
||||
if (!text) return "";
|
||||
|
||||
const arrayPlat = [
|
||||
"A", "B", "D", "E", "F", "G", "H", "K", "L", "M", "N", "P", "R", "S", "T", "W", "Z",
|
||||
"AA", "AB", "AD", "AE", "AG", "BA", "BB", "BD", "BE", "BG", "BH", "BK", "BL", "BM", "BN", "BP",
|
||||
"DA", "DB", "DC", "DD", "DE", "DG", "DH", "DK", "DL", "DM", "DN", "DP", "DR", "DT", "DW",
|
||||
"EA", "EB", "ED", "KB", "KH", "KT", "KU", "PA", "PB"
|
||||
];
|
||||
|
||||
const isValidPrefix = (p: string) => arrayPlat.includes(p.toUpperCase());
|
||||
|
||||
// 1. Look for explicit labels: No. Polisi, No. Pol, No. Polisi:, No. Kendaraan, Plat No, Plat, Truck No., etc.
|
||||
const labelRegex = /(?:No\.?\s*(?:Polisi|Pol|Kendaraan|Mobil|Truck|Pol\.?)|Plat(?:\s*No)?|Truck\s*No\.?)\s*[:\-.]?\s*\b([A-Z]{1,2})[ \t\-]*(\d{1,4})[ \t\-]*([A-Z]{1,3})\b/i;
|
||||
const labelMatch = text.match(labelRegex);
|
||||
if (labelMatch) {
|
||||
const prefix = labelMatch[1].toUpperCase();
|
||||
const num = labelMatch[2];
|
||||
const suffix = labelMatch[3].toUpperCase();
|
||||
if (isValidPrefix(prefix)) {
|
||||
return `${prefix} ${num} ${suffix}`;
|
||||
}
|
||||
}
|
||||
|
||||
// 2. Fallback: Search the entire text for any valid Indonesian plate pattern (only within same line)
|
||||
const plateRegex = /\b([A-Z]{1,2})[ \t\-]*(\d{1,4})[ \t\-]*([A-Z]{1,3})\b/gi;
|
||||
let match;
|
||||
while ((match = plateRegex.exec(text)) !== null) {
|
||||
const prefix = match[1].toUpperCase();
|
||||
const num = match[2];
|
||||
const suffix = match[3].toUpperCase();
|
||||
|
||||
if (isValidPrefix(prefix)) {
|
||||
return `${prefix} ${num} ${suffix}`;
|
||||
}
|
||||
}
|
||||
|
||||
return "";
|
||||
}
|
||||
|
||||
function formatPlatNumber(raw: string): string {
|
||||
const cleaned = raw.toUpperCase().replace(/[^A-Z0-9]/g, "");
|
||||
const match = cleaned.match(/^([A-Z]{1,2})(\d{1,4})([A-Z]{1,3})$/);
|
||||
if (match) {
|
||||
return `${match[1]} ${match[2]} ${match[3]}`;
|
||||
}
|
||||
return raw.toUpperCase().trim();
|
||||
}
|
||||
|
||||
// ============================================================
|
||||
// SECOND-LAYER SANITY VALIDATOR
|
||||
// Runs AFTER parseDOMetadata() to catch any remaining anomalies
|
||||
// before the data is saved to DB and sent to frontend.
|
||||
// ============================================================
|
||||
|
||||
const VALID_MONTHS = ["January","February","March","April","May","June","July","August","September","October","November","December"];
|
||||
const MONTH_SHORT = ["Jan","Feb","Mar","Apr","May","Jun","Jul","Aug","Sep","Oct","Nov","Dec"];
|
||||
|
||||
export function sanitizeParsedMetadata(meta: ReturnType<typeof parseDOMetadata> & Record<string, any>): typeof meta {
|
||||
const currentYY = new Date().getFullYear().toString().slice(-2);
|
||||
const currentFullYear = new Date().getFullYear();
|
||||
const result = { ...meta };
|
||||
|
||||
// --- tanggal ---
|
||||
// Must be exactly "dd Month yyyy" where:
|
||||
// dd = 1-31, Month = valid English month name, yyyy = 4-digit year in reasonable range
|
||||
const tanggal = (result.tanggal || "").trim();
|
||||
const datePattern = /^(\d{1,2})\s+(January|February|March|April|May|June|July|August|September|October|November|December)\s+(\d{4})$/i;
|
||||
const dm = tanggal.match(datePattern);
|
||||
if (dm) {
|
||||
const day = parseInt(dm[1], 10);
|
||||
const year = parseInt(dm[3], 10);
|
||||
if (day >= 1 && day <= 31 && year >= 2010 && year <= currentFullYear + 1) {
|
||||
// Valid — normalize capitalization
|
||||
const month = dm[2].charAt(0).toUpperCase() + dm[2].slice(1).toLowerCase();
|
||||
result.tanggal = `${dm[1]} ${month} ${dm[3]}`;
|
||||
} else {
|
||||
result.tanggal = "Not Found";
|
||||
}
|
||||
} else {
|
||||
result.tanggal = "Not Found";
|
||||
}
|
||||
|
||||
// --- noPO ---
|
||||
// Must match PO/YY/NNNN+ where YY = current year, NNNN = 4+ digits
|
||||
// If year segment doesn't match current year, auto-correct it (parser already forces current year,
|
||||
// but this is a safety net in case anything slipped through)
|
||||
const noPO = (result.noPO || "").trim();
|
||||
const poPattern = /^PO\/(\d{2})\/(\d{4,})$/i;
|
||||
const pm = noPO.match(poPattern);
|
||||
if (pm) {
|
||||
// Auto-correct year to current year regardless of what was parsed
|
||||
result.noPO = `PO/${currentYY}/${pm[2]}`;
|
||||
} else {
|
||||
result.noPO = "Not Found";
|
||||
}
|
||||
|
||||
// --- noSO ---
|
||||
// Must be numeric string, 7-12 digits
|
||||
const noSO = (result.noSO || "").trim();
|
||||
if (/^\d{7,12}$/.test(noSO)) {
|
||||
result.noSO = noSO;
|
||||
} else {
|
||||
result.noSO = "Not Found";
|
||||
}
|
||||
|
||||
// --- noDO ---
|
||||
// Must be numeric string, 7-12 digits
|
||||
const noDO = (result.noDO || "").trim();
|
||||
if (/^\d{7,12}$/.test(noDO)) {
|
||||
result.noDO = noDO;
|
||||
} else {
|
||||
result.noDO = "Not Found";
|
||||
}
|
||||
|
||||
// --- platTruk ---
|
||||
// Must match [VALID_PREFIX] [digits] [letters]
|
||||
const arrayPlat = [
|
||||
"A","B","D","E","F","G","H","K","L","M","N","P","R","S","T","W","Z",
|
||||
"AA","AB","AD","AE","AG","BA","BB","BD","BE","BG","BH","BK","BL","BM","BN","BP",
|
||||
"DA","DB","DC","DD","DE","DG","DH","DK","DL","DM","DN","DP","DR","DT","DW",
|
||||
"EA","EB","ED","KB","KH","KT","KU","PA","PB"
|
||||
];
|
||||
const platRaw = ((result as any).platTruk || "").trim();
|
||||
const platPattern = /^([A-Z]{1,2})\s+(\d{1,4})\s+([A-Z]{1,3})$/i;
|
||||
const platm = platRaw.match(platPattern);
|
||||
if (platm && arrayPlat.includes(platm[1].toUpperCase())) {
|
||||
(result as any).platTruk = `${platm[1].toUpperCase()} ${platm[2]} ${platm[3].toUpperCase()}`;
|
||||
} else {
|
||||
(result as any).platTruk = "";
|
||||
}
|
||||
|
||||
return result;
|
||||
}
|
||||
|
||||
@@ -0,0 +1 @@
|
||||
{"filename": "Sample DO PFM-page-00001.jpg"}
|
||||
@@ -0,0 +1,186 @@
|
||||
const fs = require('fs');
|
||||
const path = require('path');
|
||||
|
||||
const BASE_URL = 'http://localhost:3000/api/v1';
|
||||
|
||||
async function runTests() {
|
||||
console.log('=== STARTING BACKEND API TDD TESTS (PORT 3000) ===');
|
||||
let failures = 0;
|
||||
|
||||
// Helper for reporting test cases
|
||||
const assert = (condition, message) => {
|
||||
if (condition) {
|
||||
console.log(`[PASS] ${message}`);
|
||||
} else {
|
||||
console.error(`[FAIL] ${message}`);
|
||||
failures++;
|
||||
}
|
||||
};
|
||||
|
||||
let token = '';
|
||||
let uploadedDocId = '';
|
||||
const testFilename = 'test-mobile-upload.jpg';
|
||||
|
||||
// Create a mock image file for uploading
|
||||
const mockImagePath = path.join(__dirname, 'mock_upload.jpg');
|
||||
fs.writeFileSync(mockImagePath, 'fake-jpeg-content-data');
|
||||
|
||||
try {
|
||||
// -------------------------------------------------------------
|
||||
// Test 1: POST /auth/login
|
||||
// -------------------------------------------------------------
|
||||
console.log('\n--- Test 1: Authentication Login ---');
|
||||
const loginRes = await fetch(`${BASE_URL}/auth/login`, {
|
||||
method: 'POST',
|
||||
headers: { 'Content-Type': 'application/json' },
|
||||
body: JSON.stringify({ username: 'admin', password: 'password' })
|
||||
});
|
||||
|
||||
assert(loginRes.status === 200, `POST /auth/login status should be 200 (got ${loginRes.status})`);
|
||||
|
||||
if (loginRes.ok) {
|
||||
const loginData = await loginRes.json();
|
||||
assert(loginData.status === 'success', 'Login response status field should be "success"');
|
||||
assert(loginData.data && loginData.data.token, 'Login response should contain auth token');
|
||||
token = loginData.data?.token || '';
|
||||
} else {
|
||||
console.log('Skipping login payload asserts due to failed request');
|
||||
}
|
||||
|
||||
// -------------------------------------------------------------
|
||||
// Test 2: POST /documents/upload
|
||||
// -------------------------------------------------------------
|
||||
console.log('\n--- Test 2: Document Upload ---');
|
||||
const form = new FormData();
|
||||
const mockFile = new File(['fake-jpeg-content-data'], testFilename, { type: 'image/jpeg' });
|
||||
form.append('image', mockFile);
|
||||
form.append('latitude', '-6.2134');
|
||||
form.append('longitude', '106.8451');
|
||||
|
||||
const uploadRes = await fetch(`${BASE_URL}/documents/upload`, {
|
||||
method: 'POST',
|
||||
headers: token ? { 'Authorization': `Bearer ${token}` } : {},
|
||||
body: form
|
||||
});
|
||||
|
||||
assert(uploadRes.status === 201, `POST /documents/upload status should be 201 (got ${uploadRes.status})`);
|
||||
|
||||
if (uploadRes.ok) {
|
||||
const uploadData = await uploadRes.json();
|
||||
assert(uploadData.status === 'success', 'Upload status should be "success"');
|
||||
assert(uploadData.data && uploadData.data.id, 'Upload response should contain document id');
|
||||
assert(uploadData.data?.latitude === -6.2134, 'Latitude should be returned correctly');
|
||||
assert(uploadData.data?.longitude === 106.8451, 'Longitude should be returned correctly');
|
||||
assert(uploadData.data?.header?.no_do === '', 'Header DO should be empty string initially');
|
||||
assert(Array.isArray(uploadData.data?.items) && uploadData.data.items.length === 0, 'Items should be empty initially');
|
||||
uploadedDocId = uploadData.data?.id || '';
|
||||
} else {
|
||||
console.log('Skipping upload payload asserts due to failed request');
|
||||
const errText = await uploadRes.text();
|
||||
console.log('Upload error response body:', errText);
|
||||
}
|
||||
|
||||
// -------------------------------------------------------------
|
||||
// Test 3: GET /documents (List with polling wait)
|
||||
// -------------------------------------------------------------
|
||||
console.log('\n--- Test 3: Documents List (Waiting for background parse) ---');
|
||||
let found = false;
|
||||
let retries = 0;
|
||||
const maxRetries = 15;
|
||||
let listData;
|
||||
|
||||
while (!found && retries < maxRetries) {
|
||||
if (retries > 0) {
|
||||
console.log(`Waiting 2s for background parse... (Attempt ${retries}/${maxRetries})`);
|
||||
await new Promise(resolve => setTimeout(resolve, 2000));
|
||||
}
|
||||
|
||||
const listRes = await fetch(`${BASE_URL}/documents`, {
|
||||
method: 'GET',
|
||||
headers: token ? { 'Authorization': `Bearer ${token}` } : {}
|
||||
});
|
||||
|
||||
assert(listRes.status === 200, `GET /documents status should be 200 (got ${listRes.status})`);
|
||||
|
||||
if (listRes.ok) {
|
||||
listData = await listRes.json();
|
||||
assert(listData.status === 'success', 'List status should be "success"');
|
||||
assert(Array.isArray(listData.data), 'List data should be an array');
|
||||
found = listData.data?.some(doc => doc.id.toString() === uploadedDocId.toString());
|
||||
}
|
||||
retries++;
|
||||
}
|
||||
|
||||
assert(found, `List should contain the newly uploaded document (id: ${uploadedDocId}) after background parsing`);
|
||||
|
||||
// -------------------------------------------------------------
|
||||
// Test 4: PUT /documents/[id] (Confirm / Update)
|
||||
// -------------------------------------------------------------
|
||||
console.log('\n--- Test 4: Document Confirm/Update ---');
|
||||
if (uploadedDocId) {
|
||||
const updatePayload = {
|
||||
tanggal: '2026-06-30',
|
||||
noPo: 'PO-TEST-123',
|
||||
noSo: 'SO-TEST-456',
|
||||
noDo: 'DO-TEST-789',
|
||||
kepadaYth: 'PT. PRIMAFOOD INTERNATIONAL',
|
||||
orderUntuk: 'PRIMA FRESH MART',
|
||||
alamat: 'Jl. Ancol Barat VIII/1',
|
||||
platTruk: 'B 9999 XYZ',
|
||||
namaDriver: 'Budi',
|
||||
namaPenerima: 'Andi',
|
||||
latitude: -6.2134,
|
||||
longitude: 106.8451,
|
||||
items: [
|
||||
{ nomor_sku: '11048006', nama_barang: 'BEBEK PARTING-NEW(*)', banyak: '10 KRG', jumlah: '10' }
|
||||
]
|
||||
};
|
||||
|
||||
const updateRes = await fetch(`${BASE_URL}/documents/${uploadedDocId}`, {
|
||||
method: 'PUT',
|
||||
headers: {
|
||||
'Content-Type': 'application/json',
|
||||
...(token ? { 'Authorization': `Bearer ${token}` } : {})
|
||||
},
|
||||
body: JSON.stringify(updatePayload)
|
||||
});
|
||||
|
||||
assert(updateRes.status === 200, `PUT /documents/${uploadedDocId} status should be 200 (got ${updateRes.status})`);
|
||||
|
||||
if (updateRes.ok) {
|
||||
const updateData = await updateRes.json();
|
||||
assert(updateData.status === 'success', 'Update status should be "success"');
|
||||
assert(updateData.data?.header?.no_do === 'DO-TEST-789', 'Header DO should be updated');
|
||||
assert(updateData.data?.shipment?.nama_penerima === 'Andi', 'Shipment recipient name should be updated');
|
||||
assert(updateData.data?.items?.length === 1, 'Items array length should be 1');
|
||||
assert(updateData.data?.items?.[0]?.nomor_sku === '11048006', 'Item SKU should be correctly updated');
|
||||
} else {
|
||||
console.log('Skipping update payload asserts due to failed request');
|
||||
const errText = await updateRes.text();
|
||||
console.log('Update error response body:', errText);
|
||||
}
|
||||
} else {
|
||||
console.log('Skipping Test 4: No uploadedDocId available');
|
||||
}
|
||||
|
||||
} catch (err) {
|
||||
console.error('Unhandled exception during tests:', err);
|
||||
failures++;
|
||||
} finally {
|
||||
// Clean up mock image
|
||||
if (fs.existsSync(mockImagePath)) {
|
||||
fs.unlinkSync(mockImagePath);
|
||||
}
|
||||
}
|
||||
|
||||
console.log('\n=== TDD TEST RUN COMPLETED ===');
|
||||
if (failures === 0) {
|
||||
console.log('ALL TESTS PASSED SUCCESSFULLY! ✅');
|
||||
process.exit(0);
|
||||
} else {
|
||||
console.error(`${failures} TEST(S) FAILED! ❌`);
|
||||
process.exit(1);
|
||||
}
|
||||
}
|
||||
|
||||
runTests();
|
||||
@@ -0,0 +1,87 @@
|
||||
const http = require("http");
|
||||
|
||||
function dockerRequest(path, method, body = null) {
|
||||
return new Promise((resolve, reject) => {
|
||||
const options = {
|
||||
socketPath: "/var/run/docker.sock",
|
||||
path: path,
|
||||
method: method,
|
||||
headers: {
|
||||
"Content-Type": "application/json",
|
||||
},
|
||||
};
|
||||
|
||||
const req = http.request(options, (res) => {
|
||||
let data = "";
|
||||
res.on("data", (chunk) => (data += chunk));
|
||||
res.on("end", () => {
|
||||
if (res.statusCode >= 200 && res.statusCode < 300) {
|
||||
try {
|
||||
resolve(data ? JSON.parse(data) : null);
|
||||
} catch (e) {
|
||||
resolve(data);
|
||||
}
|
||||
} else {
|
||||
reject(new Error(`Docker API Error ${res.statusCode}: ${data}`));
|
||||
}
|
||||
});
|
||||
});
|
||||
|
||||
req.on("error", (err) => reject(err));
|
||||
if (body) {
|
||||
req.write(JSON.stringify(body));
|
||||
}
|
||||
req.end();
|
||||
});
|
||||
}
|
||||
|
||||
async function runExec(containerName, cmd) {
|
||||
try {
|
||||
// 1. Create exec instance
|
||||
const execConfig = {
|
||||
AttachStdout: true,
|
||||
AttachStderr: true,
|
||||
Cmd: cmd,
|
||||
};
|
||||
const createRes = await dockerRequest(`/containers/${containerName}/exec`, "POST", execConfig);
|
||||
const execId = createRes.Id;
|
||||
|
||||
// 2. Start exec instance
|
||||
// Note: Start API returns raw stream, so we use http.request directly to read it
|
||||
return new Promise((resolve, reject) => {
|
||||
const options = {
|
||||
socketPath: "/var/run/docker.sock",
|
||||
path: `/exec/${execId}/start`,
|
||||
method: "POST",
|
||||
headers: {
|
||||
"Content-Type": "application/json",
|
||||
},
|
||||
};
|
||||
|
||||
const req = http.request(options, (res) => {
|
||||
let output = "";
|
||||
res.on("data", (chunk) => (output += chunk));
|
||||
res.on("end", () => {
|
||||
// Docker multiplexes stdout/stderr in the stream.
|
||||
// First 8 bytes of each frame contain header info: [stream_type, 0, 0, 0, size1, size2, size3, size4]
|
||||
// For simple outputs, we can clean up non-printable characters or parse directly
|
||||
resolve(output);
|
||||
});
|
||||
});
|
||||
|
||||
req.on("error", (err) => reject(err));
|
||||
req.write(JSON.stringify({ Detach: false, Tty: false }));
|
||||
req.end();
|
||||
});
|
||||
} catch (err) {
|
||||
throw err;
|
||||
}
|
||||
}
|
||||
|
||||
async function main() {
|
||||
const result = await runExec("paddleocr-vllm-server", ["nvidia-smi", "--query-gpu=index,name,utilization.gpu,utilization.memory,memory.total,memory.used,memory.free,uuid", "--format=csv,noheader,nounits"]);
|
||||
console.log("Exec output:");
|
||||
console.log(result);
|
||||
}
|
||||
|
||||
main().catch(console.error);
|
||||
@@ -0,0 +1,34 @@
|
||||
{
|
||||
"compilerOptions": {
|
||||
"target": "ES2017",
|
||||
"lib": ["dom", "dom.iterable", "esnext"],
|
||||
"allowJs": true,
|
||||
"skipLibCheck": true,
|
||||
"strict": true,
|
||||
"noEmit": true,
|
||||
"esModuleInterop": true,
|
||||
"module": "esnext",
|
||||
"moduleResolution": "bundler",
|
||||
"resolveJsonModule": true,
|
||||
"isolatedModules": true,
|
||||
"jsx": "react-jsx",
|
||||
"incremental": true,
|
||||
"plugins": [
|
||||
{
|
||||
"name": "next"
|
||||
}
|
||||
],
|
||||
"paths": {
|
||||
"@/*": ["./src/*"]
|
||||
}
|
||||
},
|
||||
"include": [
|
||||
"next-env.d.ts",
|
||||
"**/*.ts",
|
||||
"**/*.tsx",
|
||||
".next/types/**/*.ts",
|
||||
".next/dev/types/**/*.ts",
|
||||
"**/*.mts"
|
||||
],
|
||||
"exclude": ["node_modules"]
|
||||
}
|
||||
@@ -0,0 +1,19 @@
|
||||
[project]
|
||||
name = "paddle-ocr-vl-vllm-server"
|
||||
version = "0.1.0"
|
||||
description = "vLLM inference service for PaddleOCR-VL-1.6"
|
||||
readme = "README.md"
|
||||
requires-python = ">=3.10,<3.13"
|
||||
dependencies = [
|
||||
"paddleocr[doc-parser]>=3.3.0",
|
||||
"transformers==4.57.6",
|
||||
"torch==2.8.0",
|
||||
"vllm==0.10.2",
|
||||
"uvloop",
|
||||
"xformers",
|
||||
"einops",
|
||||
"ultralytics>=8.4.57",
|
||||
]
|
||||
|
||||
[dependency-groups]
|
||||
dev = []
|
||||