chore: normalize line endings (CRLF -> LF)

No content changes: git diff --ignore-all-space over these files is empty.
The churn came from editing on Windows against a repo checked out with LF.
This commit is contained in:
fhanyuh committed 2026-08-27 10:40:49 +07:00
1 parent 15566a6951
commit caf8e98378
315 files changed
+86950 -86950

No files matched your search

+7 -7
View File
@@ -1,7 +1,7 @@
--- ---
name: grill-me name: grill-me
description: A relentless interview to sharpen a plan or design. description: A relentless interview to sharpen a plan or design.
disable-model-invocation: true disable-model-invocation: true
--- ---
Run a `/grilling` session. Run a `/grilling` session.
+12 -12
View File
@@ -1,12 +1,12 @@
--- ---
name: grilling name: grilling
description: Grill the user relentlessly about a plan or design. Use when the user wants to stress-test a plan before building, or uses any 'grill' trigger phrases. description: Grill the user relentlessly about a plan or design. Use when the user wants to stress-test a plan before building, or uses any 'grill' trigger phrases.
--- ---
Interview me relentlessly about every aspect of this plan until we reach a shared understanding. Walk down each branch of the design tree, resolving dependencies between decisions one-by-one. For each question, provide your recommended answer. Interview me relentlessly about every aspect of this plan until we reach a shared understanding. Walk down each branch of the design tree, resolving dependencies between decisions one-by-one. For each question, provide your recommended answer.
Ask the questions one at a time, waiting for feedback on each question before continuing. Asking multiple questions at once is bewildering. Ask the questions one at a time, waiting for feedback on each question before continuing. Asking multiple questions at once is bewildering.
If a question can be answered by exploring the codebase, explore the codebase instead. If a question can be answered by exploring the codebase, explore the codebase instead.
Do not enact the plan until I confirm we have reached a shared understanding. Do not enact the plan until I confirm we have reached a shared understanding.
+53 -53
View File
@@ -1,56 +1,56 @@
# Miscellaneous # Miscellaneous
*.class *.class
*.log *.log
*.pyc *.pyc
*.swp *.swp
.DS_Store .DS_Store
.atom/ .atom/
.build/ .build/
.buildlog/ .buildlog/
.history .history
.svn/ .svn/
.swiftpm/ .swiftpm/
migrate_working_dir/ migrate_working_dir/
# IntelliJ related # IntelliJ related
*.iml *.iml
*.ipr *.ipr
*.iws *.iws
.idea/ .idea/
# The .vscode folder contains launch configuration and tasks you configure in # The .vscode folder contains launch configuration and tasks you configure in
# VS Code which you may wish to be included in version control, so this line # VS Code which you may wish to be included in version control, so this line
# is commented out by default. # is commented out by default.
#.vscode/ #.vscode/
# Flutter/Dart/Pub related # Flutter/Dart/Pub related
**/doc/api/ **/doc/api/
**/ios/Flutter/.last_build_id **/ios/Flutter/.last_build_id
.dart_tool/ .dart_tool/
.flutter-plugins-dependencies .flutter-plugins-dependencies
.pub-cache/ .pub-cache/
.pub/ .pub/
/build/ /build/
/coverage/ /coverage/
# Symbolication related # Symbolication related
app.*.symbols app.*.symbols
# Obfuscation related # Obfuscation related
app.*.map.json app.*.map.json
# Android Studio will place build artifacts here # Android Studio will place build artifacts here
/android/app/debug /android/app/debug
/android/app/profile /android/app/profile
/android/app/release /android/app/release
# Node, Next.js, and general JS ignores # Node, Next.js, and general JS ignores
**/node_modules/ **/node_modules/
**/.next/ **/.next/
**/dist/ **/dist/
# Graphify code knowledge graph (regenerated locally via git hooks) # Graphify code knowledge graph (regenerated locally via git hooks)
graphify-out/ graphify-out/
# Client business documents kept local-only (user decision 2026-07-16) # Client business documents kept local-only (user decision 2026-07-16)
proposals/ proposals/
+45 -45
View File
@@ -1,45 +1,45 @@
# This file tracks properties of this Flutter project. # This file tracks properties of this Flutter project.
# Used by Flutter tool to assess capabilities and perform upgrades etc. # Used by Flutter tool to assess capabilities and perform upgrades etc.
# #
# This file should be version controlled and should not be manually edited. # This file should be version controlled and should not be manually edited.
version: version:
revision: "ad70ec4617166f1c38e5d2bfd388af71fda14f06" revision: "ad70ec4617166f1c38e5d2bfd388af71fda14f06"
channel: "stable" channel: "stable"
project_type: app project_type: app
# Tracks metadata for the flutter migrate command # Tracks metadata for the flutter migrate command
migration: migration:
platforms: platforms:
- platform: root - platform: root
create_revision: ad70ec4617166f1c38e5d2bfd388af71fda14f06 create_revision: ad70ec4617166f1c38e5d2bfd388af71fda14f06
base_revision: ad70ec4617166f1c38e5d2bfd388af71fda14f06 base_revision: ad70ec4617166f1c38e5d2bfd388af71fda14f06
- platform: android - platform: android
create_revision: ad70ec4617166f1c38e5d2bfd388af71fda14f06 create_revision: ad70ec4617166f1c38e5d2bfd388af71fda14f06
base_revision: ad70ec4617166f1c38e5d2bfd388af71fda14f06 base_revision: ad70ec4617166f1c38e5d2bfd388af71fda14f06
- platform: ios - platform: ios
create_revision: ad70ec4617166f1c38e5d2bfd388af71fda14f06 create_revision: ad70ec4617166f1c38e5d2bfd388af71fda14f06
base_revision: ad70ec4617166f1c38e5d2bfd388af71fda14f06 base_revision: ad70ec4617166f1c38e5d2bfd388af71fda14f06
- platform: linux - platform: linux
create_revision: ad70ec4617166f1c38e5d2bfd388af71fda14f06 create_revision: ad70ec4617166f1c38e5d2bfd388af71fda14f06
base_revision: ad70ec4617166f1c38e5d2bfd388af71fda14f06 base_revision: ad70ec4617166f1c38e5d2bfd388af71fda14f06
- platform: macos - platform: macos
create_revision: ad70ec4617166f1c38e5d2bfd388af71fda14f06 create_revision: ad70ec4617166f1c38e5d2bfd388af71fda14f06
base_revision: ad70ec4617166f1c38e5d2bfd388af71fda14f06 base_revision: ad70ec4617166f1c38e5d2bfd388af71fda14f06
- platform: web - platform: web
create_revision: ad70ec4617166f1c38e5d2bfd388af71fda14f06 create_revision: ad70ec4617166f1c38e5d2bfd388af71fda14f06
base_revision: ad70ec4617166f1c38e5d2bfd388af71fda14f06 base_revision: ad70ec4617166f1c38e5d2bfd388af71fda14f06
- platform: windows - platform: windows
create_revision: ad70ec4617166f1c38e5d2bfd388af71fda14f06 create_revision: ad70ec4617166f1c38e5d2bfd388af71fda14f06
base_revision: ad70ec4617166f1c38e5d2bfd388af71fda14f06 base_revision: ad70ec4617166f1c38e5d2bfd388af71fda14f06
# User provided section # User provided section
# List of Local paths (relative to this file) that should be # List of Local paths (relative to this file) that should be
# ignored by the migrate tool. # ignored by the migrate tool.
# #
# Files that are not part of the templates will be ignored by default. # Files that are not part of the templates will be ignored by default.
unmanaged_files: unmanaged_files:
- 'lib/main.dart' - 'lib/main.dart'
- 'ios/Runner.xcodeproj/project.pbxproj' - 'ios/Runner.xcodeproj/project.pbxproj'
+28 -28
View File
@@ -1,28 +1,28 @@
# This file configures the analyzer, which statically analyzes Dart code to # This file configures the analyzer, which statically analyzes Dart code to
# check for errors, warnings, and lints. # check for errors, warnings, and lints.
# #
# The issues identified by the analyzer are surfaced in the UI of Dart-enabled # The issues identified by the analyzer are surfaced in the UI of Dart-enabled
# IDEs (https://dart.dev/tools#ides-and-editors). The analyzer can also be # IDEs (https://dart.dev/tools#ides-and-editors). The analyzer can also be
# invoked from the command line by running `flutter analyze`. # invoked from the command line by running `flutter analyze`.
# The following line activates a set of recommended lints for Flutter apps, # The following line activates a set of recommended lints for Flutter apps,
# packages, and plugins designed to encourage good coding practices. # packages, and plugins designed to encourage good coding practices.
include: package:flutter_lints/flutter.yaml include: package:flutter_lints/flutter.yaml
linter: linter:
# The lint rules applied to this project can be customized in the # The lint rules applied to this project can be customized in the
# section below to disable rules from the `package:flutter_lints/flutter.yaml` # section below to disable rules from the `package:flutter_lints/flutter.yaml`
# included above or to enable additional rules. A list of all available lints # included above or to enable additional rules. A list of all available lints
# and their documentation is published at https://dart.dev/lints. # and their documentation is published at https://dart.dev/lints.
# #
# Instead of disabling a lint rule for the entire project in the # Instead of disabling a lint rule for the entire project in the
# section below, it can also be suppressed for a single line of code # section below, it can also be suppressed for a single line of code
# or a specific dart file by using the `// ignore: name_of_lint` and # or a specific dart file by using the `// ignore: name_of_lint` and
# `// ignore_for_file: name_of_lint` syntax on the line or in the file # `// ignore_for_file: name_of_lint` syntax on the line or in the file
# producing the lint. # producing the lint.
rules: rules:
# avoid_print: false # Uncomment to disable the `avoid_print` rule # avoid_print: false # Uncomment to disable the `avoid_print` rule
# prefer_single_quotes: true # Uncomment to enable the `prefer_single_quotes` rule # prefer_single_quotes: true # Uncomment to enable the `prefer_single_quotes` rule
# Additional information about this file can be found at # Additional information about this file can be found at
# https://dart.dev/guides/language/analysis-options # https://dart.dev/guides/language/analysis-options
+14 -14
View File
@@ -1,14 +1,14 @@
gradle-wrapper.jar gradle-wrapper.jar
/.gradle /.gradle
/captures/ /captures/
/gradlew /gradlew
/gradlew.bat /gradlew.bat
/local.properties /local.properties
GeneratedPluginRegistrant.java GeneratedPluginRegistrant.java
.cxx/ .cxx/
# Remember to never publicly share your keystore. # Remember to never publicly share your keystore.
# See https://flutter.dev/to/reference-keystore # See https://flutter.dev/to/reference-keystore
key.properties key.properties
**/*.keystore **/*.keystore
**/*.jks **/*.jks
+2 -2
View File
@@ -1,2 +1,2 @@
#Fri Jun 26 15:57:59 WIB 2026 #Fri Jun 26 15:57:59 WIB 2026
java.home=D\:\\Android\\jbr java.home=D\:\\Android\\jbr
+45 -45
View File
@@ -1,45 +1,45 @@
plugins { plugins {
id("com.android.application") id("com.android.application")
// The Flutter Gradle Plugin must be applied after the Android and Kotlin Gradle plugins. // The Flutter Gradle Plugin must be applied after the Android and Kotlin Gradle plugins.
id("dev.flutter.flutter-gradle-plugin") id("dev.flutter.flutter-gradle-plugin")
} }
android { android {
namespace = "com.databisnis.app_pfm_ocr_v2" namespace = "com.databisnis.app_pfm_ocr_v2"
compileSdk = flutter.compileSdkVersion compileSdk = flutter.compileSdkVersion
ndkVersion = flutter.ndkVersion ndkVersion = flutter.ndkVersion
compileOptions { compileOptions {
sourceCompatibility = JavaVersion.VERSION_17 sourceCompatibility = JavaVersion.VERSION_17
targetCompatibility = JavaVersion.VERSION_17 targetCompatibility = JavaVersion.VERSION_17
} }
defaultConfig { defaultConfig {
// TODO: Specify your own unique Application ID (https://developer.android.com/studio/build/application-id.html). // TODO: Specify your own unique Application ID (https://developer.android.com/studio/build/application-id.html).
applicationId = "com.databisnis.app_pfm_ocr_v2" applicationId = "com.databisnis.app_pfm_ocr_v2"
// You can update the following values to match your application needs. // You can update the following values to match your application needs.
// For more information, see: https://flutter.dev/to/review-gradle-config. // For more information, see: https://flutter.dev/to/review-gradle-config.
minSdk = flutter.minSdkVersion minSdk = flutter.minSdkVersion
targetSdk = flutter.targetSdkVersion targetSdk = flutter.targetSdkVersion
versionCode = flutter.versionCode versionCode = flutter.versionCode
versionName = flutter.versionName versionName = flutter.versionName
} }
buildTypes { buildTypes {
release { release {
// TODO: Add your own signing config for the release build. // TODO: Add your own signing config for the release build.
// Signing with the debug keys for now, so `flutter run --release` works. // Signing with the debug keys for now, so `flutter run --release` works.
signingConfig = signingConfigs.getByName("debug") signingConfig = signingConfigs.getByName("debug")
} }
} }
} }
kotlin { kotlin {
compilerOptions { compilerOptions {
jvmTarget = org.jetbrains.kotlin.gradle.dsl.JvmTarget.JVM_17 jvmTarget = org.jetbrains.kotlin.gradle.dsl.JvmTarget.JVM_17
} }
} }
flutter { flutter {
source = "../.." source = "../.."
} }
+8 -8
View File
@@ -1,8 +1,8 @@
## This file must *NOT* be checked into Version Control Systems, ## This file must *NOT* be checked into Version Control Systems,
# as it contains information specific to your local configuration. # as it contains information specific to your local configuration.
# #
# Location of the SDK. This is only used by Gradle. # Location of the SDK. This is only used by Gradle.
# For customization when using a Version Control System, please read the # For customization when using a Version Control System, please read the
# header note. # header note.
#Fri Jun 26 15:57:59 WIB 2026 #Fri Jun 26 15:57:59 WIB 2026
sdk.dir=C\:\\Users\\rafha\\AppData\\Local\\Android\\Sdk sdk.dir=C\:\\Users\\rafha\\AppData\\Local\\Android\\Sdk
+7 -7
View File
@@ -1,7 +1,7 @@
<manifest xmlns:android="http://schemas.android.com/apk/res/android"> <manifest xmlns:android="http://schemas.android.com/apk/res/android">
<!-- The INTERNET permission is required for development. Specifically, <!-- The INTERNET permission is required for development. Specifically,
the Flutter tool needs it to communicate with the running application the Flutter tool needs it to communicate with the running application
to allow setting breakpoints, to provide hot reload, etc. to allow setting breakpoints, to provide hot reload, etc.
--> -->
<uses-permission android:name="android.permission.INTERNET"/> <uses-permission android:name="android.permission.INTERNET"/>
</manifest> </manifest>
+50 -50
View File
@@ -1,50 +1,50 @@
<manifest xmlns:android="http://schemas.android.com/apk/res/android"> <manifest xmlns:android="http://schemas.android.com/apk/res/android">
<uses-permission android:name="android.permission.INTERNET" /> <uses-permission android:name="android.permission.INTERNET" />
<uses-permission android:name="android.permission.CAMERA" /> <uses-permission android:name="android.permission.CAMERA" />
<uses-permission android:name="android.permission.ACCESS_FINE_LOCATION" /> <uses-permission android:name="android.permission.ACCESS_FINE_LOCATION" />
<uses-permission android:name="android.permission.ACCESS_COARSE_LOCATION" /> <uses-permission android:name="android.permission.ACCESS_COARSE_LOCATION" />
<application <application
android:label="app_pfm_ocr_v2" android:label="app_pfm_ocr_v2"
android:name="${applicationName}" android:name="${applicationName}"
android:icon="@mipmap/launcher_icon"> android:icon="@mipmap/launcher_icon">
<activity <activity
android:name=".MainActivity" android:name=".MainActivity"
android:exported="true" android:exported="true"
android:launchMode="singleTop" android:launchMode="singleTop"
android:taskAffinity="" android:taskAffinity=""
android:theme="@style/LaunchTheme" android:theme="@style/LaunchTheme"
android:configChanges="orientation|keyboardHidden|keyboard|screenSize|smallestScreenSize|locale|layoutDirection|fontScale|screenLayout|density|uiMode" android:configChanges="orientation|keyboardHidden|keyboard|screenSize|smallestScreenSize|locale|layoutDirection|fontScale|screenLayout|density|uiMode"
android:hardwareAccelerated="true" android:hardwareAccelerated="true"
android:windowSoftInputMode="adjustResize"> android:windowSoftInputMode="adjustResize">
<!-- Specifies an Android theme to apply to this Activity as soon as <!-- Specifies an Android theme to apply to this Activity as soon as
the Android process has started. This theme is visible to the user the Android process has started. This theme is visible to the user
while the Flutter UI initializes. After that, this theme continues while the Flutter UI initializes. After that, this theme continues
to determine the Window background behind the Flutter UI. --> to determine the Window background behind the Flutter UI. -->
<meta-data <meta-data
android:name="io.flutter.embedding.android.NormalTheme" android:name="io.flutter.embedding.android.NormalTheme"
android:resource="@style/NormalTheme" android:resource="@style/NormalTheme"
/> />
<intent-filter> <intent-filter>
<action android:name="android.intent.action.MAIN"/> <action android:name="android.intent.action.MAIN"/>
<category android:name="android.intent.category.LAUNCHER"/> <category android:name="android.intent.category.LAUNCHER"/>
</intent-filter> </intent-filter>
</activity> </activity>
<!-- Don't delete the meta-data below. <!-- Don't delete the meta-data below.
This is used by the Flutter tool to generate GeneratedPluginRegistrant.java --> This is used by the Flutter tool to generate GeneratedPluginRegistrant.java -->
<meta-data <meta-data
android:name="flutterEmbedding" android:name="flutterEmbedding"
android:value="2" /> android:value="2" />
</application> </application>
<!-- Required to query activities that can process text, see: <!-- Required to query activities that can process text, see:
https://developer.android.com/training/package-visibility and https://developer.android.com/training/package-visibility and
https://developer.android.com/reference/android/content/Intent#ACTION_PROCESS_TEXT. https://developer.android.com/reference/android/content/Intent#ACTION_PROCESS_TEXT.
In particular, this is used by the Flutter engine in io.flutter.plugin.text.ProcessTextPlugin. --> In particular, this is used by the Flutter engine in io.flutter.plugin.text.ProcessTextPlugin. -->
<queries> <queries>
<intent> <intent>
<action android:name="android.intent.action.PROCESS_TEXT"/> <action android:name="android.intent.action.PROCESS_TEXT"/>
<data android:mimeType="text/plain"/> <data android:mimeType="text/plain"/>
</intent> </intent>
</queries> </queries>
</manifest> </manifest>
@@ -1,5 +1,5 @@
package com.databisnis.app_pfm_ocr_v2 package com.databisnis.app_pfm_ocr_v2
import io.flutter.embedding.android.FlutterActivity import io.flutter.embedding.android.FlutterActivity
class MainActivity : FlutterActivity() class MainActivity : FlutterActivity()
@@ -1,12 +1,12 @@
<?xml version="1.0" encoding="utf-8"?> <?xml version="1.0" encoding="utf-8"?>
<!-- Modify this file to customize your launch splash screen --> <!-- Modify this file to customize your launch splash screen -->
<layer-list xmlns:android="http://schemas.android.com/apk/res/android"> <layer-list xmlns:android="http://schemas.android.com/apk/res/android">
<item android:drawable="?android:colorBackground" /> <item android:drawable="?android:colorBackground" />
<!-- You can insert your own image assets here --> <!-- You can insert your own image assets here -->
<!-- <item> <!-- <item>
<bitmap <bitmap
android:gravity="center" android:gravity="center"
android:src="@mipmap/launch_image" /> android:src="@mipmap/launch_image" />
</item> --> </item> -->
</layer-list> </layer-list>
@@ -1,12 +1,12 @@
<?xml version="1.0" encoding="utf-8"?> <?xml version="1.0" encoding="utf-8"?>
<!-- Modify this file to customize your launch splash screen --> <!-- Modify this file to customize your launch splash screen -->
<layer-list xmlns:android="http://schemas.android.com/apk/res/android"> <layer-list xmlns:android="http://schemas.android.com/apk/res/android">
<item android:drawable="@android:color/white" /> <item android:drawable="@android:color/white" />
<!-- You can insert your own image assets here --> <!-- You can insert your own image assets here -->
<!-- <item> <!-- <item>
<bitmap <bitmap
android:gravity="center" android:gravity="center"
android:src="@mipmap/launch_image" /> android:src="@mipmap/launch_image" />
</item> --> </item> -->
</layer-list> </layer-list>
@@ -1,18 +1,18 @@
<?xml version="1.0" encoding="utf-8"?> <?xml version="1.0" encoding="utf-8"?>
<resources> <resources>
<!-- Theme applied to the Android Window while the process is starting when the OS's Dark Mode setting is on --> <!-- Theme applied to the Android Window while the process is starting when the OS's Dark Mode setting is on -->
<style name="LaunchTheme" parent="@android:style/Theme.Black.NoTitleBar"> <style name="LaunchTheme" parent="@android:style/Theme.Black.NoTitleBar">
<!-- Show a splash screen on the activity. Automatically removed when <!-- Show a splash screen on the activity. Automatically removed when
the Flutter engine draws its first frame --> the Flutter engine draws its first frame -->
<item name="android:windowBackground">@drawable/launch_background</item> <item name="android:windowBackground">@drawable/launch_background</item>
</style> </style>
<!-- Theme applied to the Android Window as soon as the process has started. <!-- Theme applied to the Android Window as soon as the process has started.
This theme determines the color of the Android Window while your This theme determines the color of the Android Window while your
Flutter UI initializes, as well as behind your Flutter UI while its Flutter UI initializes, as well as behind your Flutter UI while its
running. running.
This Theme is only used starting with V2 of Flutter's Android embedding. --> This Theme is only used starting with V2 of Flutter's Android embedding. -->
<style name="NormalTheme" parent="@android:style/Theme.Black.NoTitleBar"> <style name="NormalTheme" parent="@android:style/Theme.Black.NoTitleBar">
<item name="android:windowBackground">?android:colorBackground</item> <item name="android:windowBackground">?android:colorBackground</item>
</style> </style>
</resources> </resources>
+18 -18
View File
@@ -1,18 +1,18 @@
<?xml version="1.0" encoding="utf-8"?> <?xml version="1.0" encoding="utf-8"?>
<resources> <resources>
<!-- Theme applied to the Android Window while the process is starting when the OS's Dark Mode setting is off --> <!-- Theme applied to the Android Window while the process is starting when the OS's Dark Mode setting is off -->
<style name="LaunchTheme" parent="@android:style/Theme.Light.NoTitleBar"> <style name="LaunchTheme" parent="@android:style/Theme.Light.NoTitleBar">
<!-- Show a splash screen on the activity. Automatically removed when <!-- Show a splash screen on the activity. Automatically removed when
the Flutter engine draws its first frame --> the Flutter engine draws its first frame -->
<item name="android:windowBackground">@drawable/launch_background</item> <item name="android:windowBackground">@drawable/launch_background</item>
</style> </style>
<!-- Theme applied to the Android Window as soon as the process has started. <!-- Theme applied to the Android Window as soon as the process has started.
This theme determines the color of the Android Window while your This theme determines the color of the Android Window while your
Flutter UI initializes, as well as behind your Flutter UI while its Flutter UI initializes, as well as behind your Flutter UI while its
running. running.
This Theme is only used starting with V2 of Flutter's Android embedding. --> This Theme is only used starting with V2 of Flutter's Android embedding. -->
<style name="NormalTheme" parent="@android:style/Theme.Light.NoTitleBar"> <style name="NormalTheme" parent="@android:style/Theme.Light.NoTitleBar">
<item name="android:windowBackground">?android:colorBackground</item> <item name="android:windowBackground">?android:colorBackground</item>
</style> </style>
</resources> </resources>
+7 -7
View File
@@ -1,7 +1,7 @@
<manifest xmlns:android="http://schemas.android.com/apk/res/android"> <manifest xmlns:android="http://schemas.android.com/apk/res/android">
<!-- The INTERNET permission is required for development. Specifically, <!-- The INTERNET permission is required for development. Specifically,
the Flutter tool needs it to communicate with the running application the Flutter tool needs it to communicate with the running application
to allow setting breakpoints, to provide hot reload, etc. to allow setting breakpoints, to provide hot reload, etc.
--> -->
<uses-permission android:name="android.permission.INTERNET"/> <uses-permission android:name="android.permission.INTERNET"/>
</manifest> </manifest>
+24 -24
View File
@@ -1,24 +1,24 @@
allprojects { allprojects {
repositories { repositories {
google() google()
mavenCentral() mavenCentral()
} }
} }
val newBuildDir: Directory = val newBuildDir: Directory =
rootProject.layout.buildDirectory rootProject.layout.buildDirectory
.dir("../../build") .dir("../../build")
.get() .get()
rootProject.layout.buildDirectory.value(newBuildDir) rootProject.layout.buildDirectory.value(newBuildDir)
subprojects { subprojects {
val newSubprojectBuildDir: Directory = newBuildDir.dir(project.name) val newSubprojectBuildDir: Directory = newBuildDir.dir(project.name)
project.layout.buildDirectory.value(newSubprojectBuildDir) project.layout.buildDirectory.value(newSubprojectBuildDir)
} }
subprojects { subprojects {
project.evaluationDependsOn(":app") project.evaluationDependsOn(":app")
} }
tasks.register<Delete>("clean") { tasks.register<Delete>("clean") {
delete(rootProject.layout.buildDirectory) delete(rootProject.layout.buildDirectory)
} }
+6 -6
View File
@@ -1,6 +1,6 @@
org.gradle.jvmargs=-Xmx8G -XX:MaxMetaspaceSize=4G -XX:ReservedCodeCacheSize=512m -XX:+HeapDumpOnOutOfMemoryError org.gradle.jvmargs=-Xmx8G -XX:MaxMetaspaceSize=4G -XX:ReservedCodeCacheSize=512m -XX:+HeapDumpOnOutOfMemoryError
android.useAndroidX=true android.useAndroidX=true
# This newDsl flag was added by the Flutter template # This newDsl flag was added by the Flutter template
android.newDsl=false android.newDsl=false
# This builtInKotlin flag was added by the Flutter template # This builtInKotlin flag was added by the Flutter template
android.builtInKotlin=false android.builtInKotlin=false
+5 -5
View File
@@ -1,5 +1,5 @@
distributionBase=GRADLE_USER_HOME distributionBase=GRADLE_USER_HOME
distributionPath=wrapper/dists distributionPath=wrapper/dists
zipStoreBase=GRADLE_USER_HOME zipStoreBase=GRADLE_USER_HOME
zipStorePath=wrapper/dists zipStorePath=wrapper/dists
distributionUrl=https\://services.gradle.org/distributions/gradle-9.1.0-all.zip distributionUrl=https\://services.gradle.org/distributions/gradle-9.1.0-all.zip
+26 -26
View File
@@ -1,26 +1,26 @@
pluginManagement { pluginManagement {
val flutterSdkPath = val flutterSdkPath =
run { run {
val properties = java.util.Properties() val properties = java.util.Properties()
file("local.properties").inputStream().use { properties.load(it) } file("local.properties").inputStream().use { properties.load(it) }
val flutterSdkPath = properties.getProperty("flutter.sdk") val flutterSdkPath = properties.getProperty("flutter.sdk")
require(flutterSdkPath != null) { "flutter.sdk not set in local.properties" } require(flutterSdkPath != null) { "flutter.sdk not set in local.properties" }
flutterSdkPath flutterSdkPath
} }
includeBuild("$flutterSdkPath/packages/flutter_tools/gradle") includeBuild("$flutterSdkPath/packages/flutter_tools/gradle")
repositories { repositories {
google() google()
mavenCentral() mavenCentral()
gradlePluginPortal() gradlePluginPortal()
} }
} }
plugins { plugins {
id("dev.flutter.flutter-plugin-loader") version "1.0.0" id("dev.flutter.flutter-plugin-loader") version "1.0.0"
id("com.android.application") version "9.0.1" apply false id("com.android.application") version "9.0.1" apply false
id("org.jetbrains.kotlin.android") version "2.3.20" apply false id("org.jetbrains.kotlin.android") version "2.3.20" apply false
} }
include(":app") include(":app")
+14 -14
View File
@@ -1,14 +1,14 @@
.git .git
.github .github
.venv .venv
.venv-api .venv-api
PaddleOCR-VL-1.6_Online_Demo/.venv PaddleOCR-VL-1.6_Online_Demo/.venv
**/__pycache__ **/__pycache__
**/*.pyc **/*.pyc
.cache .cache
.python-version .python-version
issues issues
*.md *.md
pfm-web-app/node_modules pfm-web-app/node_modules
pfm-web-app/.next pfm-web-app/.next
+8 -8
View File
@@ -1,8 +1,8 @@
# Port to serve the application on the host (routed via Nginx) # Port to serve the application on the host (routed via Nginx)
APP_PORT=8000 APP_PORT=8000
# GPU index to allocate (e.g. 0, 1, or 0,1) # GPU index to allocate (e.g. 0, 1, or 0,1)
CUDA_VISIBLE_DEVICES=1 CUDA_VISIBLE_DEVICES=1
# Secret used to sign/verify account login JWTs # Secret used to sign/verify account login JWTs
JWT_SECRET=change-me JWT_SECRET=change-me
+18 -18
View File
@@ -1,18 +1,18 @@
.venv/ .venv/
.venv-api/ .venv-api/
.env .env
__pycache__/ __pycache__/
*.pyc *.pyc
.python-version .python-version
.antigravitycli/ .antigravitycli/
PaddleOCR-VL-1.6_Online_Demo/examples/do-pfm/*.jpg PaddleOCR-VL-1.6_Online_Demo/examples/do-pfm/*.jpg
pfm-web-app/public/do-pfm/* pfm-web-app/public/do-pfm/*
uploads/* uploads/*
pfm-web-app/public/produk-pfm/**/*.jpeg pfm-web-app/public/produk-pfm/**/*.jpeg
pfm-web-app/public/produk-pfm/yolo_dataset/* pfm-web-app/public/produk-pfm/yolo_dataset/*
pfm-web-app/public/produk-pfm/runs/* pfm-web-app/public/produk-pfm/runs/*
pfm-web-app/public/produk-pfm/models/* pfm-web-app/public/produk-pfm/models/*
*.pt *.pt
test_img.jpeg test_img.jpeg
screenshots/*.jpg screenshots/*.jpg
pfm-web-app/public/test-images/* pfm-web-app/public/test-images/*
+18 -18
View File
@@ -1,18 +1,18 @@
[submodule "deepseek-ocr-2-demo-2026"] [submodule "deepseek-ocr-2-demo-2026"]
path = deepseek-ocr-2-demo-2026 path = deepseek-ocr-2-demo-2026
url = https://github.com/abdshomad/deepseek-ocr-2-demo-2026.git url = https://github.com/abdshomad/deepseek-ocr-2-demo-2026.git
[submodule "LightOnOCR-2-1B-Demo-2026"] [submodule "LightOnOCR-2-1B-Demo-2026"]
path = LightOnOCR-2-1B-Demo-2026 path = LightOnOCR-2-1B-Demo-2026
url = https://github.com/abdshomad/LightOnOCR-2-1B-Demo-2026.git url = https://github.com/abdshomad/LightOnOCR-2-1B-Demo-2026.git
[submodule "nvidia-nemotron-ocr-v2-demo-2026"] [submodule "nvidia-nemotron-ocr-v2-demo-2026"]
path = nvidia-nemotron-ocr-v2-demo-2026 path = nvidia-nemotron-ocr-v2-demo-2026
url = https://github.com/abdshomad/nvidia-nemotron-ocr-v2-demo-2026.git url = https://github.com/abdshomad/nvidia-nemotron-ocr-v2-demo-2026.git
[submodule "dots.ocr-demo-2026"] [submodule "dots.ocr-demo-2026"]
path = dots.ocr-demo-2026 path = dots.ocr-demo-2026
url = https://github.com/abdshomad/dots.ocr-demo-2026.git url = https://github.com/abdshomad/dots.ocr-demo-2026.git
[submodule "glm-ocr-demo-2026"] [submodule "glm-ocr-demo-2026"]
path = glm-ocr-demo-2026 path = glm-ocr-demo-2026
url = https://github.com/abdshomad/glm-ocr-demo-2026.git url = https://github.com/abdshomad/glm-ocr-demo-2026.git
[submodule "andrej-karpathy-skills"] [submodule "andrej-karpathy-skills"]
path = andrej-karpathy-skills path = andrej-karpathy-skills
url = https://github.com/multica-ai/andrej-karpathy-skills.git url = https://github.com/multica-ai/andrej-karpathy-skills.git
+188 -188
View File
@@ -1,188 +1,188 @@
# AGENTS: PaddleOCR-VL-1.6 vLLM Service + Agents Settings Kit # AGENTS: PaddleOCR-VL-1.6 vLLM Service + Agents Settings Kit
This is the authoritative rules file for any AI coding agent (Claude Code, Cursor, This is the authoritative rules file for any AI coding agent (Claude Code, Cursor,
GitHub Copilot, Aider, etc.) working inside `backend/`. Two unrelated concerns live GitHub Copilot, Aider, etc.) working inside `backend/`. Two unrelated concerns live
here side by side: **Part A** is this repo's original vLLM/PaddleOCR service doc. here side by side: **Part A** is this repo's original vLLM/PaddleOCR service doc.
**Part B** (appended 2026-07-08) is a **backend-scoped copy** of the **Part B** (appended 2026-07-08) is a **backend-scoped copy** of the
[fhanyuh/agents-settings](https://github.com/fhanyuh/agents-settings) `e`/`enhance` [fhanyuh/agents-settings](https://github.com/fhanyuh/agents-settings) `e`/`enhance`
and `n`/`next` workflow — see root `../AGENTS.md` for the same kit covering the and `n`/`next` workflow — see root `../AGENTS.md` for the same kit covering the
Flutter side of this repo. The two copies are independent: this one's Flutter side of this repo. The two copies are independent: this one's
`plans/next-enhancements.md` and `docs/feature-list.md` only track backend work. `plans/next-enhancements.md` and `docs/feature-list.md` only track backend work.
--- ---
# Part A — vLLM Service (PaddleOCR-VL-1.6) # Part A — vLLM Service (PaddleOCR-VL-1.6)
This repository serves **PaddleOCR-VL-1.6** as a dedicated VLM inference backend using **vLLM**. All Python workflows use **uv** (never bare `pip` or system Python). Full detail (client usage examples, tuning, troubleshooting, issue-file template) moved to [docs/vllm-service.md](docs/vllm-service.md) 2026-07-08 to keep this file under the Part B kit's 256-line threshold (§3) — this section keeps only the essentials. This repository serves **PaddleOCR-VL-1.6** as a dedicated VLM inference backend using **vLLM**. All Python workflows use **uv** (never bare `pip` or system Python). Full detail (client usage examples, tuning, troubleshooting, issue-file template) moved to [docs/vllm-service.md](docs/vllm-service.md) 2026-07-08 to keep this file under the Part B kit's 256-line threshold (§3) — this section keeps only the essentials.
## Architecture ## Architecture
``` ```
Client (PaddleOCR pipeline) --> HTTP /v1 --> paddleocr genai_server (vLLM backend) Client (PaddleOCR pipeline) --> HTTP /v1 --> paddleocr genai_server (vLLM backend)
``` ```
This service exposes only the VLM stage. Clients connect with `vl_rec_backend="vllm-server"` and `vl_rec_server_url="http://<host>:8118/v1"`. This service exposes only the VLM stage. Clients connect with `vl_rec_backend="vllm-server"` and `vl_rec_server_url="http://<host>:8118/v1"`.
## Quick start ## Quick start
```bash ```bash
./scripts/install.sh # 1) Create Python 3.12 venv and install dependencies ./scripts/install.sh # 1) Create Python 3.12 venv and install dependencies
./scripts/serve.sh # 2) Start the vLLM-backed genai server ./scripts/serve.sh # 2) Start the vLLM-backed genai server
``` ```
Default endpoint: `http://0.0.0.0:8118/v1`. Never use `python -m pip`, `pip install`, or `python -m venv` directly in this repo — always `uv sync` / `uv run` / `uv add`. Default endpoint: `http://0.0.0.0:8118/v1`. Never use `python -m pip`, `pip install`, or `python -m venv` directly in this repo — always `uv sync` / `uv run` / `uv add`.
## Issue recording (always follow) ## Issue recording (always follow)
**Every problem encountered** during install, serve, debug, or client integration must be written to `issues/{NN}-{slug}.md` before moving on — even if resolved in the same session. Naming/template details: [docs/vllm-service.md](docs/vllm-service.md#issue-recording--naming-and-template). **Every problem encountered** during install, serve, debug, or client integration must be written to `issues/{NN}-{slug}.md` before moving on — even if resolved in the same session. Naming/template details: [docs/vllm-service.md](docs/vllm-service.md#issue-recording--naming-and-template).
## Environment variables ## Environment variables
Copy `.env.example` to `.env` and adjust as needed: Copy `.env.example` to `.env` and adjust as needed:
| Variable | Default | Description | | Variable | Default | Description |
|----------|---------|-------------| |----------|---------|-------------|
| `GENAI_HOST` | `0.0.0.0` | Bind address | | `GENAI_HOST` | `0.0.0.0` | Bind address |
| `GENAI_PORT` | `8118` | Service port | | `GENAI_PORT` | `8118` | Service port |
| `GENAI_MODEL` | `PaddleOCR-VL-1.6-0.9B` | Model name for `genai_server` | | `GENAI_MODEL` | `PaddleOCR-VL-1.6-0.9B` | Model name for `genai_server` |
| `GENAI_BACKEND` | `vllm` | Inference backend | | `GENAI_BACKEND` | `vllm` | Inference backend |
| `VLLM_CONFIG` | `config/vllm_config.yaml` | vLLM backend YAML config | | `VLLM_CONFIG` | `config/vllm_config.yaml` | vLLM backend YAML config |
| `CUDA_VISIBLE_DEVICES` | `1` (see `.env.example`) | GPU index(es) to use | | `CUDA_VISIBLE_DEVICES` | `1` (see `.env.example`) | GPU index(es) to use |
On dual-GPU hosts, pick the GPU with more free VRAM. If startup fails with a memory error, lower `gpu-memory-utilization` in `config/vllm_config.yaml` — see [docs/vllm-service.md](docs/vllm-service.md#gpu-memory-on-startup). On dual-GPU hosts, pick the GPU with more free VRAM. If startup fails with a memory error, lower `gpu-memory-utilization` in `config/vllm_config.yaml` — see [docs/vllm-service.md](docs/vllm-service.md#gpu-memory-on-startup).
## File map ## File map
| Path | Purpose | | Path | Purpose |
|------|---------| |------|---------|
| `issues/` | Recorded problems and fixes (`{NN}-{slug}.md`) | | `issues/` | Recorded problems and fixes (`{NN}-{slug}.md`) |
| `pyproject.toml` | uv project metadata and base dependencies | | `pyproject.toml` | uv project metadata and base dependencies |
| `scripts/install.sh` | Bootstrap venv + vLLM server deps | | `scripts/install.sh` | Bootstrap venv + vLLM server deps |
| `scripts/serve.sh` | Start `paddleocr genai_server` | | `scripts/serve.sh` | Start `paddleocr genai_server` |
| `config/vllm_config.yaml` | vLLM backend tuning | | `config/vllm_config.yaml` | vLLM backend tuning |
| `.env.example` | Environment variable template | | `.env.example` | Environment variable template |
| `docs/vllm-service.md` | Full vLLM reference (client usage, tuning, troubleshooting) | | `docs/vllm-service.md` | Full vLLM reference (client usage, tuning, troubleshooting) |
## Coding Guidelines (always follow) ## Coding Guidelines (always follow)
We use the karpathy-guidelines skill to reduce common LLM coding mistakes: We use the karpathy-guidelines skill to reduce common LLM coding mistakes:
1. **Think Before Coding**: Explicitly state assumptions and surface tradeoffs instead of making silent choices. 1. **Think Before Coding**: Explicitly state assumptions and surface tradeoffs instead of making silent choices.
2. **Simplicity First**: Write the minimum amount of code to solve the problem with zero speculative configurations. 2. **Simplicity First**: Write the minimum amount of code to solve the problem with zero speculative configurations.
3. **Surgical Changes**: Edit only what is required and match the existing coding style exactly. 3. **Surgical Changes**: Edit only what is required and match the existing coding style exactly.
4. **Goal-Driven Execution**: Define verifiable success criteria and run automated tests/screenshots to confirm correctness. 4. **Goal-Driven Execution**: Define verifiable success criteria and run automated tests/screenshots to confirm correctness.
5. **SOLID Principles**: Always design, implement, and refactor code adhering to SOLID programming principles (Single Responsibility, Open/Closed, Liskov Substitution, Interface Segregation, Dependency Inversion) to ensure modularity, scalability, and maintainability. 5. **SOLID Principles**: Always design, implement, and refactor code adhering to SOLID programming principles (Single Responsibility, Open/Closed, Liskov Substitution, Interface Segregation, Dependency Inversion) to ensure modularity, scalability, and maintainability.
## Path Guidelines (always follow) ## Path Guidelines (always follow)
Never use full paths containing the user's logged-in name (e.g., `/home/{uid}/path`). Always use relative paths instead (e.g., `.` or `./path` relative to the workspace root). Never use full paths containing the user's logged-in name (e.g., `/home/{uid}/path`). Always use relative paths instead (e.g., `.` or `./path` relative to the workspace root).
## App Testing Guidelines (always follow) ## App Testing Guidelines (always follow)
When the user intentionally asks to test the app: When the user intentionally asks to test the app:
- Use browser tools to test the app. - Use browser tools to test the app.
- Take a screenshot for each sample image, each step, and each variant/option (if any), until the OCR result appears. - Take a screenshot for each sample image, each step, and each variant/option (if any), until the OCR result appears.
- Save the screenshots in the `/screenshots/` folder. - Save the screenshots in the `/screenshots/` folder.
- Follow the file naming convention: `{2-digit-number}-{step#}-{variant_or_options_if_any}-{slug}.jpg` (e.g., `01-step1-default-upload.jpg`). - Follow the file naming convention: `{2-digit-number}-{step#}-{variant_or_options_if_any}-{slug}.jpg` (e.g., `01-step1-default-upload.jpg`).
--- ---
# Part B — Agents Settings Kit (backend-scoped `e`/`n` workflow) # Part B — Agents Settings Kit (backend-scoped `e`/`n` workflow)
Backend-scoped copy of the [fhanyuh/agents-settings](https://github.com/fhanyuh/agents-settings) kit, adopted 2026-07-08. Covers only `backend/` modules (Next.js API Gateway, OCR Pipeline & Accuracy, Postgres Data Layer, DevOps/Docker) — Flutter modules are tracked by the separate copy at root `../AGENTS.md`. `../CLAUDE.md` (root) and `CLAUDE.md` (this dir) each import their own copy. Backend-scoped copy of the [fhanyuh/agents-settings](https://github.com/fhanyuh/agents-settings) kit, adopted 2026-07-08. Covers only `backend/` modules (Next.js API Gateway, OCR Pipeline & Accuracy, Postgres Data Layer, DevOps/Docker) — Flutter modules are tracked by the separate copy at root `../AGENTS.md`. `../CLAUDE.md` (root) and `CLAUDE.md` (this dir) each import their own copy.
## B0. Adopting Into an Existing Project ## B0. Adopting Into an Existing Project
Already done for this repo (this split *is* that adoption, mirroring root's own §0 audit). Re-run "i"/"init" here to force a re-audit of `backend/` specifically (e.g. after a large refactor). Already done for this repo (this split *is* that adoption, mirroring root's own §0 audit). Re-run "i"/"init" here to force a re-audit of `backend/` specifically (e.g. after a large refactor).
## B1. Trigger "e" or "enhance" ## B1. Trigger "e" or "enhance"
- Read `plans/next-enhancements.md` (this dir) to understand current backend structure, history, and active tasks. - Read `plans/next-enhancements.md` (this dir) to understand current backend structure, history, and active tasks.
- Overwrite or update the active tasks list inside it. - Overwrite or update the active tasks list inside it.
- The plan must cover each backend section/module. - The plan must cover each backend section/module.
- Define **exactly 3 new enhancements per section**, each with a unique number (e.g. `1.1`), a clear functional description, and status `[TODO]`. - Define **exactly 3 new enhancements per section**, each with a unique number (e.g. `1.1`), a clear functional description, and status `[TODO]`.
- Present the plan to the user in your final summary. - Present the plan to the user in your final summary.
## B2. Trigger "n", "next", or "n{x}" ## B2. Trigger "n", "next", or "n{x}"
- Read `plans/next-enhancements.md` to check task status. - Read `plans/next-enhancements.md` to check task status.
- If all tasks are `[DONE]` (or none `[TODO]`), run **"e"/"enhance"** first. - If all tasks are `[DONE]` (or none `[TODO]`), run **"e"/"enhance"** first.
- Otherwise select the most impactful `[TODO]` task(s) by strategic value/impact — not just first-in-order. If `{x}` given, take the top `{x}` sequentially. - Otherwise select the most impactful `[TODO]` task(s) by strategic value/impact — not just first-in-order. If `{x}` given, take the top `{x}` sequentially.
### B2a. Clarify before building ("Grill Me" step) ### B2a. Clarify before building ("Grill Me" step)
Same rule as root AGENTS.md §2a: if scope/acceptance criteria are genuinely ambiguous, ask one question at a time (`AskUserQuestion` in Claude Code) until unambiguous, and record the resolved criteria as a 1-3 line note next to the task entry before writing code. Skip when the task is already unambiguous. Same rule as root AGENTS.md §2a: if scope/acceptance criteria are genuinely ambiguous, ask one question at a time (`AskUserQuestion` in Claude Code) until unambiguous, and record the resolved criteria as a 1-3 line note next to the task entry before writing code. Skip when the task is already unambiguous.
### B2b. TDD Workflow (Test First) ### B2b. TDD Workflow (Test First)
- **Write Tests First**: Before implementing the actual feature code for a task, write automated tests defining the expected behavior. - **Write Tests First**: Before implementing the actual feature code for a task, write automated tests defining the expected behavior.
- **Iterate Until Green**: Run the tests to confirm they fail, then write the implementation until all tests pass perfectly. - **Iterate Until Green**: Run the tests to confirm they fail, then write the implementation until all tests pass perfectly.
- **Browser Testing**: If the enhancement involves web UI or visual components, use browser tools (e.g., Chrome) to test the app visually and functionally if necessary. - **Browser Testing**: If the enhancement involves web UI or visual components, use browser tools (e.g., Chrome) to test the app visually and functionally if necessary.
- Implement the task(s) fully, applying the relevant role(s) from `SKILLS.md` (this dir). - Implement the task(s) fully, applying the relevant role(s) from `SKILLS.md` (this dir).
- On completion: - On completion:
1. Flip status to `[DONE]` in `plans/next-enhancements.md`. 1. Flip status to `[DONE]` in `plans/next-enhancements.md`.
2. Document the feature in `docs/feature-list.md` (this dir) under the right section. 2. Document the feature in `docs/feature-list.md` (this dir) under the right section.
3. **Create an Iteration Log**: Perform a code review and audit of the tasks just completed. Document this audit in `docs/iteration-log.md` (or append to it) to ensure all functions work perfectly. 3. **Create an Iteration Log**: Perform a code review and audit of the tasks just completed. Document this audit in `docs/iteration-log.md` (or append to it) to ensure all functions work perfectly.
4. **Update Documentation**: Sync any architecture or workflow changes back to `CLAUDE.md` and `SKILLS.md` to keep the agent instructions current. 4. **Update Documentation**: Sync any architecture or workflow changes back to `CLAUDE.md` and `SKILLS.md` to keep the agent instructions current.
- **Verify build integrity**: QA pass (golden path + edge cases + regression check on adjacent features — see `backend/CLAUDE.md`'s accuracy regression harness for OCR/parser changes specifically) and Hardware/Compatibility pass (cross-platform, GPU/VRAM footprint under Local/on-prem deployment — see Part A above). - **Verify build integrity**: QA pass (golden path + edge cases + regression check on adjacent features — see `backend/CLAUDE.md`'s accuracy regression harness for OCR/parser changes specifically) and Hardware/Compatibility pass (cross-platform, GPU/VRAM footprint under Local/on-prem deployment — see Part A above).
- State which task(s) were completed and the exact route/endpoint/menu path to see the new feature. - State which task(s) were completed and the exact route/endpoint/menu path to see the new feature.
## B3. File Size & Refactoring Rules ## B3. File Size & Refactoring Rules
Same 256-line threshold as root AGENTS.md §3, backend-wide. Applies to this file, `SKILLS.md`, and `CLAUDE.md` too — which is why Part A above was trimmed and linked out to `docs/vllm-service.md` rather than left inline. Same 256-line threshold as root AGENTS.md §3, backend-wide. Applies to this file, `SKILLS.md`, and `CLAUDE.md` too — which is why Part A above was trimmed and linked out to `docs/vllm-service.md` rather than left inline.
## B4. Roles ## B4. Roles
See `SKILLS.md` (this dir) — same 5 roles as root (Architect, Backend, Frontend, QA, Hardware/Compatibility), applied to backend surfaces only (API routes, OCR pipeline, DB layer, Docker/deploy). See `SKILLS.md` (this dir) — same 5 roles as root (Architect, Backend, Frontend, QA, Hardware/Compatibility), applied to backend surfaces only (API routes, OCR pipeline, DB layer, Docker/deploy).
## B5. Mockup Data & Demo/Live Mode ## B5. Mockup Data & Demo/Live Mode
Same as root AGENTS.md §5: mock data under `/data/mockup/`, a mock API layer mirroring the real backend contract, and a Demo/Live switcher. Not yet built for backend — see Adaptation Notes. Same as root AGENTS.md §5: mock data under `/data/mockup/`, a mock API layer mirroring the real backend contract, and a Demo/Live switcher. Not yet built for backend — see Adaptation Notes.
## B6. Cloud vs Local (On-Premise) ## B6. Cloud vs Local (On-Premise)
Same as root AGENTS.md §6, applied to backend service endpoints (Next.js gateway, pipeline API, vLLM server, Postgres) rather than the Flutter client's API base URL. Same as root AGENTS.md §6, applied to backend service endpoints (Next.js gateway, pipeline API, vLLM server, Postgres) rather than the Flutter client's API base URL.
## B7. Ad-hoc Feature Requests ## B7. Ad-hoc Feature Requests
Direct feature requests not using "e"/"n": implement and document in `docs/feature-list.md` (this dir). Direct feature requests not using "e"/"n": implement and document in `docs/feature-list.md` (this dir).
## Adaptation Notes (backend, split from root 2026-07-08) ## Adaptation Notes (backend, split from root 2026-07-08)
- **Origin**: sections 5-8 of root `plans/next-enhancements.md` (Backend — Next.js API - **Origin**: sections 5-8 of root `plans/next-enhancements.md` (Backend — Next.js API
Gateway, Backend — OCR Pipeline & Accuracy, Backend — Postgres Data Layer, DevOps — Gateway, Backend — OCR Pipeline & Accuracy, Backend — Postgres Data Layer, DevOps —
Docker & Dev Tunnel) copied here as sections 1-4, statuses re-verified against the Docker & Dev Tunnel) copied here as sections 1-4, statuses re-verified against the
live code before the copy (not copied blind) — see task 7.1's `withTransaction` live code before the copy (not copied blind) — see task 7.1's `withTransaction`
claim, task 5.1/5.2's dedup + timeout claims, and task 6.1's empty `models/` claim, claim, task 5.1/5.2's dedup + timeout claims, and task 6.1's empty `models/` claim,
all confirmed still accurate as of 2026-07-08. The root copy is frozen/archival all confirmed still accurate as of 2026-07-08. The root copy is frozen/archival
(see root `AGENTS.md`'s "Scope: excludes `backend/`") rather than deleted, so this (see root `AGENTS.md`'s "Scope: excludes `backend/`") rather than deleted, so this
file — not the root one — is the single active source of truth going forward. file — not the root one — is the single active source of truth going forward.
- **Real commands**: `npm run dev`/`build`/`lint` in `pfm-web-app/`; accuracy - **Real commands**: `npm run dev`/`build`/`lint` in `pfm-web-app/`; accuracy
regression harness `node pfm-web-app/scripts/accuracy-check.mts`; Python services regression harness `node pfm-web-app/scripts/accuracy-check.mts`; Python services
via `./scripts/install.sh` + `./scripts/serve.sh` (this vLLM repo) and via `./scripts/install.sh` + `./scripts/serve.sh` (this vLLM repo) and
`./scripts/install-pipeline.sh` + `./scripts/serve-pipeline.sh` (pipeline API + `./scripts/install-pipeline.sh` + `./scripts/serve-pipeline.sh` (pipeline API +
classifier). Full stack: `docker compose up -d --build` **from the repo root**, not classifier). Full stack: `docker compose up -d --build` **from the repo root**, not
from inside `backend/` (see root `CLAUDE.md` — two `docker-compose.yml` files from inside `backend/` (see root `CLAUDE.md` — two `docker-compose.yml` files
exist and running from here risks container-name conflicts). exist and running from here risks container-name conflicts).
- **Pre-existing files over the 256-line threshold** (§B3 debt, not a blocker — split - **Pre-existing files over the 256-line threshold** (§B3 debt, not a blocker — split
only if/when touched): `pfm-web-app/src/app/scan-pfm/page.tsx` (1169), only if/when touched): `pfm-web-app/src/app/scan-pfm/page.tsx` (1169),
`pfm-web-app/src/utils/parser.ts` (908), `pfm-web-app/src/app/page.tsx` (737), `pfm-web-app/src/utils/parser.ts` (908), `pfm-web-app/src/app/page.tsx` (737),
`config/classify_ocr_server.py` (691), `pfm-web-app/src/app/manual-label/page.tsx` `config/classify_ocr_server.py` (691), `pfm-web-app/src/app/manual-label/page.tsx`
(612), `pfm-web-app/src/app/api/parse/route.ts` (604), `pfm-web-app/src/db/init.ts` (612), `pfm-web-app/src/app/api/parse/route.ts` (604), `pfm-web-app/src/db/init.ts`
(477), `compare_sources_accuracy.py` (451), `pfm-web-app/src/utils/docker.ts` (362), (477), `compare_sources_accuracy.py` (451), `pfm-web-app/src/utils/docker.ts` (362),
`pfm-web-app/public/produk-pfm/train_classifier.py` (351), `compare_accuracy.py` `pfm-web-app/public/produk-pfm/train_classifier.py` (351), `compare_accuracy.py`
(308), `pfm-web-app/src/app/api/arena/route.ts` (265). This file itself (`AGENTS.md`) (308), `pfm-web-app/src/app/api/arena/route.ts` (265). This file itself (`AGENTS.md`)
was at 237 lines pre-kit and would have exceeded 256 once Part B was appended — was at 237 lines pre-kit and would have exceeded 256 once Part B was appended —
hence the split into `docs/vllm-service.md`. hence the split into `docs/vllm-service.md`.
- **No Demo/Live or Cloud/Local switch exists yet** (§B5, §B6) for the backend - **No Demo/Live or Cloud/Local switch exists yet** (§B5, §B6) for the backend
either. `docker-compose.override.yml` exposing `db`/`pipeline-api` directly to the either. `docker-compose.override.yml` exposing `db`/`pipeline-api` directly to the
host is a local-dev convenience, not a Cloud/Local deployment switch. host is a local-dev convenience, not a Cloud/Local deployment switch.
- **Naming collision resolved by this split**: `AGENTS.md` already existed in this - **Naming collision resolved by this split**: `AGENTS.md` already existed in this
directory (vLLM service doc, committed 2026-06-30, unrelated to this kit) before directory (vLLM service doc, committed 2026-06-30, unrelated to this kit) before
Part B was appended — unlike root, where `AGENTS.md` didn't previously exist. Don't Part B was appended — unlike root, where `AGENTS.md` didn't previously exist. Don't
assume backend's `AGENTS.md` is kit-only when reading it from another tool; Part A assume backend's `AGENTS.md` is kit-only when reading it from another tool; Part A
is unrelated, pre-existing content kept for a reason. is unrelated, pre-existing content kept for a reason.
- **Pre-existing, unrelated governance files left as-is**: `.agents/AGENTS.md` at the - **Pre-existing, unrelated governance files left as-is**: `.agents/AGENTS.md` at the
*repo root* (different path, OCR post-processing rules) and root *repo root* (different path, OCR post-processing rules) and root
`plans/next-enhancement-plan.md` (singular, `[DONE]` QA checklist) — neither is `plans/next-enhancement-plan.md` (singular, `[DONE]` QA checklist) — neither is
part of this kit; see root `AGENTS.md`'s own Adaptation Notes. part of this kit; see root `AGENTS.md`'s own Adaptation Notes.
+87 -87
View File
@@ -1,87 +1,87 @@
# CLAUDE.md # CLAUDE.md
This file provides guidance to Claude Code (claude.ai/code) when working with code in this repository. This file provides guidance to Claude Code (claude.ai/code) when working with code in this repository.
## What this repo is ## What this repo is
`app-pfm-ocr-v2/backend` is the **next-generation rewrite of `ai-ocr-pfm-2026`** — same underlying OCR infra (PaddleOCR-VL-1.6 on vLLM + a PaddlePaddle layout-parsing pipeline), same client (Charoen Pokphand/Primafood-branded frozen food products), but a reworked Next.js app (`pfm-web-app/`) and DB schema. If you need background on the shared OCR/vLLM infra (uv conventions, issue-recording workflow, GPU tuning), see `AGENTS.md` — it's carried over near-unchanged from the previous project. `app-pfm-ocr-v2/backend` is the **next-generation rewrite of `ai-ocr-pfm-2026`** — same underlying OCR infra (PaddleOCR-VL-1.6 on vLLM + a PaddlePaddle layout-parsing pipeline), same client (Charoen Pokphand/Primafood-branded frozen food products), but a reworked Next.js app (`pfm-web-app/`) and DB schema. If you need background on the shared OCR/vLLM infra (uv conventions, issue-recording workflow, GPU tuning), see `AGENTS.md` — it's carried over near-unchanged from the previous project.
**The active plan for porting the Product/SKU-scanning feature lives in [`plans/next-enhancements.md`](plans/next-enhancements.md) §2** — read it before touching anything related to `scan-pfm`, `produk-pfm`, or the product classifier, since it records exactly what's done vs. still missing and the decisions already made about how to build it. (This used to be a separate `next-implementation.md`; that file was deleted 2026-07-08 once its content was folded into the plan for traceability with the rest of the `e`/`n` backlog.) **The active plan for porting the Product/SKU-scanning feature lives in [`plans/next-enhancements.md`](plans/next-enhancements.md) §2** — read it before touching anything related to `scan-pfm`, `produk-pfm`, or the product classifier, since it records exactly what's done vs. still missing and the decisions already made about how to build it. (This used to be a separate `next-implementation.md`; that file was deleted 2026-07-08 once its content was folded into the plan for traceability with the rest of the `e`/`n` backlog.)
## How this project differs from `ai-ocr-pfm-2026` ## How this project differs from `ai-ocr-pfm-2026`
- **DO-PFM UI is consolidated into a single page.** Unlike the old project's per-route pages (`do-pfm/page.tsx`, `m-do-pfm/page.tsx`), v2's entire upload/history/item-review flow lives in one `pfm-web-app/src/app/page.tsx` (client component, local state, no separate routes). `nginx.conf` still has `/do-pfm`/`/m-do-pfm` location blocks left over from the old routing — these are currently dead (no matching Next.js route, would 404). - **DO-PFM UI is consolidated into a single page.** Unlike the old project's per-route pages (`do-pfm/page.tsx`, `m-do-pfm/page.tsx`), v2's entire upload/history/item-review flow lives in one `pfm-web-app/src/app/page.tsx` (client component, local state, no separate routes). `nginx.conf` still has `/do-pfm`/`/m-do-pfm` location blocks left over from the old routing — these are currently dead (no matching Next.js route, would 404).
- **Standalone-purpose pages still get their own route folder**, e.g. `pfm-web-app/src/app/manual-label/page.tsx` — a self-contained ground-truth annotation tool (own header, own theme, no shared chrome with the root page) backed by `api/manual-label/route.ts` and `sources/manual_labels.json`. This is the pattern to follow for any new single-purpose page (see `plans/next-enhancements.md` §2 for the Product-scan pages, which follow it). - **Standalone-purpose pages still get their own route folder**, e.g. `pfm-web-app/src/app/manual-label/page.tsx` — a self-contained ground-truth annotation tool (own header, own theme, no shared chrome with the root page) backed by `api/manual-label/route.ts` and `sources/manual_labels.json`. This is the pattern to follow for any new single-purpose page (see `plans/next-enhancements.md` §2 for the Product-scan pages, which follow it).
- **Real JWT auth, enforced on the production surface**: `src/utils/auth.ts` signs/verifies tokens (`signAccountToken`/`verifyAccountToken`/`getAccountFromAuthHeader`) against an `accounts` table, each account bound to exactly one `kode_toko` (store) — the intent being that an account's own store is used on upload instead of relying on OCR-based store-text matching. **Passwords are bcrypt-hashed** (`accounts.password`, via `bcryptjs` — chosen over native `bcrypt` since the `pfm-web-app` Docker stage is `node:20-slim` with no build toolchain for native addons; `pfm-web-app/src/db/init.ts` hashes the seed and idempotently migrates any pre-existing plaintext rows on every startup). As of 2026-07-08, `api/v1/documents/*` (list, PUT-by-id, upload) **reject requests with a missing/invalid token (401)** — this is the real production surface, and the Flutter client already does a real login and attaches `Authorization: Bearer <token>` to every request (`lib/features/auth/auth_provider.dart` + `lib/core/network/api_client.dart`). The **classic routes** (`/api/upload`, `/api/scan-pfm`, `/api/parse`, `/api/history`, etc.) and the root/`scan-pfm`/`manual-label` pages deliberately do **not** check auth at all and never will unless that decision changes — they're dev-only web UI with no login screen, not part of the production surface (see `plans/next-enhancements.md` task 1.3, cancelled, and 1.4, shipped instead). - **Real JWT auth, enforced on the production surface**: `src/utils/auth.ts` signs/verifies tokens (`signAccountToken`/`verifyAccountToken`/`getAccountFromAuthHeader`) against an `accounts` table, each account bound to exactly one `kode_toko` (store) — the intent being that an account's own store is used on upload instead of relying on OCR-based store-text matching. **Passwords are bcrypt-hashed** (`accounts.password`, via `bcryptjs` — chosen over native `bcrypt` since the `pfm-web-app` Docker stage is `node:20-slim` with no build toolchain for native addons; `pfm-web-app/src/db/init.ts` hashes the seed and idempotently migrates any pre-existing plaintext rows on every startup). As of 2026-07-08, `api/v1/documents/*` (list, PUT-by-id, upload) **reject requests with a missing/invalid token (401)** — this is the real production surface, and the Flutter client already does a real login and attaches `Authorization: Bearer <token>` to every request (`lib/features/auth/auth_provider.dart` + `lib/core/network/api_client.dart`). The **classic routes** (`/api/upload`, `/api/scan-pfm`, `/api/parse`, `/api/history`, etc.) and the root/`scan-pfm`/`manual-label` pages deliberately do **not** check auth at all and never will unless that decision changes — they're dev-only web UI with no login screen, not part of the production surface (see `plans/next-enhancements.md` task 1.3, cancelled, and 1.4, shipped instead).
- **Richer SKU master data**: `pfm-web-app/import_sku.js` imports from a TSV with extended packaging columns (`standar_jumlah`, `berat_kemasan`, `isi_outer_kg`, `isi_outer_pac`, `jenis_outer`) added via `ALTER TABLE sku_master ADD COLUMN IF NOT EXISTS`, superseding the old project's bare `no_sku`/`nama_item` seed list. - **Richer SKU master data**: `pfm-web-app/import_sku.js` imports from a TSV with extended packaging columns (`standar_jumlah`, `berat_kemasan`, `isi_outer_kg`, `isi_outer_pac`, `jenis_outer`) added via `ALTER TABLE sku_master ADD COLUMN IF NOT EXISTS`, superseding the old project's bare `no_sku`/`nama_item` seed list.
- **Accuracy regression harness** (new, doesn't exist in the old project): `pfm-web-app/scripts/accuracy-check.mts` hits the live `/api/parse` endpoint for every image in `sources/test-images/`, diffs against hand-labeled ground truth in `sources/manual_labels.json` at three post-processing stages (`layer1RawRegex` → `layer2Sanitized` → `layer3Final` — trace these stage names into `utils/parser.ts` to see where each is produced), and appends run-over-run results to `sources/accuracy_history.jsonl`. Run this after touching `parser.ts` to check for regressions: - **Accuracy regression harness** (new, doesn't exist in the old project): `pfm-web-app/scripts/accuracy-check.mts` hits the live `/api/parse` endpoint for every image in `sources/test-images/`, diffs against hand-labeled ground truth in `sources/manual_labels.json` at three post-processing stages (`layer1RawRegex` → `layer2Sanitized` → `layer3Final` — trace these stage names into `utils/parser.ts` to see where each is produced), and appends run-over-run results to `sources/accuracy_history.jsonl`. Run this after touching `parser.ts` to check for regressions:
```bash ```bash
node pfm-web-app/scripts/accuracy-check.mts # reuse cached OCR (fast) node pfm-web-app/scripts/accuracy-check.mts # reuse cached OCR (fast)
node pfm-web-app/scripts/accuracy-check.mts --refresh-ocr # force fresh pipeline run node pfm-web-app/scripts/accuracy-check.mts --refresh-ocr # force fresh pipeline run
node pfm-web-app/scripts/accuracy-check.mts --detail <filename> # full per-stage breakdown for one image node pfm-web-app/scripts/accuracy-check.mts --detail <filename> # full per-stage breakdown for one image
``` ```
`compare_accuracy.py` / `compare_sources_accuracy.py` / `generate_excel.py` at the repo root build human-readable Excel/HTML comparison reports from the same data (`sources/comparison_report.xlsx`, `sources/comparison_side_by_side.html`) — these are analysis tooling, not part of the running app. `compare_accuracy.py` / `compare_sources_accuracy.py` / `generate_excel.py` at the repo root build human-readable Excel/HTML comparison reports from the same data (`sources/comparison_report.xlsx`, `sources/comparison_side_by_side.html`) — these are analysis tooling, not part of the running app.
- **`api/vllm-proxy/[[...path]]/route.ts`**: a passthrough proxy to the vLLM server (`paddleocr-vllm-server:8118`) that logs every call via `logVllmCallToAll` (`utils/active-log.ts`) — used for debugging/observability, not part of the OCR pipeline itself. - **`api/vllm-proxy/[[...path]]/route.ts`**: a passthrough proxy to the vLLM server (`paddleocr-vllm-server:8118`) that logs every call via `logVllmCallToAll` (`utils/active-log.ts`) — used for debugging/observability, not part of the OCR pipeline itself.
- **`docker-compose.override.yml`** exposes `db` (`5432`) and `pipeline-api` (`8090`) directly to the host for local dev — not present in the old project's compose setup. - **`docker-compose.override.yml`** exposes `db` (`5432`) and `pipeline-api` (`8090`) directly to the host for local dev — not present in the old project's compose setup.
## Product/SKU scanning flow — status ## Product/SKU scanning flow — status
**How it works end-to-end** (architecture, endpoints, classification/OCR internals, retraining): [`docs/scan-product.md`](docs/scan-product.md). See [`plans/next-enhancements.md`](plans/next-enhancements.md) §2 (task 2.1) for full detail — kept there instead of a separate doc so status stays traceable against the rest of the `e`/`n` backlog. **Feature-complete as of 2026-07-08**: the backend (`config/classify_ocr_server.py` with DINOv2 similarity search + YOLO classifier fallback, `api/scan-pfm/route.ts`, `api/produk-pfm/route.ts`, DB schema), the reference photo dataset (`pfm-web-app/public/produk-pfm/foto-kemasan-v2/`, 81 SKU subfolders as of 2026-07-14, up from the original 16 — target ~230), the desktop frontend page (`scan-pfm/page.tsx`, full feature parity), and the trained model artifacts (`models/dinov2_index.pkl` — 2,493/2,493 photos indexed as of 2026-07-14; `models/produk-pfm-classifier-26n-100e-2026-07-14.pt` — 85.8% top-1 / 94.4% top-5 val accuracy across all 81 classes, retrained 2026-07-14 in 54m21s on an RTX 2060) all now exist and load cleanly on `pipeline-api` startup. **No mobile web page is planned**: `scan-pfm/page.tsx` is desktop-only, used to test the pipeline; real mobile product scanning goes through the Flutter app instead, so `m-scan-pfm/page.tsx` and its `nginx.conf` route are intentionally left unbuilt/dead (see plan task 2.2, cancelled 2026-07-08). Not yet done: an actual browser pass uploading a photo through `/scan-pfm` end-to-end (verified via container logs/model-loading so far, not a UI test). **How it works end-to-end** (architecture, endpoints, classification/OCR internals, retraining): [`docs/scan-product.md`](docs/scan-product.md). See [`plans/next-enhancements.md`](plans/next-enhancements.md) §2 (task 2.1) for full detail — kept there instead of a separate doc so status stays traceable against the rest of the `e`/`n` backlog. **Feature-complete as of 2026-07-08**: the backend (`config/classify_ocr_server.py` with DINOv2 similarity search + YOLO classifier fallback, `api/scan-pfm/route.ts`, `api/produk-pfm/route.ts`, DB schema), the reference photo dataset (`pfm-web-app/public/produk-pfm/foto-kemasan-v2/`, 81 SKU subfolders as of 2026-07-14, up from the original 16 — target ~230), the desktop frontend page (`scan-pfm/page.tsx`, full feature parity), and the trained model artifacts (`models/dinov2_index.pkl` — 2,493/2,493 photos indexed as of 2026-07-14; `models/produk-pfm-classifier-26n-100e-2026-07-14.pt` — 85.8% top-1 / 94.4% top-5 val accuracy across all 81 classes, retrained 2026-07-14 in 54m21s on an RTX 2060) all now exist and load cleanly on `pipeline-api` startup. **No mobile web page is planned**: `scan-pfm/page.tsx` is desktop-only, used to test the pipeline; real mobile product scanning goes through the Flutter app instead, so `m-scan-pfm/page.tsx` and its `nginx.conf` route are intentionally left unbuilt/dead (see plan task 2.2, cancelled 2026-07-08). Not yet done: an actual browser pass uploading a photo through `/scan-pfm` end-to-end (verified via container logs/model-loading so far, not a UI test).
## Confidentiality ## Confidentiality
Same concerns as `ai-ocr-pfm-2026` apply here, plus more surface area: Same concerns as `ai-ocr-pfm-2026` apply here, plus more surface area:
- `pfm-web-app/src/db/init.ts` and `db/migrations/005_create_sku_master.sql` contain the client's real product catalog and real vendor/customer identities, committed directly in source. - `pfm-web-app/src/db/init.ts` and `db/migrations/005_create_sku_master.sql` contain the client's real product catalog and real vendor/customer identities, committed directly in source.
- `sources/` holds live business data: `Rekap SKU Aktif CPI Cikande per April 2026 v2.xlsx`, `Tabel Toko Aktif Juni 2026.xlsx`, `toko_aktif.json`, `manual_labels.json`, `ai_results.json` — real SKU/store master data and hand-labeled ground truth from real scanned documents, not fixtures. - `sources/` holds live business data: `Rekap SKU Aktif CPI Cikande per April 2026 v2.xlsx`, `Tabel Toko Aktif Juni 2026.xlsx`, `toko_aktif.json`, `manual_labels.json`, `ai_results.json` — real SKU/store master data and hand-labeled ground truth from real scanned documents, not fixtures.
- `uploads/` contains real scanned delivery-order photos and their OCR JSON output. - `uploads/` contains real scanned delivery-order photos and their OCR JSON output.
- The `accounts` table stores bcrypt-hashed passwords as of 2026-07-08 (see above) — still don't log or export its contents, and it's not wired into most routes yet (task 1.3), so don't treat it as a secure boundary for anything beyond the `api/v1/*` REST layer. - The `accounts` table stores bcrypt-hashed passwords as of 2026-07-08 (see above) — still don't log or export its contents, and it's not wired into most routes yet (task 1.3), so don't treat it as a secure boundary for anything beyond the `api/v1/*` REST layer.
## Commands ## Commands
Web app (`pfm-web-app/`): Web app (`pfm-web-app/`):
```bash ```bash
npm run dev # next dev -H 0.0.0.0 (binds all interfaces — for LAN/tunnel access during mobile testing) npm run dev # next dev -H 0.0.0.0 (binds all interfaces — for LAN/tunnel access during mobile testing)
npm run build npm run build
npm run start npm run start
npm run lint npm run lint
``` ```
Accuracy regression check (see above) — run after any `parser.ts` change: Accuracy regression check (see above) — run after any `parser.ts` change:
```bash ```bash
node pfm-web-app/scripts/accuracy-check.mts node pfm-web-app/scripts/accuracy-check.mts
``` ```
`pfm-web-app/src/utils/parser.test.ts` — same standalone `node:assert` script as the old project, covering `parseDOMetadata`/`sanitizeParsedMetadata`. Run with a TS-capable runner, e.g. `npx tsx pfm-web-app/src/utils/parser.test.ts`. `pfm-web-app/src/utils/parser.test.ts` — same standalone `node:assert` script as the old project, covering `parseDOMetadata`/`sanitizeParsedMetadata`. Run with a TS-capable runner, e.g. `npx tsx pfm-web-app/src/utils/parser.test.ts`.
Python services (uv-managed, same as `ai-ocr-pfm-2026` — see `AGENTS.md`): Python services (uv-managed, same as `ai-ocr-pfm-2026` — see `AGENTS.md`):
```bash ```bash
./scripts/install.sh # bootstrap .venv for vLLM server ./scripts/install.sh # bootstrap .venv for vLLM server
./scripts/install-pipeline.sh # bootstrap .venv-api ./scripts/install-pipeline.sh # bootstrap .venv-api
./scripts/serve.sh # vLLM genai server on :8118 ./scripts/serve.sh # vLLM genai server on :8118
./scripts/serve-pipeline.sh # pipeline API on :8090 + classify_ocr_server.py on :8120 ./scripts/serve-pipeline.sh # pipeline API on :8090 + classify_ocr_server.py on :8120
``` ```
Full stack: Full stack:
```bash ```bash
docker compose up -d --build docker compose up -d --build
``` ```
## Agents Settings Kit (backend-scoped) ## Agents Settings Kit (backend-scoped)
@AGENTS.md @AGENTS.md
`AGENTS.md` in this directory now has two parts: Part A is the pre-existing vLLM `AGENTS.md` in this directory now has two parts: Part A is the pre-existing vLLM
service doc referenced above; Part B (appended 2026-07-08) is a **backend-scoped service doc referenced above; Part B (appended 2026-07-08) is a **backend-scoped
copy** of the [fhanyuh/agents-settings](https://github.com/fhanyuh/agents-settings) copy** of the [fhanyuh/agents-settings](https://github.com/fhanyuh/agents-settings)
`e`/`enhance`/`n`/`next` workflow, independent of the root-level copy that covers `e`/`enhance`/`n`/`next` workflow, independent of the root-level copy that covers
the Flutter side (see root `CLAUDE.md`/`AGENTS.md`). Roles are in `SKILLS.md` (this the Flutter side (see root `CLAUDE.md`/`AGENTS.md`). Roles are in `SKILLS.md` (this
dir). The backlog and shipped-feature log live in `plans/next-enhancements.md` and dir). The backlog and shipped-feature log live in `plans/next-enhancements.md` and
`docs/feature-list.md` (this dir) — these are backend-only and separate from the `docs/feature-list.md` (this dir) — these are backend-only and separate from the
root project's equivalents, which now only track Flutter work. root project's equivalents, which now only track Flutter work.
Claude-specific notes (same as root): Claude-specific notes (same as root):
- Spawn the relevant `SKILLS.md` role via the `Agent` tool for a fresh-context - Spawn the relevant `SKILLS.md` role via the `Agent` tool for a fresh-context
review/QA/architecture pass instead of continuing in the implementing context. review/QA/architecture pass instead of continuing in the implementing context.
- Use `AskUserQuestion` for the one-at-a-time clarification step (§B2a). - Use `AskUserQuestion` for the one-at-a-time clarification step (§B2a).
- Use `EnterPlanMode` before writing code for any `n`/`next` task that touches - Use `EnterPlanMode` before writing code for any `n`/`next` task that touches
multiple files or has more than one reasonable implementation approach. multiple files or has more than one reasonable implementation approach.
+83 -83
View File
@@ -1,83 +1,83 @@
# Stage 0: GPU Base image # Stage 0: GPU Base image
FROM nvidia/cuda:12.6.0-devel-ubuntu22.04 AS base-gpu FROM nvidia/cuda:12.6.0-devel-ubuntu22.04 AS base-gpu
ENV DEBIAN_FRONTEND=noninteractive ENV DEBIAN_FRONTEND=noninteractive
ENV PATH="/root/.local/bin:$PATH" ENV PATH="/root/.local/bin:$PATH"
# Install system dependencies (libgl and libglib are required for OpenCV) # Install system dependencies (libgl and libglib are required for OpenCV)
RUN apt-get update && apt-get install -y --no-install-recommends \ RUN apt-get update && apt-get install -y --no-install-recommends \
curl \ curl \
git \ git \
libgl1 \ libgl1 \
libglib2.0-0 \ libglib2.0-0 \
&& rm -rf /var/lib/apt/lists/* && rm -rf /var/lib/apt/lists/*
# Install uv # Install uv
RUN curl -LsSf https://astral.sh/uv/install.sh | sh RUN curl -LsSf https://astral.sh/uv/install.sh | sh
# --- vLLM Server Stage --- # --- vLLM Server Stage ---
FROM base-gpu AS vllm-server FROM base-gpu AS vllm-server
WORKDIR /app WORKDIR /app
# Install project dependencies # Install project dependencies
COPY pyproject.toml uv.lock ./ COPY pyproject.toml uv.lock ./
RUN uv python pin 3.12 && uv sync --frozen --no-dev RUN uv python pin 3.12 && uv sync --frozen --no-dev
# Install prebuilt flash-attention wheel # Install prebuilt flash-attention wheel
ARG FLASH_ATTN_WHEEL=https://github.com/mjun0812/flash-attention-prebuild-wheels/releases/download/v0.3.14/flash_attn-2.8.2+cu128torch2.8-cp312-cp312-linux_x86_64.whl ARG FLASH_ATTN_WHEEL=https://github.com/mjun0812/flash-attention-prebuild-wheels/releases/download/v0.3.14/flash_attn-2.8.2+cu128torch2.8-cp312-cp312-linux_x86_64.whl
RUN uv pip install --python .venv "${FLASH_ATTN_WHEEL}" RUN uv pip install --python .venv "${FLASH_ATTN_WHEEL}"
COPY . /app COPY . /app
RUN chmod +x /app/scripts/serve.sh RUN chmod +x /app/scripts/serve.sh
EXPOSE 8118 EXPOSE 8118
CMD ["./scripts/serve.sh"] CMD ["./scripts/serve.sh"]
# --- Pipeline API Stage --- # --- Pipeline API Stage ---
FROM base-gpu AS pipeline-api FROM base-gpu AS pipeline-api
WORKDIR /app WORKDIR /app
# Build paddlepaddle and paddlex virtual env # Build paddlepaddle and paddlex virtual env
RUN uv venv .venv-api --python 3.12 RUN uv venv .venv-api --python 3.12
RUN uv pip install --python .venv-api paddlepaddle-gpu -i https://www.paddlepaddle.org.cn/packages/stable/cu126/ RUN uv pip install --python .venv-api paddlepaddle-gpu -i https://www.paddlepaddle.org.cn/packages/stable/cu126/
RUN uv pip install --python .venv-api "paddleocr[doc-parser]>=3.3.0" RUN uv pip install --python .venv-api "paddleocr[doc-parser]>=3.3.0"
RUN uv pip install --python .venv-api "aiohttp>=3.9" "filetype>=1.2" "fastapi>=0.110" "starlette>=0.36" "uvicorn>=0.16" "ultralytics>=8.0" RUN uv pip install --python .venv-api "aiohttp>=3.9" "filetype>=1.2" "fastapi>=0.110" "starlette>=0.36" "uvicorn>=0.16" "ultralytics>=8.0"
COPY . /app COPY . /app
RUN chmod +x /app/scripts/serve-pipeline.sh RUN chmod +x /app/scripts/serve-pipeline.sh
EXPOSE 8090 EXPOSE 8090
CMD ["./scripts/serve-pipeline.sh"] CMD ["./scripts/serve-pipeline.sh"]
# --- Gradio UI Stage --- # --- Gradio UI Stage ---
FROM python:3.12-slim AS gradio-ui FROM python:3.12-slim AS gradio-ui
WORKDIR /app/PaddleOCR-VL-1.6_Online_Demo WORKDIR /app/PaddleOCR-VL-1.6_Online_Demo
RUN apt-get update && apt-get install -y --no-install-recommends \ RUN apt-get update && apt-get install -y --no-install-recommends \
curl \ curl \
&& rm -rf /var/lib/apt/lists/* && rm -rf /var/lib/apt/lists/*
COPY PaddleOCR-VL-1.6_Online_Demo/requirements.txt ./ COPY PaddleOCR-VL-1.6_Online_Demo/requirements.txt ./
RUN pip install --no-cache-dir -r requirements.txt RUN pip install --no-cache-dir -r requirements.txt
COPY PaddleOCR-VL-1.6_Online_Demo ./ COPY PaddleOCR-VL-1.6_Online_Demo ./
EXPOSE 7870 EXPOSE 7870
ENV GRADIO_SERVER_NAME="0.0.0.0" ENV GRADIO_SERVER_NAME="0.0.0.0"
ENV GRADIO_SERVER_PORT="7870" ENV GRADIO_SERVER_PORT="7870"
CMD ["python", "app.py"] CMD ["python", "app.py"]
# --- Next.js Web App Stage --- # --- Next.js Web App Stage ---
FROM node:20-slim AS pfm-web-app FROM node:20-slim AS pfm-web-app
WORKDIR /app WORKDIR /app
COPY pfm-web-app/package.json pfm-web-app/package-lock.json ./ COPY pfm-web-app/package.json pfm-web-app/package-lock.json ./
ENV PUPPETEER_SKIP_DOWNLOAD=true ENV PUPPETEER_SKIP_DOWNLOAD=true
RUN npm ci RUN npm ci
COPY pfm-web-app/ ./ COPY pfm-web-app/ ./
ENV NODE_ENV=production ENV NODE_ENV=production
RUN npm run build RUN npm run build
EXPOSE 3000 EXPOSE 3000
CMD ["npm", "start"] CMD ["npm", "start"]
+151 -151
View File
@@ -1,151 +1,151 @@
# PaddleOCR-VL-1.6 on vLLM # PaddleOCR-VL-1.6 on vLLM
Local deployment of [PaddleOCR-VL-1.6](https://www.paddleocr.ai/latest/en/version3.x/pipeline_usage/PaddleOCR-VL.html) using **vLLM** as the VLM inference backend. All Python workflows use **[uv](https://docs.astral.sh/uv/)**. Local deployment of [PaddleOCR-VL-1.6](https://www.paddleocr.ai/latest/en/version3.x/pipeline_usage/PaddleOCR-VL.html) using **vLLM** as the VLM inference backend. All Python workflows use **[uv](https://docs.astral.sh/uv/)**.
## Architecture ## Architecture
``` ```
Gradio demo (7870) Gradio demo (7870)
│ │
▼ ▼
Pipeline API (8090) ── layout + preprocessing (PaddlePaddle GPU) Pipeline API (8090) ── layout + preprocessing (PaddlePaddle GPU)
│ │
▼ ▼
vLLM genai server (8118) ── PaddleOCR-VL-1.6 VLM vLLM genai server (8118) ── PaddleOCR-VL-1.6 VLM
``` ```
| Service | Script | Default URL | | Service | Script | Default URL |
|---------|--------|-------------| |---------|--------|-------------|
| vLLM VLM server | `./scripts/serve.sh` | `http://127.0.0.1:8118/v1` | | vLLM VLM server | `./scripts/serve.sh` | `http://127.0.0.1:8118/v1` |
| Full pipeline API | `./scripts/serve-pipeline.sh` | `http://127.0.0.1:8090/layout-parsing` | | Full pipeline API | `./scripts/serve-pipeline.sh` | `http://127.0.0.1:8090/layout-parsing` |
| Online demo UI | `./scripts/run-demo.sh` | `http://127.0.0.1:7870` | | Online demo UI | `./scripts/run-demo.sh` | `http://127.0.0.1:7870` |
The vLLM server exposes only the VLM stage. For HTTP document parsing (layout + OCR), run the pipeline API, which calls vLLM via `config/pipeline_config_vllm.yaml`. The vLLM server exposes only the VLM stage. For HTTP document parsing (layout + OCR), run the pipeline API, which calls vLLM via `config/pipeline_config_vllm.yaml`.
## Prerequisites ## Prerequisites
- Linux with NVIDIA GPU (CC ≥ 8.0 recommended; CUDA 12.6+ driver) - Linux with NVIDIA GPU (CC ≥ 8.0 recommended; CUDA 12.6+ driver)
- [uv](https://docs.astral.sh/uv/) installed - [uv](https://docs.astral.sh/uv/) installed
- ~16 GB GPU VRAM for default vLLM settings (tune in `config/vllm_config.yaml`) - ~16 GB GPU VRAM for default vLLM settings (tune in `config/vllm_config.yaml`)
## Quick start ## Quick start
```bash ```bash
git clone <repo-url> ai-ocr-pfm-2026 git clone <repo-url> ai-ocr-pfm-2026
cd ai-ocr-pfm-2026 cd ai-ocr-pfm-2026
cp .env.example .env # adjust CUDA_VISIBLE_DEVICES if needed cp .env.example .env # adjust CUDA_VISIBLE_DEVICES if needed
# 1) Install vLLM server (.venv) # 1) Install vLLM server (.venv)
FLASH_ATTN_WHEEL="https://github.com/mjun0812/flash-attention-prebuild-wheels/releases/download/v0.3.14/flash_attn-2.8.2+cu128torch2.8-cp312-cp312-linux_x86_64.whl" \ FLASH_ATTN_WHEEL="https://github.com/mjun0812/flash-attention-prebuild-wheels/releases/download/v0.3.14/flash_attn-2.8.2+cu128torch2.8-cp312-cp312-linux_x86_64.whl" \
./scripts/install.sh ./scripts/install.sh
# 2) Install pipeline API (.venv-api) — optional, needed for demo / full HTTP API # 2) Install pipeline API (.venv-api) — optional, needed for demo / full HTTP API
./scripts/install-pipeline.sh ./scripts/install-pipeline.sh
``` ```
Start services (three terminals, or background each): Start services (three terminals, or background each):
```bash ```bash
./scripts/serve.sh # vLLM on :8118 ./scripts/serve.sh # vLLM on :8118
./scripts/serve-pipeline.sh # pipeline on :8090 ./scripts/serve-pipeline.sh # pipeline on :8090
./scripts/run-demo.sh # Gradio on :7870 ./scripts/run-demo.sh # Gradio on :7870
``` ```
Health checks: Health checks:
```bash ```bash
curl -s http://127.0.0.1:8118/v1/models | jq . curl -s http://127.0.0.1:8118/v1/models | jq .
curl -s http://127.0.0.1:8090/health curl -s http://127.0.0.1:8090/health
curl -s -o /dev/null -w "%{http_code}\n" http://127.0.0.1:7870/ curl -s -o /dev/null -w "%{http_code}\n" http://127.0.0.1:7870/
``` ```
## Configuration ## Configuration
Copy `.env.example` to `.env`: Copy `.env.example` to `.env`:
| Variable | Default | Description | | Variable | Default | Description |
|----------|---------|-------------| |----------|---------|-------------|
| `GENAI_HOST` | `0.0.0.0` | vLLM bind address | | `GENAI_HOST` | `0.0.0.0` | vLLM bind address |
| `GENAI_PORT` | `8118` | vLLM port | | `GENAI_PORT` | `8118` | vLLM port |
| `GENAI_MODEL` | `PaddleOCR-VL-1.6-0.9B` | Model name | | `GENAI_MODEL` | `PaddleOCR-VL-1.6-0.9B` | Model name |
| `VLLM_CONFIG` | `config/vllm_config.yaml` | vLLM tuning | | `VLLM_CONFIG` | `config/vllm_config.yaml` | vLLM tuning |
| `CUDA_VISIBLE_DEVICES` | `1` | GPU for vLLM (use least-busy GPU) | | `CUDA_VISIBLE_DEVICES` | `1` | GPU for vLLM (use least-busy GPU) |
| `PIPELINE_PORT` | `8090` | Pipeline API port | | `PIPELINE_PORT` | `8090` | Pipeline API port |
| `PIPELINE_DEVICE` | `gpu:0` | GPU for layout/preprocessing | | `PIPELINE_DEVICE` | `gpu:0` | GPU for layout/preprocessing |
| `GRADIO_PORT` | `7870` | Demo UI port | | `GRADIO_PORT` | `7870` | Demo UI port |
vLLM tuning (`config/vllm_config.yaml`): vLLM tuning (`config/vllm_config.yaml`):
```yaml ```yaml
gpu-memory-utilization: 0.75 gpu-memory-utilization: 0.75
max-num-seqs: 128 max-num-seqs: 128
``` ```
## Client usage ## Client usage
### Python (vLLM only) ### Python (vLLM only)
```python ```python
from paddleocr import PaddleOCRVL from paddleocr import PaddleOCRVL
pipeline = PaddleOCRVL( pipeline = PaddleOCRVL(
vl_rec_backend="vllm-server", vl_rec_backend="vllm-server",
vl_rec_server_url="http://127.0.0.1:8118/v1", vl_rec_server_url="http://127.0.0.1:8118/v1",
) )
output = pipeline.predict("path/to/image.png") output = pipeline.predict("path/to/image.png")
``` ```
Run the client in a **separate** environment if it needs PaddlePaddle GPU alongside Transformers. Run the client in a **separate** environment if it needs PaddlePaddle GPU alongside Transformers.
### CLI ### CLI
```bash ```bash
uv run paddleocr doc_parser \ uv run paddleocr doc_parser \
--input demo.png \ --input demo.png \
--vl_rec_backend vllm-server \ --vl_rec_backend vllm-server \
--vl_rec_server_url http://127.0.0.1:8118/v1 --vl_rec_server_url http://127.0.0.1:8118/v1
``` ```
### HTTP (full pipeline) ### HTTP (full pipeline)
```bash ```bash
curl -X POST http://127.0.0.1:8090/layout-parsing \ curl -X POST http://127.0.0.1:8090/layout-parsing \
-H "Content-Type: application/json" \ -H "Content-Type: application/json" \
-d '{"file":"<base64>", "fileType": 1, "useLayoutDetection": true}' -d '{"file":"<base64>", "fileType": 1, "useLayoutDetection": true}'
``` ```
## Project layout ## Project layout
``` ```
config/ config/
vllm_config.yaml # vLLM backend tuning vllm_config.yaml # vLLM backend tuning
pipeline_config_vllm.yaml # pipeline → vLLM server URL pipeline_config_vllm.yaml # pipeline → vLLM server URL
scripts/ scripts/
install.sh # bootstrap .venv (vLLM) install.sh # bootstrap .venv (vLLM)
install-pipeline.sh # bootstrap .venv-api (pipeline) install-pipeline.sh # bootstrap .venv-api (pipeline)
serve.sh # start vLLM genai server serve.sh # start vLLM genai server
serve-pipeline.sh # start pipeline API serve-pipeline.sh # start pipeline API
run-demo.sh # start Gradio demo run-demo.sh # start Gradio demo
PaddleOCR-VL-1.6_Online_Demo/ # bundled Hugging Face-style demo PaddleOCR-VL-1.6_Online_Demo/ # bundled Hugging Face-style demo
issues/ # recorded problems and fixes issues/ # recorded problems and fixes
AGENTS.md # agent / contributor guide AGENTS.md # agent / contributor guide
``` ```
## Troubleshooting ## Troubleshooting
See [issues/](issues/) for detailed write-ups. Common fixes: See [issues/](issues/) for detailed write-ups. Common fixes:
| Symptom | Fix | | Symptom | Fix |
|---------|-----| |---------|-----|
| GPU OOM on vLLM startup | Lower `gpu-memory-utilization` or set `CUDA_VISIBLE_DEVICES` to a free GPU | | GPU OOM on vLLM startup | Lower `gpu-memory-utilization` or set `CUDA_VISIBLE_DEVICES` to a free GPU |
| flash-attn build failure | Use prebuilt wheel via `FLASH_ATTN_WHEEL=... ./scripts/install.sh` | | flash-attn build failure | Use prebuilt wheel via `FLASH_ATTN_WHEEL=... ./scripts/install.sh` |
| Port 8080 in use | Pipeline defaults to **8090**; demo defaults to **7870** | | Port 8080 in use | Pipeline defaults to **8090**; demo defaults to **7870** |
Agent conventions and issue-recording rules: [AGENTS.md](AGENTS.md). Agent conventions and issue-recording rules: [AGENTS.md](AGENTS.md).
## References ## References
- [PaddleOCR-VL usage tutorial](https://www.paddleocr.ai/latest/en/version3.x/pipeline_usage/PaddleOCR-VL.html) - [PaddleOCR-VL usage tutorial](https://www.paddleocr.ai/latest/en/version3.x/pipeline_usage/PaddleOCR-VL.html)
- [PaddleOCR genai_server FAQ](https://github.com/PaddlePaddle/PaddleOCR/discussions/16822) - [PaddleOCR genai_server FAQ](https://github.com/PaddlePaddle/PaddleOCR/discussions/16822)
- [flash-attention prebuild wheels](https://mjunya.com/flash-attention-prebuild-wheels/) - [flash-attention prebuild wheels](https://mjunya.com/flash-attention-prebuild-wheels/)
+102 -102
View File
@@ -1,102 +1,102 @@
# Skills & Roles (backend) # Skills & Roles (backend)
Backend-scoped copy of the root `SKILLS.md` — same five roles, applied to Backend-scoped copy of the root `SKILLS.md` — same five roles, applied to
`backend/` surfaces (Next.js API gateway, OCR pipeline, Postgres, Docker/deploy) `backend/` surfaces (Next.js API gateway, OCR pipeline, Postgres, Docker/deploy)
during `n`/`next` execution (see `AGENTS.md` Part B, this dir). One agent can play during `n`/`next` execution (see `AGENTS.md` Part B, this dir). One agent can play
all of them in sequence; a multi-agent harness may spawn each as a separate all of them in sequence; a multi-agent harness may spawn each as a separate
subagent for a fresh-context pass. Order matters: Architect → Backend/Frontend → subagent for a fresh-context pass. Order matters: Architect → Backend/Frontend →
QA → Hardware/Compatibility. QA → Hardware/Compatibility.
## 1. Software Architect ## 1. Software Architect
**Responsibilities** **Responsibilities**
- Decide where new backend code lives; keep module boundaries clean (API routes vs. - Decide where new backend code lives; keep module boundaries clean (API routes vs.
`utils/` business logic vs. `db/` layer vs. the Python pipeline in `config/`). `utils/` business logic vs. `db/` layer vs. the Python pipeline in `config/`).
- Prefer deep modules (few, well-bounded files with simple interfaces) over shallow - Prefer deep modules (few, well-bounded files with simple interfaces) over shallow
ones — this is what keeps the codebase navigable for an agent. ones — this is what keeps the codebase navigable for an agent.
- Own the 256-LOC split rule (`AGENTS.md` Part B §B3): when a file crosses the - Own the 256-LOC split rule (`AGENTS.md` Part B §B3): when a file crosses the
threshold, decide the split boundary before anyone patches around it. threshold, decide the split boundary before anyone patches around it.
- Keep `plans/next-enhancements.md` (this dir) structured by real backend module - Keep `plans/next-enhancements.md` (this dir) structured by real backend module
boundaries, not arbitrary groupings. boundaries, not arbitrary groupings.
- Owns the existing-project audit (`AGENTS.md` Part B §B0) for backend specifically. - Owns the existing-project audit (`AGENTS.md` Part B §B0) for backend specifically.
**When invoked**: start of every `e`/`enhance` run; start of every `n`/`next` task, **When invoked**: start of every `e`/`enhance` run; start of every `n`/`next` task,
before implementation begins. before implementation begins.
**Handoff**: hands the Backend/Frontend roles a target file layout and interface **Handoff**: hands the Backend/Frontend roles a target file layout and interface
contract, not just a task description. contract, not just a task description.
## 2. Backend Engineer ## 2. Backend Engineer
**Responsibilities** **Responsibilities**
- Implement Next.js API route logic, DB access (`src/db/`), and the Python OCR - Implement Next.js API route logic, DB access (`src/db/`), and the Python OCR
pipeline (`config/classify_ocr_server.py`, pipeline API) as the task requires. pipeline (`config/classify_ocr_server.py`, pipeline API) as the task requires.
- Wire the mock-vs-live routing required by the Demo/Live switch (`AGENTS.md` Part B - Wire the mock-vs-live routing required by the Demo/Live switch (`AGENTS.md` Part B
§B5) and the Cloud/Local endpoint switch (§B6) if/when built — both must resolve §B5) and the Cloud/Local endpoint switch (§B6) if/when built — both must resolve
through the same contract so swapping either setting never changes calling code. through the same contract so swapping either setting never changes calling code.
- Keep business logic out of route handlers (`src/app/api/**/route.ts`); route - Keep business logic out of route handlers (`src/app/api/**/route.ts`); route
handlers stay thin, matching the existing `utils/parser.ts`-style separation. handlers stay thin, matching the existing `utils/parser.ts`-style separation.
- Use `withTransaction` (`src/db/index.ts`) for any multi-statement write that must - Use `withTransaction` (`src/db/index.ts`) for any multi-statement write that must
be atomic — see task 7.1 in the pre-kit history for why this matters here. be atomic — see task 7.1 in the pre-kit history for why this matters here.
**When invoked**: any task touching API routes, the DB layer, or the OCR pipeline. **When invoked**: any task touching API routes, the DB layer, or the OCR pipeline.
**Handoff**: gives Frontend a stable contract (types/response shape) to build **Handoff**: gives Frontend a stable contract (types/response shape) to build
against; gives QA the list of new/changed endpoints and their expected error modes. against; gives QA the list of new/changed endpoints and their expected error modes.
## 3. Frontend Engineer ## 3. Frontend Engineer
**Responsibilities** **Responsibilities**
- Implement UI for the task inside `pfm-web-app/src/app/`, including Demo/Live and - Implement UI for the task inside `pfm-web-app/src/app/`, including Demo/Live and
Cloud/Local switcher controls where relevant. Cloud/Local switcher controls where relevant.
- Follow this repo's existing page pattern: consolidated single-page flows (root - Follow this repo's existing page pattern: consolidated single-page flows (root
`page.tsx`) vs. standalone-purpose route folders (`manual-label/page.tsx`, `page.tsx`) vs. standalone-purpose route folders (`manual-label/page.tsx`,
`scan-pfm/page.tsx`) — see backend `CLAUDE.md` for which pattern a given feature `scan-pfm/page.tsx`) — see backend `CLAUDE.md` for which pattern a given feature
should follow. should follow.
- Consume the Backend Engineer's contract rather than reaching around it. - Consume the Backend Engineer's contract rather than reaching around it.
- Keep components small and composable, respecting the 256-LOC rule. - Keep components small and composable, respecting the 256-LOC rule.
**When invoked**: any task with a user-facing surface inside `pfm-web-app/`. **When invoked**: any task with a user-facing surface inside `pfm-web-app/`.
**Handoff**: gives QA the golden-path user flow and the edge cases it's aware of. **Handoff**: gives QA the golden-path user flow and the edge cases it's aware of.
## 4. QA / Test Engineer ## 4. QA / Test Engineer
**Responsibilities** **Responsibilities**
- During clarification (`AGENTS.md` Part B §B2a), turn resolved answers into - During clarification (`AGENTS.md` Part B §B2a), turn resolved answers into
concrete acceptance criteria — what "done" verifiably means. concrete acceptance criteria — what "done" verifiably means.
- Write/extend automated tests (`parser.test.ts` pattern) for the change. - Write/extend automated tests (`parser.test.ts` pattern) for the change.
- For anything touching `parser.ts` or the OCR pipeline, run the accuracy - For anything touching `parser.ts` or the OCR pipeline, run the accuracy
regression harness (`node pfm-web-app/scripts/accuracy-check.mts` or `accuracy-check-scan.mts`) and check for regression harness (`node pfm-web-app/scripts/accuracy-check.mts` or `accuracy-check-scan.mts`) and check for
regressions against the current baseline (see `sources/accuracy_history.jsonl` and `sources/product_accuracy_history.jsonl` for latest metrics), not just "it compiles." regressions against the current baseline (see `sources/accuracy_history.jsonl` and `sources/product_accuracy_history.jsonl` for latest metrics), not just "it compiles."
- Run the **verify build integrity** pass: golden path + edge cases + regression - Run the **verify build integrity** pass: golden path + edge cases + regression
check on adjacent features. check on adjacent features.
- Reject work back to the relevant role if acceptance criteria aren't met — don't - Reject work back to the relevant role if acceptance criteria aren't met — don't
patch around a failing check. patch around a failing check.
**When invoked**: acceptance-criteria drafting during §B2a; final verification pass **When invoked**: acceptance-criteria drafting during §B2a; final verification pass
before a task is marked `[DONE]`. before a task is marked `[DONE]`.
**Handoff**: reports pass/fail with specifics (what broke, under what input) back to **Handoff**: reports pass/fail with specifics (what broke, under what input) back to
whichever role owns that surface. whichever role owns that surface.
## 5. Hardware & Performance Compatibility Reviewer ## 5. Hardware & Performance Compatibility Reviewer
**Responsibilities** **Responsibilities**
- Check the change against this stack's real constraints: single vs. dual-GPU dev - Check the change against this stack's real constraints: single vs. dual-GPU dev
mode (`docker-compose.yml` runs `npm run dev`, a known throughput ceiling), VRAM mode (`docker-compose.yml` runs `npm run dev`, a known throughput ceiling), VRAM
budget for vLLM (`gpu-memory-utilization` in `config/vllm_config.yaml`), and budget for vLLM (`gpu-memory-utilization` in `config/vllm_config.yaml`), and
behavior under the Local/on-prem deployment mode from `AGENTS.md` Part B §B6. behavior under the Local/on-prem deployment mode from `AGENTS.md` Part B §B6.
- Flag newly introduced heavy Python/Node dependencies, GPU-specific assumptions, or - Flag newly introduced heavy Python/Node dependencies, GPU-specific assumptions, or
anything that would break the isolated vLLM-server-only environment (no anything that would break the isolated vLLM-server-only environment (no
`paddlepaddle-gpu` in this venv — see `AGENTS.md` Part A / `docs/vllm-service.md`). `paddlepaddle-gpu` in this venv — see `AGENTS.md` Part A / `docs/vllm-service.md`).
- Flag anything that would degrade badly on lower-spec hardware or slower networks - Flag anything that would degrade badly on lower-spec hardware or slower networks
(e.g. the mobile app's 2s polling loop against a slow backend response), and (e.g. the mobile app's 2s polling loop against a slow backend response), and
suggest a lighter-weight alternative when one exists. suggest a lighter-weight alternative when one exists.
**When invoked**: final verification pass, alongside QA, before a task is marked **When invoked**: final verification pass, alongside QA, before a task is marked
`[DONE]`; also whenever a task adds a new dependency or changes the deployment/ `[DONE]`; also whenever a task adds a new dependency or changes the deployment/
runtime surface. runtime surface.
**Handoff**: blocks `[DONE]` status until concerns are resolved or explicitly **Handoff**: blocks `[DONE]` status until concerns are resolved or explicitly
accepted as a documented trade-off in `docs/feature-list.md` (this dir). accepted as a documented trade-off in `docs/feature-list.md` (this dir).
+308 -308
View File
@@ -1,308 +1,308 @@
import json import json
import os import os
import glob import glob
import pandas as pd import pandas as pd
from openpyxl.styles import Font, PatternFill, Alignment, Border, Side from openpyxl.styles import Font, PatternFill, Alignment, Border, Side
from openpyxl.utils import get_column_letter from openpyxl.utils import get_column_letter
def clean_val(val): def clean_val(val):
if val is None: if val is None:
return "" return ""
s = str(val).strip().upper() s = str(val).strip().upper()
s = " ".join(s.split()) s = " ".join(s.split())
s = s.replace("PT. ", "PT.") s = s.replace("PT. ", "PT.")
s = s.replace("✓", "").replace("✔", "").strip() s = s.replace("✓", "").replace("✔", "").strip()
return s return s
def main(): def main():
jsonl_file = "backend/uploads/test_images_results.jsonl" jsonl_file = "backend/uploads/test_images_results.jsonl"
manual_labels_pattern = "backend/uploads/manual_label_*.json" manual_labels_pattern = "backend/uploads/manual_label_*.json"
xlsx_file = "backend/pfm-web-app/public/comparison_report.xlsx" xlsx_file = "backend/pfm-web-app/public/comparison_report.xlsx"
if not os.path.exists(jsonl_file): if not os.path.exists(jsonl_file):
# Fallback to backend/uploads if run from different dir # Fallback to backend/uploads if run from different dir
jsonl_file = "uploads/test_images_results.jsonl" jsonl_file = "uploads/test_images_results.jsonl"
manual_labels_pattern = "uploads/manual_label_*.json" manual_labels_pattern = "uploads/manual_label_*.json"
xlsx_file = "pfm-web-app/public/comparison_report.xlsx" xlsx_file = "pfm-web-app/public/comparison_report.xlsx"
if not os.path.exists(jsonl_file): if not os.path.exists(jsonl_file):
print(f"Error: JSONL file not found at {jsonl_file}") print(f"Error: JSONL file not found at {jsonl_file}")
return return
# Load automated results # Load automated results
auto_results = {} auto_results = {}
with open(jsonl_file, "r", encoding="utf-8") as f: with open(jsonl_file, "r", encoding="utf-8") as f:
for line in f: for line in f:
if not line.strip(): if not line.strip():
continue continue
try: try:
data = json.loads(line) data = json.loads(line)
filename = data.get("filename") filename = data.get("filename")
if filename: if filename:
auto_results[filename] = data auto_results[filename] = data
except Exception as e: except Exception as e:
print(f"Skipping line: {e}") print(f"Skipping line: {e}")
# Load manual labels # Load manual labels
manual_files = glob.glob(manual_labels_pattern) manual_files = glob.glob(manual_labels_pattern)
manual_labels = {} manual_labels = {}
for mf in manual_files: for mf in manual_files:
try: try:
with open(mf, "r", encoding="utf-8") as f: with open(mf, "r", encoding="utf-8") as f:
data = json.load(f) data = json.load(f)
filename = data.get("filename") filename = data.get("filename")
if filename: if filename:
manual_labels[filename] = data manual_labels[filename] = data
except Exception as e: except Exception as e:
print(f"Error reading manual label {mf}: {e}") print(f"Error reading manual label {mf}: {e}")
print(f"Loaded {len(auto_results)} automated results.") print(f"Loaded {len(auto_results)} automated results.")
print(f"Loaded {len(manual_labels)} manual labels.") print(f"Loaded {len(manual_labels)} manual labels.")
# Fields to compare in headers # Fields to compare in headers
header_fields = [ header_fields = [
("noPO", "noPO", "PO Number"), ("noPO", "noPO", "PO Number"),
("noSO", "noSO", "SO Number"), ("noSO", "noSO", "SO Number"),
("noDO", "noDO", "DO Number"), ("noDO", "noDO", "DO Number"),
("tanggal", "tanggal", "Date"), ("tanggal", "tanggal", "Date"),
("plat", "platTruk", "Plat Nomor"), ("plat", "platTruk", "Plat Nomor"),
("customer", "customerInfo", "Customer Name"), ("customer", "customerInfo", "Customer Name"),
("store", "orderUntuk", "Store Name"), ("store", "orderUntuk", "Store Name"),
("alamat", "alamat", "Alamat") ("alamat", "alamat", "Alamat")
] ]
doc_comparison_rows = [] doc_comparison_rows = []
item_comparison_rows = [] item_comparison_rows = []
# Counters for accuracy calculation # Counters for accuracy calculation
stats = { stats = {
"PO Number": {"match": 0, "total": 0}, "PO Number": {"match": 0, "total": 0},
"SO Number": {"match": 0, "total": 0}, "SO Number": {"match": 0, "total": 0},
"DO Number": {"match": 0, "total": 0}, "DO Number": {"match": 0, "total": 0},
"Date": {"match": 0, "total": 0}, "Date": {"match": 0, "total": 0},
"Plat Nomor": {"match": 0, "total": 0}, "Plat Nomor": {"match": 0, "total": 0},
"Customer Name": {"match": 0, "total": 0}, "Customer Name": {"match": 0, "total": 0},
"Store Name": {"match": 0, "total": 0}, "Store Name": {"match": 0, "total": 0},
"Alamat": {"match": 0, "total": 0}, "Alamat": {"match": 0, "total": 0},
"Item SKU": {"match": 0, "total": 0}, "Item SKU": {"match": 0, "total": 0},
"Item Banyak": {"match": 0, "total": 0}, "Item Banyak": {"match": 0, "total": 0},
"Item Jumlah": {"match": 0, "total": 0} "Item Jumlah": {"match": 0, "total": 0}
} }
for filename, manual in manual_labels.items(): for filename, manual in manual_labels.items():
auto = auto_results.get(filename) auto = auto_results.get(filename)
if not auto: if not auto:
print(f"Warning: Automated result not found for {filename}") print(f"Warning: Automated result not found for {filename}")
continue continue
auto_meta = auto.get("metadata", {}) auto_meta = auto.get("metadata", {})
# 1. Compare header fields # 1. Compare header fields
for manual_key, auto_key, field_label in header_fields: for manual_key, auto_key, field_label in header_fields:
m_val = clean_val(manual.get(manual_key)) m_val = clean_val(manual.get(manual_key))
a_val = clean_val(auto_meta.get(auto_key)) a_val = clean_val(auto_meta.get(auto_key))
is_match = (m_val == a_val) is_match = (m_val == a_val)
doc_comparison_rows.append({ doc_comparison_rows.append({
"Filename": filename, "Filename": filename,
"Field": field_label, "Field": field_label,
"Automated Value (OCR)": a_val if a_val else "(empty)", "Automated Value (OCR)": a_val if a_val else "(empty)",
"Manual Value (Ground Truth)": m_val if m_val else "(empty)", "Manual Value (Ground Truth)": m_val if m_val else "(empty)",
"Match": "Match" if is_match else "Mismatch" "Match": "Match" if is_match else "Mismatch"
}) })
stats[field_label]["total"] += 1 stats[field_label]["total"] += 1
if is_match: if is_match:
stats[field_label]["match"] += 1 stats[field_label]["match"] += 1
# 2. Compare items # 2. Compare items
m_items = manual.get("items", []) m_items = manual.get("items", [])
# We also look at auto.get("items") or auto_meta.get("items") # We also look at auto.get("items") or auto_meta.get("items")
a_items = auto.get("items", []) a_items = auto.get("items", [])
if not a_items and "items" in auto_meta: if not a_items and "items" in auto_meta:
a_items = auto_meta.get("items", []) a_items = auto_meta.get("items", [])
# Create dictionaries of items indexed by codeBarang (SKU) # Create dictionaries of items indexed by codeBarang (SKU)
m_items_dict = {clean_val(item.get("kodeBarang")): item for item in m_items if clean_val(item.get("kodeBarang"))} m_items_dict = {clean_val(item.get("kodeBarang")): item for item in m_items if clean_val(item.get("kodeBarang"))}
a_items_dict = {clean_val(item.get("kodeBarang")): item for item in a_items if clean_val(item.get("kodeBarang"))} a_items_dict = {clean_val(item.get("kodeBarang")): item for item in a_items if clean_val(item.get("kodeBarang"))}
# Check all unique SKUs across both manual and automated # Check all unique SKUs across both manual and automated
all_skus = set(list(m_items_dict.keys()) + list(a_items_dict.keys())) all_skus = set(list(m_items_dict.keys()) + list(a_items_dict.keys()))
for sku in all_skus: for sku in all_skus:
m_item = m_items_dict.get(sku) m_item = m_items_dict.get(sku)
a_item = a_items_dict.get(sku) a_item = a_items_dict.get(sku)
# Check SKU existence match # Check SKU existence match
sku_match = (m_item is not None) and (a_item is not None) sku_match = (m_item is not None) and (a_item is not None)
stats["Item SKU"]["total"] += 1 stats["Item SKU"]["total"] += 1
if sku_match: if sku_match:
stats["Item SKU"]["match"] += 1 stats["Item SKU"]["match"] += 1
m_banyak = clean_val(m_item.get("banyak")) if m_item else "" m_banyak = clean_val(m_item.get("banyak")) if m_item else ""
a_banyak = clean_val(a_item.get("banyak")) if a_item else "" a_banyak = clean_val(a_item.get("banyak")) if a_item else ""
banyak_match = (m_banyak == a_banyak) banyak_match = (m_banyak == a_banyak)
stats["Item Banyak"]["total"] += 1 stats["Item Banyak"]["total"] += 1
if banyak_match: if banyak_match:
stats["Item Banyak"]["match"] += 1 stats["Item Banyak"]["match"] += 1
m_jumlah = clean_val(m_item.get("jumlah")) if m_item else "" m_jumlah = clean_val(m_item.get("jumlah")) if m_item else ""
a_jumlah = clean_val(a_item.get("jumlah")) if a_item else "" a_jumlah = clean_val(a_item.get("jumlah")) if a_item else ""
jumlah_match = (m_jumlah == a_jumlah) jumlah_match = (m_jumlah == a_jumlah)
stats["Item Jumlah"]["total"] += 1 stats["Item Jumlah"]["total"] += 1
if jumlah_match: if jumlah_match:
stats["Item Jumlah"]["match"] += 1 stats["Item Jumlah"]["match"] += 1
# Log code comparison # Log code comparison
item_comparison_rows.append({ item_comparison_rows.append({
"Filename": filename, "Filename": filename,
"Kode Barang (SKU)": sku, "Kode Barang (SKU)": sku,
"Field": "SKU Existence", "Field": "SKU Existence",
"Automated Value (OCR)": sku if a_item else "(not found)", "Automated Value (OCR)": sku if a_item else "(not found)",
"Manual Value (Ground Truth)": sku if m_item else "(not found)", "Manual Value (Ground Truth)": sku if m_item else "(not found)",
"Match": "Match" if sku_match else "Mismatch" "Match": "Match" if sku_match else "Mismatch"
}) })
# Log Banyak comparison # Log Banyak comparison
item_comparison_rows.append({ item_comparison_rows.append({
"Filename": filename, "Filename": filename,
"Kode Barang (SKU)": sku, "Kode Barang (SKU)": sku,
"Field": "Banyak (Qty Package)", "Field": "Banyak (Qty Package)",
"Automated Value (OCR)": a_banyak if a_banyak else "(empty)", "Automated Value (OCR)": a_banyak if a_banyak else "(empty)",
"Manual Value (Ground Truth)": m_banyak if m_banyak else "(empty)", "Manual Value (Ground Truth)": m_banyak if m_banyak else "(empty)",
"Match": "Match" if banyak_match else "Mismatch" "Match": "Match" if banyak_match else "Mismatch"
}) })
# Log Jumlah comparison # Log Jumlah comparison
item_comparison_rows.append({ item_comparison_rows.append({
"Filename": filename, "Filename": filename,
"Kode Barang (SKU)": sku, "Kode Barang (SKU)": sku,
"Field": "Jumlah (Qty Unit)", "Field": "Jumlah (Qty Unit)",
"Automated Value (OCR)": a_jumlah if a_jumlah else "(empty)", "Automated Value (OCR)": a_jumlah if a_jumlah else "(empty)",
"Manual Value (Ground Truth)": m_jumlah if m_jumlah else "(empty)", "Manual Value (Ground Truth)": m_jumlah if m_jumlah else "(empty)",
"Match": "Match" if jumlah_match else "Mismatch" "Match": "Match" if jumlah_match else "Mismatch"
}) })
# Prepare summary data # Prepare summary data
summary_rows = [] summary_rows = []
total_matches = 0 total_matches = 0
total_fields = 0 total_fields = 0
for field_label, counts in stats.items(): for field_label, counts in stats.items():
match_cnt = counts["match"] match_cnt = counts["match"]
total_cnt = counts["total"] total_cnt = counts["total"]
pct = (match_cnt / total_cnt * 100.0) if total_cnt > 0 else 100.0 pct = (match_cnt / total_cnt * 100.0) if total_cnt > 0 else 100.0
summary_rows.append({ summary_rows.append({
"Field / Area": field_label, "Field / Area": field_label,
"Total Checks": total_cnt, "Total Checks": total_cnt,
"Matches": match_cnt, "Matches": match_cnt,
"Mismatches": total_cnt - match_cnt, "Mismatches": total_cnt - match_cnt,
"Accuracy (%)": round(pct, 2) "Accuracy (%)": round(pct, 2)
}) })
total_matches += match_cnt total_matches += match_cnt
total_fields += total_cnt total_fields += total_cnt
overall_accuracy = (total_matches / total_fields * 100.0) if total_fields > 0 else 100.0 overall_accuracy = (total_matches / total_fields * 100.0) if total_fields > 0 else 100.0
summary_rows.append({ summary_rows.append({
"Field / Area": "OVERALL TOTAL", "Field / Area": "OVERALL TOTAL",
"Total Checks": total_fields, "Total Checks": total_fields,
"Matches": total_matches, "Matches": total_matches,
"Mismatches": total_fields - total_matches, "Mismatches": total_fields - total_matches,
"Accuracy (%)": round(overall_accuracy, 2) "Accuracy (%)": round(overall_accuracy, 2)
}) })
df_summary = pd.DataFrame(summary_rows) df_summary = pd.DataFrame(summary_rows)
df_docs = pd.DataFrame(doc_comparison_rows) df_docs = pd.DataFrame(doc_comparison_rows)
df_items = pd.DataFrame(item_comparison_rows) df_items = pd.DataFrame(item_comparison_rows)
# Styling setup # Styling setup
font_family = "Segoe UI" font_family = "Segoe UI"
header_font = Font(name=font_family, size=11, bold=True, color="FFFFFF") header_font = Font(name=font_family, size=11, bold=True, color="FFFFFF")
regular_font = Font(name=font_family, size=10) regular_font = Font(name=font_family, size=10)
bold_font = Font(name=font_family, size=10, bold=True) bold_font = Font(name=font_family, size=10, bold=True)
header_fill = PatternFill(start_color="1F4E78", end_color="1F4E78", fill_type="solid") # Dark Blue header_fill = PatternFill(start_color="1F4E78", end_color="1F4E78", fill_type="solid") # Dark Blue
zebra_fill = PatternFill(start_color="F2F5F8", end_color="F2F5F8", fill_type="solid") # Zebra light blue-gray zebra_fill = PatternFill(start_color="F2F5F8", end_color="F2F5F8", fill_type="solid") # Zebra light blue-gray
match_fill = PatternFill(start_color="E2EFDA", end_color="E2EFDA", fill_type="solid") # Light green match_fill = PatternFill(start_color="E2EFDA", end_color="E2EFDA", fill_type="solid") # Light green
mismatch_fill = PatternFill(start_color="FCE4D6", end_color="FCE4D6", fill_type="solid") # Light orange mismatch_fill = PatternFill(start_color="FCE4D6", end_color="FCE4D6", fill_type="solid") # Light orange
center_align = Alignment(horizontal="center", vertical="center") center_align = Alignment(horizontal="center", vertical="center")
left_align = Alignment(horizontal="left", vertical="center") left_align = Alignment(horizontal="left", vertical="center")
right_align = Alignment(horizontal="right", vertical="center") right_align = Alignment(horizontal="right", vertical="center")
thin_side = Side(border_style="thin", color="D9D9D9") thin_side = Side(border_style="thin", color="D9D9D9")
cell_border = Border(left=thin_side, right=thin_side, top=thin_side, bottom=thin_side) cell_border = Border(left=thin_side, right=thin_side, top=thin_side, bottom=thin_side)
# Save to Excel # Save to Excel
os.makedirs(os.path.dirname(xlsx_file), exist_ok=True) os.makedirs(os.path.dirname(xlsx_file), exist_ok=True)
with pd.ExcelWriter(xlsx_file, engine='openpyxl') as writer: with pd.ExcelWriter(xlsx_file, engine='openpyxl') as writer:
df_summary.to_excel(writer, sheet_name='Summary Accuracy', index=False) df_summary.to_excel(writer, sheet_name='Summary Accuracy', index=False)
df_docs.to_excel(writer, sheet_name='Header Field Comparison', index=False) df_docs.to_excel(writer, sheet_name='Header Field Comparison', index=False)
df_items.to_excel(writer, sheet_name='Item SKU Comparison', index=False) df_items.to_excel(writer, sheet_name='Item SKU Comparison', index=False)
# Style worksheets # Style worksheets
for sheet_name in ['Summary Accuracy', 'Header Field Comparison', 'Item SKU Comparison']: for sheet_name in ['Summary Accuracy', 'Header Field Comparison', 'Item SKU Comparison']:
ws = writer.sheets[sheet_name] ws = writer.sheets[sheet_name]
max_row = ws.max_row max_row = ws.max_row
max_col = ws.max_column max_col = ws.max_column
# Header row styling # Header row styling
for col in range(1, max_col + 1): for col in range(1, max_col + 1):
cell = ws.cell(row=1, column=col) cell = ws.cell(row=1, column=col)
cell.font = header_font cell.font = header_font
cell.fill = header_fill cell.fill = header_fill
cell.alignment = center_align cell.alignment = center_align
# Data rows styling # Data rows styling
for row in range(2, max_row + 1): for row in range(2, max_row + 1):
is_zebra = (row % 2 == 0) is_zebra = (row % 2 == 0)
# Check for Match/Mismatch to apply colors on sheets 2 & 3 # Check for Match/Mismatch to apply colors on sheets 2 & 3
match_val = None match_val = None
if sheet_name in ['Header Field Comparison', 'Item SKU Comparison']: if sheet_name in ['Header Field Comparison', 'Item SKU Comparison']:
# Match column is the last column # Match column is the last column
match_cell = ws.cell(row=row, column=max_col) match_cell = ws.cell(row=row, column=max_col)
match_val = match_cell.value match_val = match_cell.value
for col in range(1, max_col + 1): for col in range(1, max_col + 1):
cell = ws.cell(row=row, column=col) cell = ws.cell(row=row, column=col)
cell.font = regular_font cell.font = regular_font
cell.border = cell_border cell.border = cell_border
# Apply alignments based on column # Apply alignments based on column
if sheet_name == 'Summary Accuracy': if sheet_name == 'Summary Accuracy':
if col == 1: if col == 1:
cell.alignment = left_align cell.alignment = left_align
else: else:
cell.alignment = right_align cell.alignment = right_align
# Highlight overall total row # Highlight overall total row
if row == max_row: if row == max_row:
cell.font = bold_font cell.font = bold_font
cell.fill = match_fill if overall_accuracy > 80 else mismatch_fill cell.fill = match_fill if overall_accuracy > 80 else mismatch_fill
else: else:
# For detail sheets # For detail sheets
if col in [1, 3, 4]: if col in [1, 3, 4]:
cell.alignment = left_align cell.alignment = left_align
else: else:
cell.alignment = center_align cell.alignment = center_align
# Color match / mismatch # Color match / mismatch
if match_val == "Match": if match_val == "Match":
cell.fill = match_fill cell.fill = match_fill
elif match_val == "Mismatch": elif match_val == "Mismatch":
cell.fill = mismatch_fill cell.fill = mismatch_fill
elif is_zebra: elif is_zebra:
cell.fill = zebra_fill cell.fill = zebra_fill
# Auto-fit columns # Auto-fit columns
for col in ws.columns: for col in ws.columns:
max_len = max(len(str(cell.value or '')) for cell in col) max_len = max(len(str(cell.value or '')) for cell in col)
col_letter = get_column_letter(col[0].column) col_letter = get_column_letter(col[0].column)
ws.column_dimensions[col_letter].width = max(max_len + 4, 12) ws.column_dimensions[col_letter].width = max(max_len + 4, 12)
print(f"Comparison report generated at {xlsx_file}") print(f"Comparison report generated at {xlsx_file}")
if __name__ == "__main__": if __name__ == "__main__":
main() main()
+451 -451
View File
@@ -1,451 +1,451 @@
import json import json
import os import os
import pandas as pd import pandas as pd
from openpyxl.styles import Font, PatternFill, Alignment, Border, Side from openpyxl.styles import Font, PatternFill, Alignment, Border, Side
from openpyxl.utils import get_column_letter from openpyxl.utils import get_column_letter
def clean_val(val): def clean_val(val):
if val is None: if val is None:
return "" return ""
s = str(val).strip().upper() s = str(val).strip().upper()
if s in ["N/A", "NOT FOUND", "NOTFOUND", "EMPTY", "NONE", "-", "N / A"]: if s in ["N/A", "NOT FOUND", "NOTFOUND", "EMPTY", "NONE", "-", "N / A"]:
return "" return ""
s = " ".join(s.split()) s = " ".join(s.split())
s = s.replace("PT. ", "PT.") s = s.replace("PT. ", "PT.")
s = s.replace("✓", "").replace("✔", "").strip() s = s.replace("✓", "").replace("✔", "").strip()
return s return s
def main(): def main():
ai_file = "sources/ai_results.json" ai_file = "sources/ai_results.json"
manual_file = "sources/manual_labels.json" manual_file = "sources/manual_labels.json"
xlsx_file = "sources/comparison_report.xlsx" xlsx_file = "sources/comparison_report.xlsx"
images_dir = "sources/test-images" images_dir = "sources/test-images"
if not os.path.exists(ai_file): if not os.path.exists(ai_file):
# Fallback to backend/sources # Fallback to backend/sources
ai_file = "backend/sources/ai_results.json" ai_file = "backend/sources/ai_results.json"
manual_file = "backend/sources/manual_labels.json" manual_file = "backend/sources/manual_labels.json"
xlsx_file = "backend/sources/comparison_report.xlsx" xlsx_file = "backend/sources/comparison_report.xlsx"
images_dir = "backend/sources/test-images" images_dir = "backend/sources/test-images"
if not os.path.exists(ai_file): if not os.path.exists(ai_file):
print(f"Error: AI results file not found at {ai_file}") print(f"Error: AI results file not found at {ai_file}")
return return
if not os.path.exists(manual_file): if not os.path.exists(manual_file):
print(f"Error: Manual labels file not found at {manual_file}") print(f"Error: Manual labels file not found at {manual_file}")
return return
# Load data # Load data
with open(ai_file, "r", encoding="utf-8") as f: with open(ai_file, "r", encoding="utf-8") as f:
ai_data = json.load(f) ai_data = json.load(f)
with open(manual_file, "r", encoding="utf-8") as f: with open(manual_file, "r", encoding="utf-8") as f:
manual_data = json.load(f) manual_data = json.load(f)
# Convert to dict for lookup by filename # Convert to dict for lookup by filename
ai_dict = {item.get("filename"): item for item in ai_data if item.get("filename")} ai_dict = {item.get("filename"): item for item in ai_data if item.get("filename")}
manual_dict = {item.get("filename"): item for item in manual_data if item.get("filename")} manual_dict = {item.get("filename"): item for item in manual_data if item.get("filename")}
print(f"Loaded {len(ai_dict)} AI results from file.") print(f"Loaded {len(ai_dict)} AI results from file.")
print(f"Loaded {len(manual_dict)} manual labels from file.") print(f"Loaded {len(manual_dict)} manual labels from file.")
# Scan for physical image files in test-images folder # Scan for physical image files in test-images folder
existing_images = None existing_images = None
if os.path.exists(images_dir): if os.path.exists(images_dir):
existing_images = set(os.listdir(images_dir)) existing_images = set(os.listdir(images_dir))
print(f"Found {len(existing_images)} physical images in '{images_dir}'.") print(f"Found {len(existing_images)} physical images in '{images_dir}'.")
else: else:
print(f"Warning: Images directory not found at '{images_dir}'.") print(f"Warning: Images directory not found at '{images_dir}'.")
# Find mismatches in file lists # Find mismatches in file lists
only_in_ai = set(ai_dict.keys()) - set(manual_dict.keys()) only_in_ai = set(ai_dict.keys()) - set(manual_dict.keys())
only_in_manual = set(manual_dict.keys()) - set(ai_dict.keys()) only_in_manual = set(manual_dict.keys()) - set(ai_dict.keys())
if only_in_ai: if only_in_ai:
print(f"Warning: {len(only_in_ai)} files exist only in AI results: {only_in_ai}") print(f"Warning: {len(only_in_ai)} files exist only in AI results: {only_in_ai}")
if only_in_manual: if only_in_manual:
print(f"Warning: {len(only_in_manual)} files exist only in Manual labels: {only_in_manual}") print(f"Warning: {len(only_in_manual)} files exist only in Manual labels: {only_in_manual}")
# Determine files to compare (must exist in AI results, Manual labels, and physically as images if directory is available) # Determine files to compare (must exist in AI results, Manual labels, and physically as images if directory is available)
common_filenames = set(ai_dict.keys()) & set(manual_dict.keys()) common_filenames = set(ai_dict.keys()) & set(manual_dict.keys())
if existing_images is not None: if existing_images is not None:
deleted_images = common_filenames - existing_images deleted_images = common_filenames - existing_images
if deleted_images: if deleted_images:
print(f"Info: Excluded {len(deleted_images)} files that were physically deleted from images folder: {deleted_images}") print(f"Info: Excluded {len(deleted_images)} files that were physically deleted from images folder: {deleted_images}")
all_filenames = sorted(list(common_filenames & existing_images)) all_filenames = sorted(list(common_filenames & existing_images))
else: else:
all_filenames = sorted(list(common_filenames)) all_filenames = sorted(list(common_filenames))
print(f"Comparing {len(all_filenames)} matching images.") print(f"Comparing {len(all_filenames)} matching images.")
header_fields = [ header_fields = [
("noPO", "noPO", "PO Number"), ("noPO", "noPO", "PO Number"),
("noSO", "noSO", "SO Number"), ("noSO", "noSO", "SO Number"),
("noDO", "noDO", "DO Number"), ("noDO", "noDO", "DO Number"),
("tanggal", "tanggal", "Date"), ("tanggal", "tanggal", "Date"),
("plat", "platTruk", "Plat Nomor"), ("plat", "platTruk", "Plat Nomor"),
("customer", "customerInfo", "Customer Name"), ("customer", "customerInfo", "Customer Name"),
("store", "orderUntuk", "Store Name"), ("store", "orderUntuk", "Store Name"),
("alamat", "alamat", "Alamat") ("alamat", "alamat", "Alamat")
] ]
doc_comparison_rows = [] doc_comparison_rows = []
item_comparison_rows = [] item_comparison_rows = []
# Counters for accuracy calculation # Counters for accuracy calculation
stats = { stats = {
"PO Number": {"match": 0, "total": 0}, "PO Number": {"match": 0, "total": 0},
"SO Number": {"match": 0, "total": 0}, "SO Number": {"match": 0, "total": 0},
"DO Number": {"match": 0, "total": 0}, "DO Number": {"match": 0, "total": 0},
"Date": {"match": 0, "total": 0}, "Date": {"match": 0, "total": 0},
"Plat Nomor": {"match": 0, "total": 0}, "Plat Nomor": {"match": 0, "total": 0},
"Customer Name": {"match": 0, "total": 0}, "Customer Name": {"match": 0, "total": 0},
"Store Name": {"match": 0, "total": 0}, "Store Name": {"match": 0, "total": 0},
"Alamat": {"match": 0, "total": 0}, "Alamat": {"match": 0, "total": 0},
"Item SKU": {"match": 0, "total": 0}, "Item SKU": {"match": 0, "total": 0},
"Item Banyak": {"match": 0, "total": 0}, "Item Banyak": {"match": 0, "total": 0},
"Item Jumlah": {"match": 0, "total": 0} "Item Jumlah": {"match": 0, "total": 0}
} }
# Document-level side-by-side rows # Document-level side-by-side rows
doc_side_by_side_rows = [] doc_side_by_side_rows = []
for filename in all_filenames: for filename in all_filenames:
manual = manual_dict.get(filename) manual = manual_dict.get(filename)
ai = ai_dict.get(filename) ai = ai_dict.get(filename)
if not manual: if not manual:
print(f"Warning: Manual label not found for {filename} (exists only in AI results)") print(f"Warning: Manual label not found for {filename} (exists only in AI results)")
continue continue
if not ai: if not ai:
print(f"Warning: AI result not found for {filename} (exists only in Manual labels)") print(f"Warning: AI result not found for {filename} (exists only in Manual labels)")
continue continue
ai_meta = ai.get("layer3Final", {}) ai_meta = ai.get("layer3Final", {})
# 1. Compare header fields (Vertical format for filtering) # 1. Compare header fields (Vertical format for filtering)
sxs_row = {"Filename": filename} sxs_row = {"Filename": filename}
for manual_key, ai_key, field_label in header_fields: for manual_key, ai_key, field_label in header_fields:
m_val = clean_val(manual.get(manual_key)) m_val = clean_val(manual.get(manual_key))
a_val = clean_val(ai_meta.get(ai_key)) a_val = clean_val(ai_meta.get(ai_key))
is_match = (m_val == a_val) is_match = (m_val == a_val)
doc_comparison_rows.append({ doc_comparison_rows.append({
"Filename": filename, "Filename": filename,
"Field": field_label, "Field": field_label,
"AI Value (OCR)": a_val if a_val else "(empty)", "AI Value (OCR)": a_val if a_val else "(empty)",
"Manual Value (Ground Truth)": m_val if m_val else "(empty)", "Manual Value (Ground Truth)": m_val if m_val else "(empty)",
"Match": "Match" if is_match else "Mismatch" "Match": "Match" if is_match else "Mismatch"
}) })
# Side-by-side # Side-by-side
sxs_row[f"{field_label} (AI)"] = a_val if a_val else "" sxs_row[f"{field_label} (AI)"] = a_val if a_val else ""
sxs_row[f"{field_label} (Manual)"] = m_val if m_val else "" sxs_row[f"{field_label} (Manual)"] = m_val if m_val else ""
sxs_row[f"{field_label} Status"] = "Match" if is_match else "Mismatch" sxs_row[f"{field_label} Status"] = "Match" if is_match else "Mismatch"
stats[field_label]["total"] += 1 stats[field_label]["total"] += 1
if is_match: if is_match:
stats[field_label]["match"] += 1 stats[field_label]["match"] += 1
doc_side_by_side_rows.append(sxs_row) doc_side_by_side_rows.append(sxs_row)
# 2. Compare items # 2. Compare items
m_items = manual.get("items", []) m_items = manual.get("items", [])
ai_items = ai.get("items", []) ai_items = ai.get("items", [])
if not ai_items and "items" in ai_meta: if not ai_items and "items" in ai_meta:
ai_items = ai_meta.get("items", []) ai_items = ai_meta.get("items", [])
# Create dictionaries of items indexed by codeBarang (SKU) # Create dictionaries of items indexed by codeBarang (SKU)
m_items_dict = {clean_val(item.get("kodeBarang")): item for item in m_items if clean_val(item.get("kodeBarang"))} m_items_dict = {clean_val(item.get("kodeBarang")): item for item in m_items if clean_val(item.get("kodeBarang"))}
ai_items_dict = {clean_val(item.get("kodeBarang")): item for item in ai_items if clean_val(item.get("kodeBarang"))} ai_items_dict = {clean_val(item.get("kodeBarang")): item for item in ai_items if clean_val(item.get("kodeBarang"))}
# Check all unique SKUs across both manual and AI # Check all unique SKUs across both manual and AI
all_skus = set(list(m_items_dict.keys()) + list(ai_items_dict.keys())) all_skus = set(list(m_items_dict.keys()) + list(ai_items_dict.keys()))
for sku in all_skus: for sku in all_skus:
m_item = m_items_dict.get(sku) m_item = m_items_dict.get(sku)
ai_item = ai_items_dict.get(sku) ai_item = ai_items_dict.get(sku)
# Check SKU existence match # Check SKU existence match
sku_match = (m_item is not None) and (ai_item is not None) sku_match = (m_item is not None) and (ai_item is not None)
stats["Item SKU"]["total"] += 1 stats["Item SKU"]["total"] += 1
if sku_match: if sku_match:
stats["Item SKU"]["match"] += 1 stats["Item SKU"]["match"] += 1
m_banyak = clean_val(m_item.get("banyak")) if m_item else "" m_banyak = clean_val(m_item.get("banyak")) if m_item else ""
ai_banyak = clean_val(ai_item.get("banyak")) if ai_item else "" ai_banyak = clean_val(ai_item.get("banyak")) if ai_item else ""
banyak_match = (m_banyak == ai_banyak) banyak_match = (m_banyak == ai_banyak)
stats["Item Banyak"]["total"] += 1 stats["Item Banyak"]["total"] += 1
if banyak_match: if banyak_match:
stats["Item Banyak"]["match"] += 1 stats["Item Banyak"]["match"] += 1
m_jumlah = clean_val(m_item.get("jumlah")) if m_item else "" m_jumlah = clean_val(m_item.get("jumlah")) if m_item else ""
ai_jumlah = clean_val(ai_item.get("jumlah")) if ai_item else "" ai_jumlah = clean_val(ai_item.get("jumlah")) if ai_item else ""
jumlah_match = (m_jumlah == ai_jumlah) jumlah_match = (m_jumlah == ai_jumlah)
stats["Item Jumlah"]["total"] += 1 stats["Item Jumlah"]["total"] += 1
if jumlah_match: if jumlah_match:
stats["Item Jumlah"]["match"] += 1 stats["Item Jumlah"]["match"] += 1
# Log code comparison # Log code comparison
item_comparison_rows.append({ item_comparison_rows.append({
"Filename": filename, "Filename": filename,
"Kode Barang (SKU)": sku, "Kode Barang (SKU)": sku,
"Field": "SKU Existence", "Field": "SKU Existence",
"AI Value (OCR)": sku if ai_item else "(not found)", "AI Value (OCR)": sku if ai_item else "(not found)",
"Manual Value (Ground Truth)": sku if m_item else "(not found)", "Manual Value (Ground Truth)": sku if m_item else "(not found)",
"Match": "Match" if sku_match else "Mismatch" "Match": "Match" if sku_match else "Mismatch"
}) })
# Log Banyak comparison # Log Banyak comparison
item_comparison_rows.append({ item_comparison_rows.append({
"Filename": filename, "Filename": filename,
"Kode Barang (SKU)": sku, "Kode Barang (SKU)": sku,
"Field": "Banyak (Qty Package)", "Field": "Banyak (Qty Package)",
"AI Value (OCR)": ai_banyak if ai_banyak else "(empty)", "AI Value (OCR)": ai_banyak if ai_banyak else "(empty)",
"Manual Value (Ground Truth)": m_banyak if m_banyak else "(empty)", "Manual Value (Ground Truth)": m_banyak if m_banyak else "(empty)",
"Match": "Match" if banyak_match else "Mismatch" "Match": "Match" if banyak_match else "Mismatch"
}) })
# Log Jumlah comparison # Log Jumlah comparison
item_comparison_rows.append({ item_comparison_rows.append({
"Filename": filename, "Filename": filename,
"Kode Barang (SKU)": sku, "Kode Barang (SKU)": sku,
"Field": "Jumlah (Qty Unit)", "Field": "Jumlah (Qty Unit)",
"AI Value (OCR)": ai_jumlah if ai_jumlah else "(empty)", "AI Value (OCR)": ai_jumlah if ai_jumlah else "(empty)",
"Manual Value (Ground Truth)": m_jumlah if m_jumlah else "(empty)", "Manual Value (Ground Truth)": m_jumlah if m_jumlah else "(empty)",
"Match": "Match" if jumlah_match else "Mismatch" "Match": "Match" if jumlah_match else "Mismatch"
}) })
# Prepare summary data # Prepare summary data
summary_rows = [] summary_rows = []
total_matches = 0 total_matches = 0
total_fields = 0 total_fields = 0
for field_label, counts in stats.items(): for field_label, counts in stats.items():
match_cnt = counts["match"] match_cnt = counts["match"]
total_cnt = counts["total"] total_cnt = counts["total"]
pct = (match_cnt / total_cnt * 100.0) if total_cnt > 0 else 100.0 pct = (match_cnt / total_cnt * 100.0) if total_cnt > 0 else 100.0
summary_rows.append({ summary_rows.append({
"Field / Area": field_label, "Field / Area": field_label,
"Total Checks": total_cnt, "Total Checks": total_cnt,
"Matches": match_cnt, "Matches": match_cnt,
"Mismatches": total_cnt - match_cnt, "Mismatches": total_cnt - match_cnt,
"Accuracy (%)": round(pct, 2) "Accuracy (%)": round(pct, 2)
}) })
total_matches += match_cnt total_matches += match_cnt
total_fields += total_cnt total_fields += total_cnt
overall_accuracy = (total_matches / total_fields * 100.0) if total_fields > 0 else 100.0 overall_accuracy = (total_matches / total_fields * 100.0) if total_fields > 0 else 100.0
summary_rows.append({ summary_rows.append({
"Field / Area": "OVERALL TOTAL", "Field / Area": "OVERALL TOTAL",
"Total Checks": total_fields, "Total Checks": total_fields,
"Matches": total_matches, "Matches": total_matches,
"Mismatches": total_fields - total_matches, "Mismatches": total_fields - total_matches,
"Accuracy (%)": round(overall_accuracy, 2) "Accuracy (%)": round(overall_accuracy, 2)
}) })
df_summary = pd.DataFrame(summary_rows) df_summary = pd.DataFrame(summary_rows)
df_docs = pd.DataFrame(doc_comparison_rows) df_docs = pd.DataFrame(doc_comparison_rows)
df_sxs = pd.DataFrame(doc_side_by_side_rows) df_sxs = pd.DataFrame(doc_side_by_side_rows)
df_items = pd.DataFrame(item_comparison_rows) df_items = pd.DataFrame(item_comparison_rows)
# Styling setup # Styling setup
font_family = "Segoe UI" font_family = "Segoe UI"
header_font = Font(name=font_family, size=11, bold=True, color="FFFFFF") header_font = Font(name=font_family, size=11, bold=True, color="FFFFFF")
regular_font = Font(name=font_family, size=10) regular_font = Font(name=font_family, size=10)
bold_font = Font(name=font_family, size=10, bold=True) bold_font = Font(name=font_family, size=10, bold=True)
title_font = Font(name=font_family, size=16, bold=True, color="1F4E78") title_font = Font(name=font_family, size=16, bold=True, color="1F4E78")
header_fill = PatternFill(start_color="1F4E78", end_color="1F4E78", fill_type="solid") # Dark Blue header_fill = PatternFill(start_color="1F4E78", end_color="1F4E78", fill_type="solid") # Dark Blue
zebra_fill = PatternFill(start_color="F2F5F8", end_color="F2F5F8", fill_type="solid") # Zebra light blue-gray zebra_fill = PatternFill(start_color="F2F5F8", end_color="F2F5F8", fill_type="solid") # Zebra light blue-gray
match_fill = PatternFill(start_color="E2EFDA", end_color="E2EFDA", fill_type="solid") # Light green match_fill = PatternFill(start_color="E2EFDA", end_color="E2EFDA", fill_type="solid") # Light green
mismatch_fill = PatternFill(start_color="FCE4D6", end_color="FCE4D6", fill_type="solid") # Light orange mismatch_fill = PatternFill(start_color="FCE4D6", end_color="FCE4D6", fill_type="solid") # Light orange
center_align = Alignment(horizontal="center", vertical="center") center_align = Alignment(horizontal="center", vertical="center")
left_align = Alignment(horizontal="left", vertical="center") left_align = Alignment(horizontal="left", vertical="center")
right_align = Alignment(horizontal="right", vertical="center") right_align = Alignment(horizontal="right", vertical="center")
thin_side = Side(border_style="thin", color="D9D9D9") thin_side = Side(border_style="thin", color="D9D9D9")
cell_border = Border(left=thin_side, right=thin_side, top=thin_side, bottom=thin_side) cell_border = Border(left=thin_side, right=thin_side, top=thin_side, bottom=thin_side)
# Save to Excel # Save to Excel
os.makedirs(os.path.dirname(xlsx_file), exist_ok=True) os.makedirs(os.path.dirname(xlsx_file), exist_ok=True)
writer = None writer = None
for attempt in range(1, 10): for attempt in range(1, 10):
try: try:
writer = pd.ExcelWriter(xlsx_file, engine='openpyxl') writer = pd.ExcelWriter(xlsx_file, engine='openpyxl')
break break
except PermissionError: except PermissionError:
base_dir = os.path.dirname(xlsx_file) base_dir = os.path.dirname(xlsx_file)
filename = os.path.basename(xlsx_file) filename = os.path.basename(xlsx_file)
name, ext = os.path.splitext(filename) name, ext = os.path.splitext(filename)
if "_" in name and name.split("_")[-1].isdigit(): if "_" in name and name.split("_")[-1].isdigit():
name = "_".join(name.split("_")[:-1]) name = "_".join(name.split("_")[:-1])
xlsx_file = os.path.join(base_dir, f"{name}_{attempt}{ext}") xlsx_file = os.path.join(base_dir, f"{name}_{attempt}{ext}")
if writer is None: if writer is None:
print("Error: Could not open the Excel writer because the file is locked.") print("Error: Could not open the Excel writer because the file is locked.")
return return
with writer: with writer:
df_summary.to_excel(writer, sheet_name='Summary Accuracy', index=False, startrow=3) df_summary.to_excel(writer, sheet_name='Summary Accuracy', index=False, startrow=3)
df_sxs.to_excel(writer, sheet_name='Side-by-Side Comparison', index=False) df_sxs.to_excel(writer, sheet_name='Side-by-Side Comparison', index=False)
df_docs.to_excel(writer, sheet_name='Header Field Comparison', index=False) df_docs.to_excel(writer, sheet_name='Header Field Comparison', index=False)
df_items.to_excel(writer, sheet_name='Item SKU Comparison', index=False) df_items.to_excel(writer, sheet_name='Item SKU Comparison', index=False)
# 1. Style Summary Sheet with a Title Banner # 1. Style Summary Sheet with a Title Banner
ws_sum = writer.sheets['Summary Accuracy'] ws_sum = writer.sheets['Summary Accuracy']
ws_sum.views.sheetView[0].showGridLines = True ws_sum.views.sheetView[0].showGridLines = True
ws_sum.cell(row=1, column=1, value="AI OCR vs. Manual Ground Truth Accuracy Report").font = title_font ws_sum.cell(row=1, column=1, value="AI OCR vs. Manual Ground Truth Accuracy Report").font = title_font
ws_sum.row_dimensions[1].height = 30 ws_sum.row_dimensions[1].height = 30
# Style Summary Table Headers # Style Summary Table Headers
max_col_sum = df_summary.shape[1] max_col_sum = df_summary.shape[1]
for col in range(1, max_col_sum + 1): for col in range(1, max_col_sum + 1):
cell = ws_sum.cell(row=4, column=col) cell = ws_sum.cell(row=4, column=col)
cell.font = header_font cell.font = header_font
cell.fill = header_fill cell.fill = header_fill
cell.alignment = center_align cell.alignment = center_align
cell.border = cell_border cell.border = cell_border
# Style Summary Data # Style Summary Data
max_row_sum = ws_sum.max_row max_row_sum = ws_sum.max_row
for row in range(5, max_row_sum + 1): for row in range(5, max_row_sum + 1):
for col in range(1, max_col_sum + 1): for col in range(1, max_col_sum + 1):
cell = ws_sum.cell(row=row, column=col) cell = ws_sum.cell(row=row, column=col)
cell.font = regular_font cell.font = regular_font
cell.border = cell_border cell.border = cell_border
if col == 1: if col == 1:
cell.alignment = left_align cell.alignment = left_align
else: else:
cell.alignment = right_align cell.alignment = right_align
# Zebra style # Zebra style
if row % 2 == 0 and row != max_row_sum: if row % 2 == 0 and row != max_row_sum:
cell.fill = zebra_fill cell.fill = zebra_fill
# Format percentage # Format percentage
if col == 5 and isinstance(cell.value, (int, float)): if col == 5 and isinstance(cell.value, (int, float)):
cell.number_format = '0.00"%"' cell.number_format = '0.00"%"'
# Bold overall total row # Bold overall total row
if row == max_row_sum: if row == max_row_sum:
for col in range(1, max_col_sum + 1): for col in range(1, max_col_sum + 1):
c = ws_sum.cell(row=row, column=col) c = ws_sum.cell(row=row, column=col)
c.font = bold_font c.font = bold_font
c.fill = match_fill if overall_accuracy > 80 else mismatch_fill c.fill = match_fill if overall_accuracy > 80 else mismatch_fill
# Auto-adjust column width for Summary # Auto-adjust column width for Summary
for col in ws_sum.columns: for col in ws_sum.columns:
max_len = max(len(str(cell.value or '')) for cell in col) max_len = max(len(str(cell.value or '')) for cell in col)
col_letter = get_column_letter(col[0].column) col_letter = get_column_letter(col[0].column)
ws_sum.column_dimensions[col_letter].width = max(max_len + 4, 12) ws_sum.column_dimensions[col_letter].width = max(max_len + 4, 12)
# Style detail worksheets # Style detail worksheets
for sheet_name in ['Side-by-Side Comparison', 'Header Field Comparison', 'Item SKU Comparison']: for sheet_name in ['Side-by-Side Comparison', 'Header Field Comparison', 'Item SKU Comparison']:
ws = writer.sheets[sheet_name] ws = writer.sheets[sheet_name]
ws.views.sheetView[0].showGridLines = True ws.views.sheetView[0].showGridLines = True
max_row = ws.max_row max_row = ws.max_row
max_col = ws.max_column max_col = ws.max_column
# Header row styling # Header row styling
for col in range(1, max_col + 1): for col in range(1, max_col + 1):
cell = ws.cell(row=1, column=col) cell = ws.cell(row=1, column=col)
cell.font = header_font cell.font = header_font
cell.fill = header_fill cell.fill = header_fill
cell.alignment = center_align cell.alignment = center_align
cell.border = cell_border cell.border = cell_border
# Data rows styling # Data rows styling
for row in range(2, max_row + 1): for row in range(2, max_row + 1):
is_zebra = (row % 2 == 0) is_zebra = (row % 2 == 0)
# For Side-by-Side Comparison # For Side-by-Side Comparison
if sheet_name == 'Side-by-Side Comparison': if sheet_name == 'Side-by-Side Comparison':
for col in range(1, max_col + 1): for col in range(1, max_col + 1):
cell = ws.cell(row=row, column=col) cell = ws.cell(row=row, column=col)
cell.font = regular_font cell.font = regular_font
cell.border = cell_border cell.border = cell_border
if col == 1: if col == 1:
cell.alignment = left_align cell.alignment = left_align
if is_zebra: if is_zebra:
cell.fill = zebra_fill cell.fill = zebra_fill
else: else:
# Apply alignments and color mismatch/match # Apply alignments and color mismatch/match
# Format of headers: # Format of headers:
# Col 1: Filename # Col 1: Filename
# Col 2: PO AI, Col 3: PO Manual, Col 4: PO Status # Col 2: PO AI, Col 3: PO Manual, Col 4: PO Status
# ... and so on # ... and so on
# So status is at index col where (col - 1) % 3 == 0 (4, 7, 10, 13, 16, 19, 22, 25) # So status is at index col where (col - 1) % 3 == 0 (4, 7, 10, 13, 16, 19, 22, 25)
col_pos = col - 1 col_pos = col - 1
if col_pos % 3 == 0: # This is a Status column if col_pos % 3 == 0: # This is a Status column
status_val = cell.value status_val = cell.value
cell.alignment = center_align cell.alignment = center_align
if status_val == "Match": if status_val == "Match":
cell.fill = match_fill cell.fill = match_fill
else: else:
cell.fill = mismatch_fill cell.fill = mismatch_fill
else: # This is AI or Manual value column else: # This is AI or Manual value column
cell.alignment = left_align cell.alignment = left_align
# Match background of the cell with its corresponding status cell (two columns to the right if AI, one if Manual) # Match background of the cell with its corresponding status cell (two columns to the right if AI, one if Manual)
status_col_idx = col + (2 if col_pos % 3 == 1 else 1) status_col_idx = col + (2 if col_pos % 3 == 1 else 1)
status_val = ws.cell(row=row, column=status_col_idx).value status_val = ws.cell(row=row, column=status_col_idx).value
if status_val == "Match": if status_val == "Match":
if is_zebra: if is_zebra:
# Let's keep it subtle # Let's keep it subtle
pass pass
else: else:
# Highlight mismatches clearly # Highlight mismatches clearly
cell.fill = mismatch_fill cell.fill = mismatch_fill
# For vertical comparison sheets # For vertical comparison sheets
else: else:
# Match column is the last column # Match column is the last column
match_cell = ws.cell(row=row, column=max_col) match_cell = ws.cell(row=row, column=max_col)
match_val = match_cell.value match_val = match_cell.value
for col in range(1, max_col + 1): for col in range(1, max_col + 1):
cell = ws.cell(row=row, column=col) cell = ws.cell(row=row, column=col)
cell.font = regular_font cell.font = regular_font
cell.border = cell_border cell.border = cell_border
# Apply alignments based on column # Apply alignments based on column
if col in [1, 2, 3, 4]: if col in [1, 2, 3, 4]:
cell.alignment = left_align cell.alignment = left_align
else: else:
cell.alignment = center_align cell.alignment = center_align
# Color match / mismatch # Color match / mismatch
if match_val == "Match": if match_val == "Match":
cell.fill = match_fill cell.fill = match_fill
elif match_val == "Mismatch": elif match_val == "Mismatch":
cell.fill = mismatch_fill cell.fill = mismatch_fill
elif is_zebra: elif is_zebra:
cell.fill = zebra_fill cell.fill = zebra_fill
# Auto-fit columns # Auto-fit columns
for col in ws.columns: for col in ws.columns:
max_len = 0 max_len = 0
for cell in col: for cell in col:
val_str = str(cell.value or '') val_str = str(cell.value or '')
# Limit long text like Alamat from making column excessively wide # Limit long text like Alamat from making column excessively wide
if sheet_name == 'Side-by-Side Comparison' and cell.column in [22, 23]: # Alamat if sheet_name == 'Side-by-Side Comparison' and cell.column in [22, 23]: # Alamat
max_len = max(max_len, min(len(val_str), 30)) max_len = max(max_len, min(len(val_str), 30))
elif sheet_name == 'Header Field Comparison' and cell.column in [3, 4]: # Values elif sheet_name == 'Header Field Comparison' and cell.column in [3, 4]: # Values
max_len = max(max_len, min(len(val_str), 40)) max_len = max(max_len, min(len(val_str), 40))
else: else:
max_len = max(max_len, len(val_str)) max_len = max(max_len, len(val_str))
col_letter = get_column_letter(col[0].column) col_letter = get_column_letter(col[0].column)
ws.column_dimensions[col_letter].width = max(max_len + 4, 12) ws.column_dimensions[col_letter].width = max(max_len + 4, 12)
print("\n=== Accuracy Report Summary ===") print("\n=== Accuracy Report Summary ===")
print(df_summary.to_string(index=False)) print(df_summary.to_string(index=False))
print("===============================\n") print("===============================\n")
print(f"Comparison report generated at {xlsx_file}") print(f"Comparison report generated at {xlsx_file}")
if __name__ == "__main__": if __name__ == "__main__":
main() main()
File diff suppressed because it is too large. Load diff
+266 -266
View File
@@ -1,266 +1,266 @@
# Expiry-date extraction cascade, split out of classify_ocr_server.py so it # Expiry-date extraction cascade, split out of classify_ocr_server.py so it
# can be imported (and offline-tested against captured OCR lines) without # can be imported (and offline-tested against captured OCR lines) without
# loading any models. Pure regex/string logic - no torch/paddle imports. # loading any models. Pure regex/string logic - no torch/paddle imports.
import re import re
EXP_KEYWORD_RE = re.compile( EXP_KEYWORD_RE = re.compile(
r'(?:exp(?:\.|ired)?|tgl(?:\s*exp)?|expiry|bbd|best\s*before|before|best|baik\s*digunakan|\bbb\b)', r'(?:exp(?:\.|ired)?|tgl(?:\s*exp)?|expiry|bbd|best\s*before|before|best|baik\s*digunakan|\bbb\b)',
re.IGNORECASE, re.IGNORECASE,
) )
DD_MM_YYYY_RE = re.compile( DD_MM_YYYY_RE = re.compile(
r'(?<!\d)(0[1-9]|[12]\d|3[01]).*?(0[1-9]|1[0-2]).*?(20\d{2})(?!\d)' r'(?<!\d)(0[1-9]|[12]\d|3[01]).*?(0[1-9]|1[0-2]).*?(20\d{2})(?!\d)'
) )
DDMMYYYY_RE = re.compile( DDMMYYYY_RE = re.compile(
r'(?<!\d)(0[1-9]|[12]\d|3[01])(0[1-9]|1[0-2])(20\d{2})(?!\d)' r'(?<!\d)(0[1-9]|[12]\d|3[01])(0[1-9]|1[0-2])(20\d{2})(?!\d)'
) )
BB_ATTACHED_DATE_RE = re.compile( BB_ATTACHED_DATE_RE = re.compile(
r'\b(?:bb|bestbefore)\s*[:.-]?\s*(0[1-9]|[12]\d|3[01])(0[1-9]|1[0-2])(20\d{2})(?!\d)', r'\b(?:bb|bestbefore)\s*[:.-]?\s*(0[1-9]|[12]\d|3[01])(0[1-9]|1[0-2])(20\d{2})(?!\d)',
re.IGNORECASE, re.IGNORECASE,
) )
KEYWORD_DIGITS_RE = re.compile( KEYWORD_DIGITS_RE = re.compile(
r'(?:exp|expired|tgl|expiry|bbd|before|best|bb|baik|digunakan)\s*[:.-]?\s*(\d{6,8})\b', r'(?:exp|expired|tgl|expiry|bbd|before|best|bb|baik|digunakan)\s*[:.-]?\s*(\d{6,8})\b',
re.IGNORECASE, re.IGNORECASE,
) )
LENIENT_DATE_RE = re.compile( LENIENT_DATE_RE = re.compile(
r'(?<!\d)(\d{1,2}).*?(\d{1,2}).*?((?:20)?\d{2})(?!\d)' r'(?<!\d)(\d{1,2}).*?(\d{1,2}).*?((?:20)?\d{2})(?!\d)'
) )
# Store price-tag / label-printer lines ("Printed:04/05/2026 19:53", # Store price-tag / label-printer lines ("Printed:04/05/2026 19:53",
# "Rp.6,800/PC"). The date on these is the moment the shelf label was # "Rp.6,800/PC"). The date on these is the moment the shelf label was
# printed, never the product's expiry - excluded from the keyword-less # printed, never the product's expiry - excluded from the keyword-less
# stages so it can't shadow the real date elsewhere on the package. # stages so it can't shadow the real date elsewhere on the package.
PRICE_TAG_RE = re.compile(r'(?i)printed\s*[:.]?|rp\s*[.,]?\s*\d') PRICE_TAG_RE = re.compile(r'(?i)printed\s*[:.]?|rp\s*[.,]?\s*\d')
def is_valid_ddmmyyyy_digits(val: str) -> bool: def is_valid_ddmmyyyy_digits(val: str) -> bool:
if len(val) != 8 or not val.isdigit(): if len(val) != 8 or not val.isdigit():
return False return False
day, month, year = int(val[0:2]), int(val[2:4]), int(val[4:8]) day, month, year = int(val[0:2]), int(val[2:4]), int(val[4:8])
return 1 <= day <= 31 and 1 <= month <= 12 and 2000 <= year <= 2099 return 1 <= day <= 31 and 1 <= month <= 12 and 2000 <= year <= 2099
def is_plausible_date_parts(day: str, month: str, year: str) -> bool: def is_plausible_date_parts(day: str, month: str, year: str) -> bool:
# Sanity gate for the lenient stage: it happily assembles junk like # Sanity gate for the lenient stage: it happily assembles junk like
# "00/22/26" or "1/3/06" out of garbled digit runs. A frozen-food # "00/22/26" or "1/3/06" out of garbled digit runs. A frozen-food
# expiry is always a real calendar day within a few years of today. # expiry is always a real calendar day within a few years of today.
if not (day.isdigit() and month.isdigit() and year.isdigit()): if not (day.isdigit() and month.isdigit() and year.isdigit()):
return False return False
d, m = int(day), int(month) d, m = int(day), int(month)
y = int(year) if len(year) == 4 else 2000 + int(year) y = int(year) if len(year) == 4 else 2000 + int(year)
return 1 <= d <= 31 and 1 <= m <= 12 and 2020 <= y <= 2039 return 1 <= d <= 31 and 1 <= m <= 12 and 2020 <= y <= 2039
def format_ddmmyyyy(val: str) -> str: def format_ddmmyyyy(val: str) -> str:
if is_valid_ddmmyyyy_digits(val): if is_valid_ddmmyyyy_digits(val):
return f"{val[0:2]}/{val[2:4]}/{val[4:8]}" return f"{val[0:2]}/{val[2:4]}/{val[4:8]}"
return val.upper() return val.upper()
def format_ddmmyy(val: str) -> str: def format_ddmmyy(val: str) -> str:
if len(val) == 6 and val.isdigit(): if len(val) == 6 and val.isdigit():
day, month = int(val[0:2]), int(val[2:4]) day, month = int(val[0:2]), int(val[2:4])
if 1 <= day <= 31 and 1 <= month <= 12: if 1 <= day <= 31 and 1 <= month <= 12:
return f"{val[0:2]}/{val[2:4]}/{val[4:6]}" return f"{val[0:2]}/{val[2:4]}/{val[4:6]}"
return val.upper() return val.upper()
def line_has_exp_keyword(line: str) -> bool: def line_has_exp_keyword(line: str) -> bool:
if EXP_KEYWORD_RE.search(line): if EXP_KEYWORD_RE.search(line):
return True return True
# BB05032027 — keyword directly followed by digits # BB05032027 — keyword directly followed by digits
return bool(re.search(r'(?i)\b(?:bb|bestbefore)(?:\s*[:.-]?\s*)?\d', line)) return bool(re.search(r'(?i)\b(?:bb|bestbefore)(?:\s*[:.-]?\s*)?\d', line))
def clean_date_line(line: str) -> str: def clean_date_line(line: str) -> str:
# 1) Replace "1)" with "0" # 1) Replace "1)" with "0"
cleaned = line.replace("1)", "0") cleaned = line.replace("1)", "0")
# 2) Replace "()" with "0" # 2) Replace "()" with "0"
cleaned = cleaned.replace("()", "0") cleaned = cleaned.replace("()", "0")
# Clean BB misrecognitions (convert B8, 8B, 88 to BB when followed by digits) # Clean BB misrecognitions (convert B8, 8B, 88 to BB when followed by digits)
cleaned = re.sub(r'\b(?:B8|8B|88)(?=\d)', 'BB', cleaned) cleaned = re.sub(r'\b(?:B8|8B|88)(?=\d)', 'BB', cleaned)
cleaned = re.sub(r'^(?:B8|8B|88)(?=\d)', 'BB', cleaned) cleaned = re.sub(r'^(?:B8|8B|88)(?=\d)', 'BB', cleaned)
# Clean 012/112 month misrecognitions (e.g. 020122027 -> 02022027) — # Clean 012/112 month misrecognitions (e.g. 020122027 -> 02022027) —
# but only when the line does NOT already hold a valid date: a real # but only when the line does NOT already hold a valid date: a real
# "01122026" (= 01/12/2026) also matches the 112 pattern (0+112+2026) # "01122026" (= 01/12/2026) also matches the 112 pattern (0+112+2026)
# and would be mangled into 7-digit junk. # and would be mangled into 7-digit junk.
if not (DDMMYYYY_RE.search(cleaned) or DD_MM_YYYY_RE.search(cleaned)): if not (DDMMYYYY_RE.search(cleaned) or DD_MM_YYYY_RE.search(cleaned)):
cleaned = re.sub(r'(?<!\d)(\d{1,2})012(20\d{2})(?!\d)', r'\g<1>02\g<2>', cleaned) cleaned = re.sub(r'(?<!\d)(\d{1,2})012(20\d{2})(?!\d)', r'\g<1>02\g<2>', cleaned)
cleaned = re.sub(r'(?<!\d)(\d{1,2})([-./\s]+)012([-./\s]+)(20\d{2})(?!\d)', r'\g<1>\g<2>02\g<3>\g<4>', cleaned) cleaned = re.sub(r'(?<!\d)(\d{1,2})([-./\s]+)012([-./\s]+)(20\d{2})(?!\d)', r'\g<1>\g<2>02\g<3>\g<4>', cleaned)
cleaned = re.sub(r'(?<!\d)(\d{1,2})112(20\d{2})(?!\d)', r'\g<1>02\g<2>', cleaned) cleaned = re.sub(r'(?<!\d)(\d{1,2})112(20\d{2})(?!\d)', r'\g<1>02\g<2>', cleaned)
cleaned = re.sub(r'(?<!\d)(\d{1,2})([-./\s]+)112([-./\s]+)(20\d{2})(?!\d)', r'\g<1>\g<2>02\g<3>\g<4>', cleaned) cleaned = re.sub(r'(?<!\d)(\d{1,2})([-./\s]+)112([-./\s]+)(20\d{2})(?!\d)', r'\g<1>\g<2>02\g<3>\g<4>', cleaned)
# Run contextual replacements # Run contextual replacements
for _ in range(3): for _ in range(3):
# letter o/O flanked by digits or boundary -> 0 # letter o/O flanked by digits or boundary -> 0
cleaned = re.compile(r'(\d)[oO](\d|\b)').sub(r'\g<1>0\g<2>', cleaned) cleaned = re.compile(r'(\d)[oO](\d|\b)').sub(r'\g<1>0\g<2>', cleaned)
cleaned = re.compile(r'(\b|\d)[oO](\d)').sub(r'\g<1>0\g<2>', cleaned) cleaned = re.compile(r'(\b|\d)[oO](\d)').sub(r'\g<1>0\g<2>', cleaned)
# letter I/i/l/| flanked by digits -> 1 # letter I/i/l/| flanked by digits -> 1
cleaned = re.compile(r'(\d)[Ii|l](\d|\b)').sub(r'\g<1>1\g<2>', cleaned) cleaned = re.compile(r'(\d)[Ii|l](\d|\b)').sub(r'\g<1>1\g<2>', cleaned)
cleaned = re.compile(r'(\b|\d)[Ii|l](\d)').sub(r'\g<1>1\g<2>', cleaned) cleaned = re.compile(r'(\b|\d)[Ii|l](\d)').sub(r'\g<1>1\g<2>', cleaned)
# letter S/s flanked by digits -> 5 # letter S/s flanked by digits -> 5
cleaned = re.compile(r'(\d)[Ss](\d|\b)').sub(r'\g<1>5\g<2>', cleaned) cleaned = re.compile(r'(\d)[Ss](\d|\b)').sub(r'\g<1>5\g<2>', cleaned)
cleaned = re.compile(r'(\b|\d)[Ss](\d)').sub(r'\g<1>5\g<2>', cleaned) cleaned = re.compile(r'(\b|\d)[Ss](\d)').sub(r'\g<1>5\g<2>', cleaned)
# letter Z/z flanked by digits -> 2 # letter Z/z flanked by digits -> 2
cleaned = re.compile(r'(\d)[Zz](\d|\b)').sub(r'\g<1>2\g<2>', cleaned) cleaned = re.compile(r'(\d)[Zz](\d|\b)').sub(r'\g<1>2\g<2>', cleaned)
cleaned = re.compile(r'(\b|\d)[Zz](\d)').sub(r'\g<1>2\g<2>', cleaned) cleaned = re.compile(r'(\b|\d)[Zz](\d)').sub(r'\g<1>2\g<2>', cleaned)
# letter B flanked by digits -> 8 # letter B flanked by digits -> 8
cleaned = re.compile(r'(\d)B(\d|\b)').sub(r'\g<1>8\g<2>', cleaned) cleaned = re.compile(r'(\d)B(\d|\b)').sub(r'\g<1>8\g<2>', cleaned)
cleaned = re.compile(r'(\b|\d)B(\d)').sub(r'\g<1>8\g<2>', cleaned) cleaned = re.compile(r'(\b|\d)B(\d)').sub(r'\g<1>8\g<2>', cleaned)
return cleaned return cleaned
def extract_expired_date(text_lines): def extract_expired_date(text_lines):
"""Return (formatted_date, line_index, source_line). Prioritises BB/EXP + DDMMYYYY or DD MM YYYY.""" """Return (formatted_date, line_index, source_line). Prioritises BB/EXP + DDMMYYYY or DD MM YYYY."""
if not text_lines: if not text_lines:
return None, None, None return None, None, None
cleaned_lines = [clean_date_line(line) for line in text_lines] cleaned_lines = [clean_date_line(line) for line in text_lines]
def pick(match, idx, cleaned_line, formatter=None): def pick(match, idx, cleaned_line, formatter=None):
raw = match.group(0) raw = match.group(0)
original_line = text_lines[idx].strip() original_line = text_lines[idx].strip()
if match.lastindex and match.lastindex >= 3: if match.lastindex and match.lastindex >= 3:
formatted = f"{match.group(1)}/{match.group(2)}/{match.group(3)}" formatted = f"{match.group(1)}/{match.group(2)}/{match.group(3)}"
elif match.lastindex and match.lastindex >= 1 and match.group(1).isdigit(): elif match.lastindex and match.lastindex >= 1 and match.group(1).isdigit():
digits = match.group(1) digits = match.group(1)
if len(digits) == 8: if len(digits) == 8:
formatted = format_ddmmyyyy(digits) formatted = format_ddmmyyyy(digits)
elif len(digits) == 6: elif len(digits) == 6:
formatted = format_ddmmyy(digits) formatted = format_ddmmyy(digits)
else: else:
formatted = digits formatted = digits
elif formatter: elif formatter:
formatted = formatter(raw) formatted = formatter(raw)
else: else:
formatted = raw.strip().upper() formatted = raw.strip().upper()
return formatted, idx, original_line return formatted, idx, original_line
# 1) BB/EXP keyword lines — compact DDMMYYYY (e.g. BB05032027, EXP 05032027) # 1) BB/EXP keyword lines — compact DDMMYYYY (e.g. BB05032027, EXP 05032027)
for idx, line in enumerate(cleaned_lines): for idx, line in enumerate(cleaned_lines):
if not line_has_exp_keyword(line): if not line_has_exp_keyword(line):
continue continue
match = BB_ATTACHED_DATE_RE.search(line) or DDMMYYYY_RE.search(line) match = BB_ATTACHED_DATE_RE.search(line) or DDMMYYYY_RE.search(line)
if match: if match:
return pick(match, idx, line) return pick(match, idx, line)
# 2) BB/EXP keyword lines — spaced DD MM YYYY (e.g. BB 05 03 2027) # 2) BB/EXP keyword lines — spaced DD MM YYYY (e.g. BB 05 03 2027)
for idx, line in enumerate(cleaned_lines): for idx, line in enumerate(cleaned_lines):
if not line_has_exp_keyword(line): if not line_has_exp_keyword(line):
continue continue
match = DD_MM_YYYY_RE.search(line) match = DD_MM_YYYY_RE.search(line)
if match: if match:
return pick(match, idx, line) return pick(match, idx, line)
# 3) Keyword + 6–8 digit run (BB05032027 via keyword_digits) # 3) Keyword + 6–8 digit run (BB05032027 via keyword_digits)
for idx, line in enumerate(cleaned_lines): for idx, line in enumerate(cleaned_lines):
match = KEYWORD_DIGITS_RE.search(line) match = KEYWORD_DIGITS_RE.search(line)
if match: if match:
digits = match.group(1) digits = match.group(1)
if len(digits) == 8 and is_valid_ddmmyyyy_digits(digits): if len(digits) == 8 and is_valid_ddmmyyyy_digits(digits):
return format_ddmmyyyy(digits), idx, text_lines[idx].strip() return format_ddmmyyyy(digits), idx, text_lines[idx].strip()
if len(digits) == 6: if len(digits) == 6:
return format_ddmmyy(digits), idx, text_lines[idx].strip() return format_ddmmyy(digits), idx, text_lines[idx].strip()
# 3.5) BB/EXP keyword lines — lenient check for unclear/noisy date formats (e.g. BB 02J 132027) # 3.5) BB/EXP keyword lines — lenient check for unclear/noisy date formats (e.g. BB 02J 132027)
for idx, line in enumerate(cleaned_lines): for idx, line in enumerate(cleaned_lines):
if not line_has_exp_keyword(line): if not line_has_exp_keyword(line):
continue continue
match = LENIENT_DATE_RE.search(line) match = LENIENT_DATE_RE.search(line)
if match and is_plausible_date_parts(match.group(1), match.group(2), match.group(3)): if match and is_plausible_date_parts(match.group(1), match.group(2), match.group(3)):
return pick(match, idx, line) return pick(match, idx, line)
# 3.6) Keyword line + date split onto an adjacent line (PaddleOCR sometimes # 3.6) Keyword line + date split onto an adjacent line (PaddleOCR sometimes
# detects "BB"/"Baik digunakan" as its own box, separate from the date # detects "BB"/"Baik digunakan" as its own box, separate from the date
# digits in a neighboring box, e.g. "BB" / "05032027" as two lines). # digits in a neighboring box, e.g. "BB" / "05032027" as two lines).
for idx, line in enumerate(cleaned_lines): for idx, line in enumerate(cleaned_lines):
if not line_has_exp_keyword(line): if not line_has_exp_keyword(line):
continue continue
for j in (idx + 1, idx - 1, idx + 2): for j in (idx + 1, idx - 1, idx + 2):
if j < 0 or j >= len(cleaned_lines) or j == idx: if j < 0 or j >= len(cleaned_lines) or j == idx:
continue continue
neighbor = cleaned_lines[j] neighbor = cleaned_lines[j]
combined = f"{line} {neighbor}" if j > idx else f"{neighbor} {line}" combined = f"{line} {neighbor}" if j > idx else f"{neighbor} {line}"
match = (BB_ATTACHED_DATE_RE.search(combined) match = (BB_ATTACHED_DATE_RE.search(combined)
or DDMMYYYY_RE.search(combined) or DDMMYYYY_RE.search(combined)
or DD_MM_YYYY_RE.search(combined)) or DD_MM_YYYY_RE.search(combined))
if match: if match:
report_idx = j if sum(c.isdigit() for c in neighbor) > sum(c.isdigit() for c in line) else idx report_idx = j if sum(c.isdigit() for c in neighbor) > sum(c.isdigit() for c in line) else idx
return pick(match, report_idx, combined) return pick(match, report_idx, combined)
# 4) Any line — spaced DD MM YYYY (excluding store price-tag lines) # 4) Any line — spaced DD MM YYYY (excluding store price-tag lines)
for idx, line in enumerate(cleaned_lines): for idx, line in enumerate(cleaned_lines):
if PRICE_TAG_RE.search(line): if PRICE_TAG_RE.search(line):
continue continue
match = DD_MM_YYYY_RE.search(line) match = DD_MM_YYYY_RE.search(line)
if match: if match:
return pick(match, idx, line) return pick(match, idx, line)
# 5) Any line — compact DDMMYYYY (skip likely SKU: same line has 8-digit product code context) # 5) Any line — compact DDMMYYYY (skip likely SKU: same line has 8-digit product code context)
for idx, line in enumerate(cleaned_lines): for idx, line in enumerate(cleaned_lines):
if PRICE_TAG_RE.search(line): if PRICE_TAG_RE.search(line):
continue continue
for match in DDMMYYYY_RE.finditer(line): for match in DDMMYYYY_RE.finditer(line):
digits = f"{match.group(1)}{match.group(2)}{match.group(3)}" digits = f"{match.group(1)}{match.group(2)}{match.group(3)}"
if is_valid_ddmmyyyy_digits(digits): if is_valid_ddmmyyyy_digits(digits):
# Skip if this 8-digit block is the only digits and looks like SKU on label top # Skip if this 8-digit block is the only digits and looks like SKU on label top
if re.search(r'\b\d{8}\b', line) and not line_has_exp_keyword(line): if re.search(r'\b\d{8}\b', line) and not line_has_exp_keyword(line):
if re.search(r'(?:nugget|chicken|fiesta|champ|okey|akumo|frozen|gr)', line, re.I): if re.search(r'(?:nugget|chicken|fiesta|champ|okey|akumo|frozen|gr)', line, re.I):
continue continue
return format_ddmmyyyy(digits), idx, text_lines[idx].strip() return format_ddmmyyyy(digits), idx, text_lines[idx].strip()
# 6) Legacy patterns (slashes, month names, etc.) # 6) Legacy patterns (slashes, month names, etc.)
date_patterns = [ date_patterns = [
r'\b\d{2}[-./]\d{2}[-./]\d{2,4}\b', r'\b\d{2}[-./]\d{2}[-./]\d{2,4}\b',
r'\b\d{4}[-./]\d{2}[-./]\d{2}\b', r'\b\d{4}[-./]\d{2}[-./]\d{2}\b',
r'\b\d{2}\s+(?:JAN|FEB|MAR|APR|MAY|JUN|JUL|AUG|SEP|OCT|NOV|DEC)[a-zA-Z]*\s+\d{2,4}\b', r'\b\d{2}\s+(?:JAN|FEB|MAR|APR|MAY|JUN|JUL|AUG|SEP|OCT|NOV|DEC)[a-zA-Z]*\s+\d{2,4}\b',
] ]
for idx, line in enumerate(cleaned_lines): for idx, line in enumerate(cleaned_lines):
if not line_has_exp_keyword(line): if not line_has_exp_keyword(line):
continue continue
for pat in date_patterns: for pat in date_patterns:
match = re.search(pat, line, re.IGNORECASE) match = re.search(pat, line, re.IGNORECASE)
if match: if match:
return match.group(0).upper(), idx, text_lines[idx].strip() return match.group(0).upper(), idx, text_lines[idx].strip()
return None, None, None return None, None, None
def line_contains_expired_date(line: str, expired_date: str) -> bool: def line_contains_expired_date(line: str, expired_date: str) -> bool:
if not line or not expired_date: if not line or not expired_date:
return False return False
digits_only = re.sub(r"\D", "", expired_date) digits_only = re.sub(r"\D", "", expired_date)
line_digits = re.sub(r"\D", "", line) line_digits = re.sub(r"\D", "", line)
if len(digits_only) >= 6 and digits_only in line_digits: if len(digits_only) >= 6 and digits_only in line_digits:
return True return True
compact = expired_date.replace("/", "") compact = expired_date.replace("/", "")
return compact in line.replace(" ", "") or expired_date in line return compact in line.replace(" ", "") or expired_date in line
def find_expired_crop_index(text_lines, expired_idx, expired_date, polys_len): def find_expired_crop_index(text_lines, expired_idx, expired_date, polys_len):
"""Pick OCR box index for cropping; prefer the line that actually contains the date.""" """Pick OCR box index for cropping; prefer the line that actually contains the date."""
if not expired_date or polys_len <= 0: if not expired_date or polys_len <= 0:
return None return None
if ( if (
expired_idx is not None expired_idx is not None
and expired_idx < polys_len and expired_idx < polys_len
and expired_idx < len(text_lines) and expired_idx < len(text_lines)
and line_contains_expired_date(text_lines[expired_idx], expired_date) and line_contains_expired_date(text_lines[expired_idx], expired_date)
): ):
return expired_idx return expired_idx
keyword_match = None keyword_match = None
for idx, line in enumerate(text_lines): for idx, line in enumerate(text_lines):
if idx >= polys_len: if idx >= polys_len:
break break
if not line_contains_expired_date(line, expired_date): if not line_contains_expired_date(line, expired_date):
continue continue
if line_has_exp_keyword(line): if line_has_exp_keyword(line):
return idx return idx
if keyword_match is None: if keyword_match is None:
keyword_match = idx keyword_match = idx
if keyword_match is not None: if keyword_match is not None:
return keyword_match return keyword_match
if expired_idx is not None and expired_idx < polys_len: if expired_idx is not None and expired_idx < polys_len:
return expired_idx return expired_idx
return None return None
+85 -85
View File
@@ -1,85 +1,85 @@
pipeline_name: PaddleOCR-VL-1.6 pipeline_name: PaddleOCR-VL-1.6
batch_size: 64 batch_size: 64
use_queues: True use_queues: True
use_doc_preprocessor: True use_doc_preprocessor: True
use_layout_detection: True use_layout_detection: True
use_chart_recognition: False use_chart_recognition: False
use_seal_recognition: False use_seal_recognition: False
format_block_content: False format_block_content: False
merge_layout_blocks: True merge_layout_blocks: True
markdown_ignore_labels: [] markdown_ignore_labels: []
# - number # - number
# - footnote # - footnote
# - header # - header
# - header_image # - header_image
# - footer # - footer
# - footer_image # - footer_image
# - aside_text # - aside_text
SubModules: SubModules:
LayoutDetection: LayoutDetection:
module_name: layout_detection module_name: layout_detection
model_name: PP-DocLayoutV3 model_name: PP-DocLayoutV3
model_dir: null model_dir: null
batch_size: 8 batch_size: 8
threshold: 0.2 threshold: 0.2
layout_nms: True layout_nms: True
layout_unclip_ratio: [1.0, 1.0] layout_unclip_ratio: [1.0, 1.0]
layout_merge_bboxes_mode: layout_merge_bboxes_mode:
0: "union" 0: "union"
1: "union" 1: "union"
2: "union" 2: "union"
3: "large" 3: "large"
4: "union" 4: "union"
5: "large" 5: "large"
6: "large" 6: "large"
7: "union" 7: "union"
8: "union" 8: "union"
9: "union" 9: "union"
10: "union" 10: "union"
11: "union" 11: "union"
12: "union" 12: "union"
13: "union" 13: "union"
14: "union" 14: "union"
15: "large" 15: "large"
16: "union" 16: "union"
17: "large" 17: "large"
18: "union" 18: "union"
19: "union" 19: "union"
20: "union" 20: "union"
21: "union" 21: "union"
22: "union" 22: "union"
23: "union" 23: "union"
24: "union" 24: "union"
VLRecognition: VLRecognition:
module_name: vl_recognition module_name: vl_recognition
model_name: PaddleOCR-VL-1.6-0.9B model_name: PaddleOCR-VL-1.6-0.9B
model_dir: null model_dir: null
batch_size: 4096 batch_size: 4096
genai_config: genai_config:
backend: vllm-server backend: vllm-server
server_url: http://127.0.0.1:8118/v1 server_url: http://127.0.0.1:8118/v1
SubPipelines: SubPipelines:
DocPreprocessor: DocPreprocessor:
pipeline_name: doc_preprocessor pipeline_name: doc_preprocessor
batch_size: 8 batch_size: 8
use_doc_orientation_classify: True use_doc_orientation_classify: True
use_doc_unwarping: True use_doc_unwarping: True
SubModules: SubModules:
DocOrientationClassify: DocOrientationClassify:
module_name: doc_text_orientation module_name: doc_text_orientation
model_name: PP-LCNet_x1_0_doc_ori model_name: PP-LCNet_x1_0_doc_ori
model_dir: null model_dir: null
batch_size: 8 batch_size: 8
DocUnwarping: DocUnwarping:
module_name: image_unwarping module_name: image_unwarping
model_name: UVDoc model_name: UVDoc
model_dir: null model_dir: null
Serving: Serving:
extra: extra:
max_num_input_imgs: null max_num_input_imgs: null
+69 -69
View File
@@ -1,69 +1,69 @@
import re import re
def clean_date_line(line: str) -> str: def clean_date_line(line: str) -> str:
# 1) Replace "1)" with "0" # 1) Replace "1)" with "0"
cleaned = line.replace("1)", "0") cleaned = line.replace("1)", "0")
# 2) Replace "()" with "0" # 2) Replace "()" with "0"
cleaned = cleaned.replace("()", "0") cleaned = cleaned.replace("()", "0")
# Clean BB misrecognitions (convert B8, 8B, 88 to BB when followed by digits) # Clean BB misrecognitions (convert B8, 8B, 88 to BB when followed by digits)
cleaned = re.sub(r'\b(?:B8|8B|88)(?=\d)', 'BB', cleaned) cleaned = re.sub(r'\b(?:B8|8B|88)(?=\d)', 'BB', cleaned)
cleaned = re.sub(r'^(?:B8|8B|88)(?=\d)', 'BB', cleaned) cleaned = re.sub(r'^(?:B8|8B|88)(?=\d)', 'BB', cleaned)
# Clean 012 month misrecognition (e.g. 020122027 -> 02022027) # Clean 012 month misrecognition (e.g. 020122027 -> 02022027)
cleaned = re.sub(r'(?<!\d)(\d{1,2})012(20\d{2})(?!\d)', r'\g<1>02\g<2>', cleaned) cleaned = re.sub(r'(?<!\d)(\d{1,2})012(20\d{2})(?!\d)', r'\g<1>02\g<2>', cleaned)
cleaned = re.sub(r'(?<!\d)(\d{1,2})([-./\s]+)012([-./\s]+)(20\d{2})(?!\d)', r'\g<1>\g<2>02\g<3>\g<4>', cleaned) cleaned = re.sub(r'(?<!\d)(\d{1,2})([-./\s]+)012([-./\s]+)(20\d{2})(?!\d)', r'\g<1>\g<2>02\g<3>\g<4>', cleaned)
# Clean 112 month misrecognition (e.g. 021122027 -> 02022027) # Clean 112 month misrecognition (e.g. 021122027 -> 02022027)
cleaned = re.sub(r'(?<!\d)(\d{1,2})112(20\d{2})(?!\d)', r'\g<1>02\g<2>', cleaned) cleaned = re.sub(r'(?<!\d)(\d{1,2})112(20\d{2})(?!\d)', r'\g<1>02\g<2>', cleaned)
cleaned = re.sub(r'(?<!\d)(\d{1,2})([-./\s]+)112([-./\s]+)(20\d{2})(?!\d)', r'\g<1>\g<2>02\g<3>\g<4>', cleaned) cleaned = re.sub(r'(?<!\d)(\d{1,2})([-./\s]+)112([-./\s]+)(20\d{2})(?!\d)', r'\g<1>\g<2>02\g<3>\g<4>', cleaned)
# Run contextual replacements # Run contextual replacements
for _ in range(3): for _ in range(3):
# letter o/O flanked by digits or boundary -> 0 # letter o/O flanked by digits or boundary -> 0
cleaned = re.compile(r'(\d)[oO](\d|\b)').sub(r'\g<1>0\g<2>', cleaned) cleaned = re.compile(r'(\d)[oO](\d|\b)').sub(r'\g<1>0\g<2>', cleaned)
cleaned = re.compile(r'(\b|\d)[oO](\d)').sub(r'\g<1>0\g<2>', cleaned) cleaned = re.compile(r'(\b|\d)[oO](\d)').sub(r'\g<1>0\g<2>', cleaned)
# letter I/i/l/| flanked by digits -> 1 # letter I/i/l/| flanked by digits -> 1
cleaned = re.compile(r'(\d)[Ii|l](\d|\b)').sub(r'\g<1>1\g<2>', cleaned) cleaned = re.compile(r'(\d)[Ii|l](\d|\b)').sub(r'\g<1>1\g<2>', cleaned)
cleaned = re.compile(r'(\b|\d)[Ii|l](\d)').sub(r'\g<1>1\g<2>', cleaned) cleaned = re.compile(r'(\b|\d)[Ii|l](\d)').sub(r'\g<1>1\g<2>', cleaned)
# letter S/s flanked by digits -> 5 # letter S/s flanked by digits -> 5
cleaned = re.compile(r'(\d)[Ss](\d|\b)').sub(r'\g<1>5\g<2>', cleaned) cleaned = re.compile(r'(\d)[Ss](\d|\b)').sub(r'\g<1>5\g<2>', cleaned)
cleaned = re.compile(r'(\b|\d)[Ss](\d)').sub(r'\g<1>5\g<2>', cleaned) cleaned = re.compile(r'(\b|\d)[Ss](\d)').sub(r'\g<1>5\g<2>', cleaned)
# letter Z/z flanked by digits -> 2 # letter Z/z flanked by digits -> 2
cleaned = re.compile(r'(\d)[Zz](\d|\b)').sub(r'\g<1>2\g<2>', cleaned) cleaned = re.compile(r'(\d)[Zz](\d|\b)').sub(r'\g<1>2\g<2>', cleaned)
cleaned = re.compile(r'(\b|\d)[Zz](\d)').sub(r'\g<1>2\g<2>', cleaned) cleaned = re.compile(r'(\b|\d)[Zz](\d)').sub(r'\g<1>2\g<2>', cleaned)
# letter B flanked by digits -> 8 # letter B flanked by digits -> 8
cleaned = re.compile(r'(\d)B(\d|\b)').sub(r'\g<1>8\g<2>', cleaned) cleaned = re.compile(r'(\d)B(\d|\b)').sub(r'\g<1>8\g<2>', cleaned)
cleaned = re.compile(r'(\b|\d)B(\d)').sub(r'\g<1>8\g<2>', cleaned) cleaned = re.compile(r'(\b|\d)B(\d)').sub(r'\g<1>8\g<2>', cleaned)
return cleaned return cleaned
test_cases = [ test_cases = [
"231)92026", "231)92026",
"23o92026", "23o92026",
"23O92026", "23O92026",
"2309202l", "2309202l",
"2309202I", "2309202I",
"230920Z6", "230920Z6",
"230920s6", "230920s6",
"2309202B", "2309202B",
"BB 231)92026", "BB 231)92026",
"BB: 23()92026", "BB: 23()92026",
"12010111", "12010111",
"B8021122027", "B8021122027",
"88021122027", "88021122027",
"020122027", "020122027",
"BB 02/012/2027", "BB 02/012/2027",
"021122027", "021122027",
"BB 02/112/2027" "BB 02/112/2027"
] ]
for tc in test_cases: for tc in test_cases:
cleaned = clean_date_line(tc) cleaned = clean_date_line(tc)
print(f"Original: {tc:<18} -> Cleaned: {cleaned}") print(f"Original: {tc:<18} -> Cleaned: {cleaned}")
+7 -7
View File
@@ -1,7 +1,7 @@
# vLLM backend tuning for paddleocr genai_server # vLLM backend tuning for paddleocr genai_server
# Docs: https://www.paddleocr.ai/latest/en/version3.x/pipeline_usage/PaddleOCR-VL.html#331-server-side-parameter-adjustment # Docs: https://www.paddleocr.ai/latest/en/version3.x/pipeline_usage/PaddleOCR-VL.html#331-server-side-parameter-adjustment
gpu-memory-utilization: 0.35 gpu-memory-utilization: 0.35
max-num-seqs: 4 max-num-seqs: 4
enforce-eager: true enforce-eager: true
max-model-len: 2048 max-model-len: 2048
max-num-batched-tokens: 2048 max-num-batched-tokens: 2048
+8 -8
View File
@@ -1,8 +1,8 @@
#!/bin/bash #!/bin/bash
# Create the target directory inside the Next.js app # Create the target directory inside the Next.js app
mkdir -p pfm-web-app/public/do-pfm mkdir -p pfm-web-app/public/do-pfm
# Copy DO-PFM images # Copy DO-PFM images
cp -v PaddleOCR-VL-1.6_Online_Demo/examples/do-pfm/*.jpg pfm-web-app/public/do-pfm/ cp -v PaddleOCR-VL-1.6_Online_Demo/examples/do-pfm/*.jpg pfm-web-app/public/do-pfm/
echo "DO-PFM examples copied successfully!" echo "DO-PFM examples copied successfully!"
+110 -110
View File
@@ -1,110 +1,110 @@
# Database Entity Relationship Diagram (ERD) # Database Entity Relationship Diagram (ERD)
This document describes the PostgreSQL database schema used to store OCR documents, parsed layout elements, inline cell edits, and row flagging status for the DO-PFM system. This document describes the PostgreSQL database schema used to store OCR documents, parsed layout elements, inline cell edits, and row flagging status for the DO-PFM system.
## Relationship Diagram ## Relationship Diagram
```mermaid ```mermaid
erDiagram erDiagram
documents { documents {
integer id PK "SERIAL" integer id PK "SERIAL"
varchar filename UK "VARCHAR(255)" varchar filename UK "VARCHAR(255)"
timestamp upload_time "TIMESTAMP" timestamp upload_time "TIMESTAMP"
integer size "INTEGER" integer size "INTEGER"
boolean parsed "BOOLEAN" boolean parsed "BOOLEAN"
jsonb metadata "JSONB" jsonb metadata "JSONB"
jsonb layout_parsing_result "JSONB" jsonb layout_parsing_result "JSONB"
boolean is_sample "BOOLEAN" boolean is_sample "BOOLEAN"
varchar file_hash "VARCHAR(64)" varchar file_hash "VARCHAR(64)"
} }
ocr_items { ocr_items {
integer id PK "SERIAL" integer id PK "SERIAL"
integer document_id FK "INTEGER" integer document_id FK "INTEGER"
integer row_index "INTEGER" integer row_index "INTEGER"
varchar kode_barang_original "VARCHAR(255)" varchar kode_barang_original "VARCHAR(255)"
varchar kode_barang "VARCHAR(255)" varchar kode_barang "VARCHAR(255)"
varchar nama_barang "VARCHAR(255)" varchar nama_barang "VARCHAR(255)"
varchar banyak_original "VARCHAR(255)" varchar banyak_original "VARCHAR(255)"
varchar banyak "VARCHAR(255)" varchar banyak "VARCHAR(255)"
varchar jumlah_original "VARCHAR(255)" varchar jumlah_original "VARCHAR(255)"
varchar jumlah "VARCHAR(255)" varchar jumlah "VARCHAR(255)"
boolean is_flagged "BOOLEAN" boolean is_flagged "BOOLEAN"
varchar remark "VARCHAR(1000)" varchar remark "VARCHAR(1000)"
} }
documents ||--o{ ocr_items : "has" documents ||--o{ ocr_items : "has"
vendors { vendors {
integer id PK "SERIAL" integer id PK "SERIAL"
varchar name UK "VARCHAR(255)" varchar name UK "VARCHAR(255)"
timestamp created_at "TIMESTAMP" timestamp created_at "TIMESTAMP"
} }
customers { customers {
integer id PK "SERIAL" integer id PK "SERIAL"
varchar name UK "VARCHAR(255)" varchar name UK "VARCHAR(255)"
timestamp created_at "TIMESTAMP" timestamp created_at "TIMESTAMP"
} }
``` ```
## Schema Definitions ## Schema Definitions
### 1. `documents` Table ### 1. `documents` Table
Stores parsed OCR files (both static sample pages and user-uploaded invoices/documents). Stores parsed OCR files (both static sample pages and user-uploaded invoices/documents).
| Column | Type | Constraints | Description | | Column | Type | Constraints | Description |
|---|---|---|---| |---|---|---|---|
| `id` | `SERIAL` | `PRIMARY KEY` | Unique autoincrement ID of the document. | | `id` | `SERIAL` | `PRIMARY KEY` | Unique autoincrement ID of the document. |
| `filename` | `VARCHAR(255)` | `UNIQUE`, `NOT NULL` | The unique name of the document file. | | `filename` | `VARCHAR(255)` | `UNIQUE`, `NOT NULL` | The unique name of the document file. |
| `upload_time` | `TIMESTAMP` | `DEFAULT NOW()`, `NOT NULL` | The timestamp of the file upload. | | `upload_time` | `TIMESTAMP` | `DEFAULT NOW()`, `NOT NULL` | The timestamp of the file upload. |
| `size` | `INTEGER` | `DEFAULT 0`, `NOT NULL` | The file size in bytes. | | `size` | `INTEGER` | `DEFAULT 0`, `NOT NULL` | The file size in bytes. |
| `parsed` | `BOOLEAN` | `DEFAULT FALSE`, `NOT NULL` | Indicates whether the document layout parsing has completed. | | `parsed` | `BOOLEAN` | `DEFAULT FALSE`, `NOT NULL` | Indicates whether the document layout parsing has completed. |
| `metadata` | `JSONB` | | Structured general metadata (Vendor, Customer, PO, SO, DO, etc.). | | `metadata` | `JSONB` | | Structured general metadata (Vendor, Customer, PO, SO, DO, etc.). |
| `layout_parsing_result` | `JSONB` | | Raw layout parser response JSON from pipeline backend. | | `layout_parsing_result` | `JSONB` | | Raw layout parser response JSON from pipeline backend. |
| `is_sample` | `BOOLEAN` | `DEFAULT FALSE`, `NOT NULL` | True if the file belongs to the pre-seeded static sample pages. | | `is_sample` | `BOOLEAN` | `DEFAULT FALSE`, `NOT NULL` | True if the file belongs to the pre-seeded static sample pages. |
| `file_hash` | `VARCHAR(64)` | | SHA-256 hash of the document file contents. | | `file_hash` | `VARCHAR(64)` | | SHA-256 hash of the document file contents. |
--- ---
### 2. `ocr_items` Table ### 2. `ocr_items` Table
Stores the extracted row items from tabular components of the document, supporting inline modifications and flagging details. Stores the extracted row items from tabular components of the document, supporting inline modifications and flagging details.
| Column | Type | Constraints | Description | | Column | Type | Constraints | Description |
|---|---|---|---| |---|---|---|---|
| `id` | `SERIAL` | `PRIMARY KEY` | Unique autoincrement ID of the item row. | | `id` | `SERIAL` | `PRIMARY KEY` | Unique autoincrement ID of the item row. |
| `document_id` | `INTEGER` | `REFERENCES documents(id) ON DELETE CASCADE`, `NOT NULL` | The associated document ID. | | `document_id` | `INTEGER` | `REFERENCES documents(id) ON DELETE CASCADE`, `NOT NULL` | The associated document ID. |
| `row_index` | `INTEGER` | `NOT NULL` | The index of the item row in the document table list (0-indexed). | | `row_index` | `INTEGER` | `NOT NULL` | The index of the item row in the document table list (0-indexed). |
| `kode_barang_original` | `VARCHAR(255)` | | The initial "Kode Barang" value extracted directly from OCR. | | `kode_barang_original` | `VARCHAR(255)` | | The initial "Kode Barang" value extracted directly from OCR. |
| `kode_barang` | `VARCHAR(255)` | | The edited/current "Kode Barang" value. | | `kode_barang` | `VARCHAR(255)` | | The edited/current "Kode Barang" value. |
| `nama_barang` | `VARCHAR(255)` | | The "Nama Barang" value (read-only reference). | | `nama_barang` | `VARCHAR(255)` | | The "Nama Barang" value (read-only reference). |
| `banyak_original` | `VARCHAR(255)` | | The initial "Banyak" value extracted from OCR. | | `banyak_original` | `VARCHAR(255)` | | The initial "Banyak" value extracted from OCR. |
| `banyak` | `VARCHAR(255)` | | The edited/current "Banyak" value. | | `banyak` | `VARCHAR(255)` | | The edited/current "Banyak" value. |
| `jumlah_original` | `VARCHAR(255)` | | The initial "Jumlah" value extracted from OCR. | | `jumlah_original` | `VARCHAR(255)` | | The initial "Jumlah" value extracted from OCR. |
| `jumlah` | `VARCHAR(255)` | | The edited/current "Jumlah" value. | | `jumlah` | `VARCHAR(255)` | | The edited/current "Jumlah" value. |
| `is_flagged` | `BOOLEAN` | `DEFAULT FALSE`, `NOT NULL` | True if the line item is flagged/strikethrough ("dicoret"). | | `is_flagged` | `BOOLEAN` | `DEFAULT FALSE`, `NOT NULL` | True if the line item is flagged/strikethrough ("dicoret"). |
| `remark` | `VARCHAR(1000)` | | Custom notes/remarks provided for flagging. | | `remark` | `VARCHAR(1000)` | | Custom notes/remarks provided for flagging. |
* **Unique Constraints**: A unique index on `(document_id, row_index)` prevents duplicate indexes for the same page. * **Unique Constraints**: A unique index on `(document_id, row_index)` prevents duplicate indexes for the same page.
--- ---
### 3. `vendors` Table ### 3. `vendors` Table
Stores the Vendor Master registry. Stores the Vendor Master registry.
| Column | Type | Constraints | Description | | Column | Type | Constraints | Description |
|---|---|---|---| |---|---|---|---|
| `id` | `SERIAL` | `PRIMARY KEY` | Unique autoincrement ID of the vendor. | | `id` | `SERIAL` | `PRIMARY KEY` | Unique autoincrement ID of the vendor. |
| `name` | `VARCHAR(255)` | `UNIQUE`, `NOT NULL` | The unique name of the vendor (e.g. including kawasan/address). | | `name` | `VARCHAR(255)` | `UNIQUE`, `NOT NULL` | The unique name of the vendor (e.g. including kawasan/address). |
| `created_at` | `TIMESTAMP` | `DEFAULT NOW()`, `NOT NULL` | The registration timestamp. | | `created_at` | `TIMESTAMP` | `DEFAULT NOW()`, `NOT NULL` | The registration timestamp. |
--- ---
### 4. `customers` Table ### 4. `customers` Table
Stores the Customer Master registry. Stores the Customer Master registry.
| Column | Type | Constraints | Description | | Column | Type | Constraints | Description |
|---|---|---|---| |---|---|---|---|
| `id` | `SERIAL` | `PRIMARY KEY` | Unique autoincrement ID of the customer. | | `id` | `SERIAL` | `PRIMARY KEY` | Unique autoincrement ID of the customer. |
| `name` | `VARCHAR(255)` | `UNIQUE`, `NOT NULL` | The unique name of the customer (e.g. including branch/address). | | `name` | `VARCHAR(255)` | `UNIQUE`, `NOT NULL` | The unique name of the customer (e.g. including branch/address). |
| `created_at` | `TIMESTAMP` | `DEFAULT NOW()`, `NOT NULL` | The registration timestamp. | | `created_at` | `TIMESTAMP` | `DEFAULT NOW()`, `NOT NULL` | The registration timestamp. |
+29 -29
View File
@@ -1,29 +1,29 @@
-- Migration: 001_init_schema -- Migration: 001_init_schema
-- Description: Initialize schema for documents and ocr_items -- Description: Initialize schema for documents and ocr_items
CREATE TABLE IF NOT EXISTS documents ( CREATE TABLE IF NOT EXISTS documents (
id SERIAL PRIMARY KEY, id SERIAL PRIMARY KEY,
filename VARCHAR(255) UNIQUE NOT NULL, filename VARCHAR(255) UNIQUE NOT NULL,
upload_time TIMESTAMP NOT NULL DEFAULT NOW(), upload_time TIMESTAMP NOT NULL DEFAULT NOW(),
size INTEGER NOT NULL DEFAULT 0, size INTEGER NOT NULL DEFAULT 0,
parsed BOOLEAN NOT NULL DEFAULT FALSE, parsed BOOLEAN NOT NULL DEFAULT FALSE,
metadata JSONB, metadata JSONB,
layout_parsing_result JSONB, layout_parsing_result JSONB,
is_sample BOOLEAN NOT NULL DEFAULT FALSE is_sample BOOLEAN NOT NULL DEFAULT FALSE
); );
CREATE TABLE IF NOT EXISTS ocr_items ( CREATE TABLE IF NOT EXISTS ocr_items (
id SERIAL PRIMARY KEY, id SERIAL PRIMARY KEY,
document_id INTEGER NOT NULL REFERENCES documents(id) ON DELETE CASCADE, document_id INTEGER NOT NULL REFERENCES documents(id) ON DELETE CASCADE,
row_index INTEGER NOT NULL, row_index INTEGER NOT NULL,
kode_barang_original VARCHAR(255), kode_barang_original VARCHAR(255),
kode_barang VARCHAR(255), kode_barang VARCHAR(255),
nama_barang VARCHAR(255), nama_barang VARCHAR(255),
banyak_original VARCHAR(255), banyak_original VARCHAR(255),
banyak VARCHAR(255), banyak VARCHAR(255),
jumlah_original VARCHAR(255), jumlah_original VARCHAR(255),
jumlah VARCHAR(255), jumlah VARCHAR(255),
is_flagged BOOLEAN NOT NULL DEFAULT FALSE, is_flagged BOOLEAN NOT NULL DEFAULT FALSE,
remark VARCHAR(1000), remark VARCHAR(1000),
UNIQUE(document_id, row_index) UNIQUE(document_id, row_index)
); );
+4 -4
View File
@@ -1,4 +1,4 @@
-- Migration: 002_add_file_hash -- Migration: 002_add_file_hash
-- Description: Add file_hash column to documents table for duplicate content detection -- Description: Add file_hash column to documents table for duplicate content detection
ALTER TABLE documents ADD COLUMN IF NOT EXISTS file_hash VARCHAR(64); ALTER TABLE documents ADD COLUMN IF NOT EXISTS file_hash VARCHAR(64);
@@ -1,12 +1,12 @@
-- Migration: 003_create_vendor_master -- Migration: 003_create_vendor_master
-- Description: Create vendors table and seed the initial vendor entry -- Description: Create vendors table and seed the initial vendor entry
CREATE TABLE IF NOT EXISTS vendors ( CREATE TABLE IF NOT EXISTS vendors (
id SERIAL PRIMARY KEY, id SERIAL PRIMARY KEY,
name VARCHAR(255) UNIQUE NOT NULL, name VARCHAR(255) UNIQUE NOT NULL,
created_at TIMESTAMP NOT NULL DEFAULT NOW() created_at TIMESTAMP NOT NULL DEFAULT NOW()
); );
INSERT INTO vendors (name) INSERT INTO vendors (name)
VALUES ('PT. CHAROEN POKPHAND INDONESIA Tbk KAWASAN INDUSTRI MODERN, BANTEN') VALUES ('PT. CHAROEN POKPHAND INDONESIA Tbk KAWASAN INDUSTRI MODERN, BANTEN')
ON CONFLICT (name) DO NOTHING; ON CONFLICT (name) DO NOTHING;
@@ -1,12 +1,12 @@
-- Migration: 004_create_customer_master -- Migration: 004_create_customer_master
-- Description: Create customers table and seed the initial customer entry -- Description: Create customers table and seed the initial customer entry
CREATE TABLE IF NOT EXISTS customers ( CREATE TABLE IF NOT EXISTS customers (
id SERIAL PRIMARY KEY, id SERIAL PRIMARY KEY,
name VARCHAR(255) UNIQUE NOT NULL, name VARCHAR(255) UNIQUE NOT NULL,
created_at TIMESTAMP NOT NULL DEFAULT NOW() created_at TIMESTAMP NOT NULL DEFAULT NOW()
); );
INSERT INTO customers (name) INSERT INTO customers (name)
VALUES ('PT.PRIMAFOOD INTERNATIONAL, JL. ANCOL BARAT VIII/1, ANCOL, PADEMANGAN, JAKARTA UTARA, 14430') VALUES ('PT.PRIMAFOOD INTERNATIONAL, JL. ANCOL BARAT VIII/1, ANCOL, PADEMANGAN, JAKARTA UTARA, 14430')
ON CONFLICT (name) DO NOTHING; ON CONFLICT (name) DO NOTHING;
+244 -244
View File
@@ -1,244 +1,244 @@
-- Migration: 005_create_sku_master -- Migration: 005_create_sku_master
-- Description: Create sku_master table and seed the initial SKU entries -- Description: Create sku_master table and seed the initial SKU entries
CREATE TABLE IF NOT EXISTS sku_master ( CREATE TABLE IF NOT EXISTS sku_master (
id SERIAL PRIMARY KEY, id SERIAL PRIMARY KEY,
no_sku VARCHAR(255) UNIQUE NOT NULL, no_sku VARCHAR(255) UNIQUE NOT NULL,
nama_item VARCHAR(255) NOT NULL, nama_item VARCHAR(255) NOT NULL,
created_at TIMESTAMP NOT NULL DEFAULT NOW() created_at TIMESTAMP NOT NULL DEFAULT NOW()
); );
INSERT INTO sku_master (no_sku, nama_item) VALUES INSERT INTO sku_master (no_sku, nama_item) VALUES
('11048006', 'BEBEK PARTING-NEW(*)'), ('11048006', 'BEBEK PARTING-NEW(*)'),
('11110059', 'CEKER BERKUKU FROZEN PACK 1 KG(*)'), ('11110059', 'CEKER BERKUKU FROZEN PACK 1 KG(*)'),
('11110074', 'CEKER 1 KG FROZEN (20 PAC/KARUNG)(*)'), ('11110074', 'CEKER 1 KG FROZEN (20 PAC/KARUNG)(*)'),
('11140051', 'AMPELA FROZEN PACK 1 KG(*)'), ('11140051', 'AMPELA FROZEN PACK 1 KG(*)'),
('11140062', 'AMPELA 1 KG FROZEN (20 PAC/KARUNG)(*)'), ('11140062', 'AMPELA 1 KG FROZEN (20 PAC/KARUNG)(*)'),
('11148002', 'AMPELA BEBEK FROZEN 1 KG/PACK (NEW)(*)'), ('11148002', 'AMPELA BEBEK FROZEN 1 KG/PACK (NEW)(*)'),
('11150052', 'HATI FROZEN PACK 1 KG(*)'), ('11150052', 'HATI FROZEN PACK 1 KG(*)'),
('11150055', 'JANTUNG FROZEN PACK 1 KG(*)'), ('11150055', 'JANTUNG FROZEN PACK 1 KG(*)'),
('11150064', 'HATI 1 KG FROZEN (20 PAC/KARUNG)(*)'), ('11150064', 'HATI 1 KG FROZEN (20 PAC/KARUNG)(*)'),
('11150065', 'JANTUNG 1 KG FROZEN (20 PAC/KARUNG)(*)'), ('11150065', 'JANTUNG 1 KG FROZEN (20 PAC/KARUNG)(*)'),
('11310012', 'AYAM SIZE 0 (0.6-0.7)KG(*)'), ('11310012', 'AYAM SIZE 0 (0.6-0.7)KG(*)'),
('11310013', 'AYAM SIZE1 FROZEN (0.75-0.8) KG(*)'), ('11310013', 'AYAM SIZE1 FROZEN (0.75-0.8) KG(*)'),
('11310014', 'AYAM SIZE 2 FROZEN (0.8-0.9)KG(*)'), ('11310014', 'AYAM SIZE 2 FROZEN (0.8-0.9)KG(*)'),
('11310016', 'AYAM SIZE Z PR FROZ(*)'), ('11310016', 'AYAM SIZE Z PR FROZ(*)'),
('11310017', 'AYAM SIZE 0 PR FROZEN(*)'), ('11310017', 'AYAM SIZE 0 PR FROZEN(*)'),
('11310018', 'AYAM SIZE 1 PR FROZEN(*)'), ('11310018', 'AYAM SIZE 1 PR FROZEN(*)'),
('11310019', 'AYAM SIZE 2 PR FROZEN(*)'), ('11310019', 'AYAM SIZE 2 PR FROZEN(*)'),
('11310021', 'AYAM SIZE BESAR (B) FROZ (1-1.1)KG/PC(*)'), ('11310021', 'AYAM SIZE BESAR (B) FROZ (1-1.1)KG/PC(*)'),
('11310022', 'AYAM SIZE A PR (0.9-1)KG/PC(*)'), ('11310022', 'AYAM SIZE A PR (0.9-1)KG/PC(*)'),
('11310024', 'AYAM SIZE A FROZEN (0.9-1)KG/PC(*)'), ('11310024', 'AYAM SIZE A FROZEN (0.9-1)KG/PC(*)'),
('11310025', 'AYAM SIZE SUPER (C) FROZ(1.1-1.2)KG/ PC(*)'), ('11310025', 'AYAM SIZE SUPER (C) FROZ(1.1-1.2)KG/ PC(*)'),
('11310026', 'AYAM SIZE JUMBO (D) FROZ (1.2- 1.3)KG/PC(*)'), ('11310026', 'AYAM SIZE JUMBO (D) FROZ (1.2- 1.3)KG/PC(*)'),
('11318301', 'BEBEK MUDA-BD1(1.0-1.1 KG)-NEW(*)'), ('11318301', 'BEBEK MUDA-BD1(1.0-1.1 KG)-NEW(*)'),
('11318306', 'CP DUCK PEKING 1.5-1.6 KG/PC(*)'), ('11318306', 'CP DUCK PEKING 1.5-1.6 KG/PC(*)'),
('11318308', 'BEBEK PEKING SPR BD5(1.7 -1.8 )Kg-NEW(*)'), ('11318308', 'BEBEK PEKING SPR BD5(1.7 -1.8 )Kg-NEW(*)'),
('11410043', 'PARTING 10 SIZE D FRESH BENSU 1.25 KG/PAC(*)'), ('11410043', 'PARTING 10 SIZE D FRESH BENSU 1.25 KG/PAC(*)'),
('11420055', 'PARTING 12 ALL SIZE FROZ/PAC(*)'), ('11420055', 'PARTING 12 ALL SIZE FROZ/PAC(*)'),
('11600053', 'BONELESS LEG FROZEN 1 KG(*)'), ('11600053', 'BONELESS LEG FROZEN 1 KG(*)'),
('11620056', 'SBL (FILLET PAHA) 1 KG(*)'), ('11620056', 'SBL (FILLET PAHA) 1 KG(*)'),
('11640053', 'PAHA UTUH (1 KG)(*)'), ('11640053', 'PAHA UTUH (1 KG)(*)'),
('11650053', 'PAHA ATAS 1 KG(*)'), ('11650053', 'PAHA ATAS 1 KG(*)'),
('11660050', 'PAHA BAWAH (1 KG)(*)'), ('11660050', 'PAHA BAWAH (1 KG)(*)'),
('11690053', 'SBB (FILLET DADA )1 KG(*)'), ('11690053', 'SBB (FILLET DADA )1 KG(*)'),
('11690081', 'SBB JUMBO FZ (2.0 - 2.2 KG/PAC)(*)'), ('11690081', 'SBB JUMBO FZ (2.0 - 2.2 KG/PAC)(*)'),
('11710051', 'DADA UTUH (1 KG)(*)'), ('11710051', 'DADA UTUH (1 KG)(*)'),
('11720055', 'FULL WING FROZ PACK 1 KG(*)'), ('11720055', 'FULL WING FROZ PACK 1 KG(*)'),
('11730050', 'MIDDLE WING FROZ PACK 1 KG(*)'), ('11730050', 'MIDDLE WING FROZ PACK 1 KG(*)'),
('11750050', 'FILLET MITRA 1 KG(*)'), ('11750050', 'FILLET MITRA 1 KG(*)'),
('11818300', 'CP-BEBEK GORENG 400GR/PAC'), ('11818300', 'CP-BEBEK GORENG 400GR/PAC'),
('11840002', 'AYAM JANTAN BKKL SZ 0 (600-700) GR/PC(*)'), ('11840002', 'AYAM JANTAN BKKL SZ 0 (600-700) GR/PC(*)'),
('11959937', 'SATE AYAM FRESHMART 360 GR (PAC)'), ('11959937', 'SATE AYAM FRESHMART 360 GR (PAC)'),
('12010111', 'FIESTA CRISPY BUBBLE 400 GR/PAC'), ('12010111', 'FIESTA CRISPY BUBBLE 400 GR/PAC'),
('12010112', 'FIESTA CHICKEN NUGGET 400 GR/PAC'), ('12010112', 'FIESTA CHICKEN NUGGET 400 GR/PAC'),
('12010113', 'FIESTA CHICKEN NUGGET 200 GR/PAC'), ('12010113', 'FIESTA CHICKEN NUGGET 200 GR/PAC'),
('12010115', 'FIESTA NUGGET ZOO 400 GR/PAC'), ('12010115', 'FIESTA NUGGET ZOO 400 GR/PAC'),
('12010116', 'FIESTA NUGGET DINO 400 GR/PAC'), ('12010116', 'FIESTA NUGGET DINO 400 GR/PAC'),
('12010117', 'FIESTA NUGGET HAPPY STAR 400 GR/PAC'), ('12010117', 'FIESTA NUGGET HAPPY STAR 400 GR/PAC'),
('12010119', 'FIESTA NUGGET CHEESE 123 400 GR/PAC'), ('12010119', 'FIESTA NUGGET CHEESE 123 400 GR/PAC'),
('12010121', 'FIESTA NUGGET PIZZABC 400 GR/PAC'), ('12010121', 'FIESTA NUGGET PIZZABC 400 GR/PAC'),
('12010122', 'FIESTA CHEESY LOVER 400 GR/PAC'), ('12010122', 'FIESTA CHEESY LOVER 400 GR/PAC'),
('12010123', 'FIESTA GARLIC CHEESE 400 GR/PAC'), ('12010123', 'FIESTA GARLIC CHEESE 400 GR/PAC'),
('12010124', 'FIESTA CHEESY CHIC W/BROCCOLI 400 GR/PAC'), ('12010124', 'FIESTA CHEESY CHIC W/BROCCOLI 400 GR/PAC'),
('12010127', 'FIESTA SPICY NUGGET 400 GR/PAC'), ('12010127', 'FIESTA SPICY NUGGET 400 GR/PAC'),
('12010128', 'FIESTA VOLCANO CHEESE 400 GR/PAC'), ('12010128', 'FIESTA VOLCANO CHEESE 400 GR/PAC'),
('12010129', 'FIESTA CHEESY BOMBS CHICKEN NUGGET 400 GR'), ('12010129', 'FIESTA CHEESY BOMBS CHICKEN NUGGET 400 GR'),
('12010402', 'GOLDEN FIESTA NUGGET W/PINEAPPLE SAUCE 500 GR'), ('12010402', 'GOLDEN FIESTA NUGGET W/PINEAPPLE SAUCE 500 GR'),
('12010509', 'CHAMP CRUNCHY NUGGET 450 GR/PAC'), ('12010509', 'CHAMP CRUNCHY NUGGET 450 GR/PAC'),
('12010510', 'CHAMP NUGGET AYAM 225 GR/PAC'), ('12010510', 'CHAMP NUGGET AYAM 225 GR/PAC'),
('12010511', 'CHAMP NUGGET AYAM 450 GR/PAC'), ('12010511', 'CHAMP NUGGET AYAM 450 GR/PAC'),
('12010512', 'CHAMP NUGGET AYAM 900 GR/PAC'), ('12010512', 'CHAMP NUGGET AYAM 900 GR/PAC'),
('12010513', 'CHAMP NUGGET ABC KOMBINASI 225 GR/PAC'), ('12010513', 'CHAMP NUGGET ABC KOMBINASI 225 GR/PAC'),
('12010514', 'CHAMP NUGGET ABC KOMBINASI 450 GR/PAC'), ('12010514', 'CHAMP NUGGET ABC KOMBINASI 450 GR/PAC'),
('12010515', 'CHAMP KOIN KOMBINASI 450 GR/PAC'), ('12010515', 'CHAMP KOIN KOMBINASI 450 GR/PAC'),
('12010516', 'CHAMP KOIN KOMBINASI 200 GR/PAC'), ('12010516', 'CHAMP KOIN KOMBINASI 200 GR/PAC'),
('12010517', 'CHAMP NUGGET STICK 225 GR/PAC'), ('12010517', 'CHAMP NUGGET STICK 225 GR/PAC'),
('12010518', 'CHAMP NUGGET STICK 450 GR/PAC'), ('12010518', 'CHAMP NUGGET STICK 450 GR/PAC'),
('12010519', 'CHAMP NUGGET STICK 900 GR/PAC'), ('12010519', 'CHAMP NUGGET STICK 900 GR/PAC'),
('12010520', 'CHAMP CHICKEN NUGGET BENTUK 123 450 GR/PAC'), ('12010520', 'CHAMP CHICKEN NUGGET BENTUK 123 450 GR/PAC'),
('12010521', 'CHAMP NUGGET HOTZZ LEVEL 5 450 GR/PAC'), ('12010521', 'CHAMP NUGGET HOTZZ LEVEL 5 450 GR/PAC'),
('12010606', 'CHAMP CRUNCHY NUGGET 225 GR/PAC'), ('12010606', 'CHAMP CRUNCHY NUGGET 225 GR/PAC'),
('12010707', 'CHAMP MITRA NUGGET COIN 200 GR (NEW)'), ('12010707', 'CHAMP MITRA NUGGET COIN 200 GR (NEW)'),
('12010801', 'OKEY NUGGET 500GR'), ('12010801', 'OKEY NUGGET 500GR'),
('12012201', 'ASIMO NUGGET KOMBINASI 500 GR/PAC'), ('12012201', 'ASIMO NUGGET KOMBINASI 500 GR/PAC'),
('12012202', 'ASIMO NUGGET KOMBINASI 1 KG/PAC'), ('12012202', 'ASIMO NUGGET KOMBINASI 1 KG/PAC'),
('12012203', 'ASIMO NUGGET KOMBINASI 250 GR/PAC'), ('12012203', 'ASIMO NUGGET KOMBINASI 250 GR/PAC'),
('12012501', 'AKUMO CHICKEN NAGET 250 GR'), ('12012501', 'AKUMO CHICKEN NAGET 250 GR'),
('12012502', 'AKUMO CHICKEN NUGGET 500 GR'), ('12012502', 'AKUMO CHICKEN NUGGET 500 GR'),
('12012503', 'AKUMO CHICKEN NUGGET 1000 GR'), ('12012503', 'AKUMO CHICKEN NUGGET 1000 GR'),
('12012504', 'AKUMO COIN 200 GR/PAC'), ('12012504', 'AKUMO COIN 200 GR/PAC'),
('12012505', 'AKUMO KOIN 400 GR/PAC'), ('12012505', 'AKUMO KOIN 400 GR/PAC'),
('12020102', 'FIESTA SPICY WING 400 GR/PAC'), ('12020102', 'FIESTA SPICY WING 400 GR/PAC'),
('12020401', 'GOLDEN FIESTA SP WING 500 GR'), ('12020401', 'GOLDEN FIESTA SP WING 500 GR'),
('12030101', 'FIESTA STIKIE 400 GR/PAC'), ('12030101', 'FIESTA STIKIE 400 GR/PAC'),
('12030102', 'FIESTA STIKIE 200 GR/PAC'), ('12030102', 'FIESTA STIKIE 200 GR/PAC'),
('12030403', 'GOLDEN FIESTA STIKIE W/ SWEET CHILLI SAUCE 500GR'), ('12030403', 'GOLDEN FIESTA STIKIE W/ SWEET CHILLI SAUCE 500GR'),
('12030801', 'OKEY STICK 1000 GR'), ('12030801', 'OKEY STICK 1000 GR'),
('12030802', 'OKEY STICK 500GR'), ('12030802', 'OKEY STICK 500GR'),
('12032201', 'ASIMO STICK KOMBINASI 500 GR/PAC'), ('12032201', 'ASIMO STICK KOMBINASI 500 GR/PAC'),
('12032202', 'ASIMO STICK KOMBINASI 1000 GR/PAC'), ('12032202', 'ASIMO STICK KOMBINASI 1000 GR/PAC'),
('12032203', 'ASIMO STIK KOMBINASI 250 GR/PAC'), ('12032203', 'ASIMO STIK KOMBINASI 250 GR/PAC'),
('12032501', 'AKUMO CHICKEN STICK 250 GR'), ('12032501', 'AKUMO CHICKEN STICK 250 GR'),
('12032502', 'AKUMO CHICKEN STIK 500 GR'), ('12032502', 'AKUMO CHICKEN STIK 500 GR'),
('12032503', 'AKUMO CHICKEN STICK 1000 GR'), ('12032503', 'AKUMO CHICKEN STICK 1000 GR'),
('12040101', 'FIESTA SCHNITZEL 400 GR/PAC'), ('12040101', 'FIESTA SCHNITZEL 400 GR/PAC'),
('12040102', 'FIESTA CRISPY BUBBLE KATSU 400 GR/PAC'), ('12040102', 'FIESTA CRISPY BUBBLE KATSU 400 GR/PAC'),
('12040404', 'GOLDEN FIESTA CORDON BLEU BBQ SAUCE 500 GR'), ('12040404', 'GOLDEN FIESTA CORDON BLEU BBQ SAUCE 500 GR'),
('12040406', 'GOLDEN FIESTA KATSU W/CHEESE SAUCE 500 GR/PAC'), ('12040406', 'GOLDEN FIESTA KATSU W/CHEESE SAUCE 500 GR/PAC'),
('12050103', 'FIESTA FRIED CHICKEN 400 GR/PAC'), ('12050103', 'FIESTA FRIED CHICKEN 400 GR/PAC'),
('12050104', 'FIESTA HOT & CRISPY FRIED CHICKEN 400 GR/PAC'), ('12050104', 'FIESTA HOT & CRISPY FRIED CHICKEN 400 GR/PAC'),
('12050401', 'GOLDEN FIESTA CRISPY WING W/SP GLAZ SC 500 GR/PAC'), ('12050401', 'GOLDEN FIESTA CRISPY WING W/SP GLAZ SC 500 GR/PAC'),
('12060103', 'FIESTA KARAGE 200 GR/PAC'), ('12060103', 'FIESTA KARAGE 200 GR/PAC'),
('12060104', 'FIESTA KARAGE 400 GR/PAC'), ('12060104', 'FIESTA KARAGE 400 GR/PAC'),
('12060105', 'FIESTA SPICY KARAGE 400 GR/PAC'), ('12060105', 'FIESTA SPICY KARAGE 400 GR/PAC'),
('12060402', 'GOLDEN FIESTA KARAGE CHILI SAUCE 500GR'), ('12060402', 'GOLDEN FIESTA KARAGE CHILI SAUCE 500GR'),
('12070101', 'FIESTA POK-POK 400 GR/PAC (NEW)'), ('12070101', 'FIESTA POK-POK 400 GR/PAC (NEW)'),
('12080101', 'FIESTA SPICY CHICK 400 GR/PAC'), ('12080101', 'FIESTA SPICY CHICK 400 GR/PAC'),
('12130102', 'FIESTA CRISPY BURGER 360 GR (NEW)'), ('12130102', 'FIESTA CRISPY BURGER 360 GR (NEW)'),
('12130504', 'CHAMP BURGER 315 GR (NEW)'), ('12130504', 'CHAMP BURGER 315 GR (NEW)'),
('12140105', 'FIESTA CHICK TOFU 400 GR/PAC'), ('12140105', 'FIESTA CHICK TOFU 400 GR/PAC'),
('12150201', 'FIESTA DS CRISPY CRUNCH 300 GR/PAC'), ('12150201', 'FIESTA DS CRISPY CRUNCH 300 GR/PAC'),
('12150501', 'CHAMP CRUNCHY HOTZZ 300 GR/PAC'), ('12150501', 'CHAMP CRUNCHY HOTZZ 300 GR/PAC'),
('12190103', 'FIESTA DELISTRIPE 400 GR/PAC'), ('12190103', 'FIESTA DELISTRIPE 400 GR/PAC'),
('12240102', 'FIESTA CHEESY ITALIAN R/BITES 400 GR/PAC'), ('12240102', 'FIESTA CHEESY ITALIAN R/BITES 400 GR/PAC'),
('12240103', 'FIESTA YAKINIKU R/BITES 400 GR/PAC'), ('12240103', 'FIESTA YAKINIKU R/BITES 400 GR/PAC'),
('13010101', 'FIESTA CHICK SSG 300 GR'), ('13010101', 'FIESTA CHICK SSG 300 GR'),
('13010102', 'FIESTA CHICK SSG 500 GR'), ('13010102', 'FIESTA CHICK SSG 500 GR'),
('13010103', 'FIESTA CHICK SSG 200 GR/PAC'), ('13010103', 'FIESTA CHICK SSG 200 GR/PAC'),
('13010111', 'FIESTA SOSIS BRATWURST 300 GR'), ('13010111', 'FIESTA SOSIS BRATWURST 300 GR'),
('13010112', 'FIESTA CHEESE SSG 300 GR'), ('13010112', 'FIESTA CHEESE SSG 300 GR'),
('13010113', 'FIESTA SOSIS CURRYWURST 300 GR'), ('13010113', 'FIESTA SOSIS CURRYWURST 300 GR'),
('13010114', 'FIESTA SSG BOCKWURST 300GR'), ('13010114', 'FIESTA SSG BOCKWURST 300GR'),
('13010115', 'FIESTA SSG WIENER 300GR'), ('13010115', 'FIESTA SSG WIENER 300GR'),
('13010116', 'FIESTA SSG ORIGINAL 300 GR'), ('13010116', 'FIESTA SSG ORIGINAL 300 GR'),
('13010117', 'FIESTA SSG FRANKFURTER 300GR'), ('13010117', 'FIESTA SSG FRANKFURTER 300GR'),
('13010118', 'FIESTA RTG SSG 65 GR/PAC'), ('13010118', 'FIESTA RTG SSG 65 GR/PAC'),
('13010119', 'FIESTA RTG C/SPICY KOREAN 60 GR/PAC'), ('13010119', 'FIESTA RTG C/SPICY KOREAN 60 GR/PAC'),
('13010120', 'FIESTA RTG C/CHEESY MELTS 65 GR/PAC'), ('13010120', 'FIESTA RTG C/CHEESY MELTS 65 GR/PAC'),
('13010122', 'FIESTA RTG SAUSAGE WITH HOT LAVA 60G'), ('13010122', 'FIESTA RTG SAUSAGE WITH HOT LAVA 60G'),
('13010123', 'FIESTA RTG SAUSAGE WITH CHEESE LAVA 60G'), ('13010123', 'FIESTA RTG SAUSAGE WITH CHEESE LAVA 60G'),
('13010124', 'FIESTA RTG SAUSAGE HICKORY SAUCE 60GR'), ('13010124', 'FIESTA RTG SAUSAGE HICKORY SAUCE 60GR'),
('13010125', 'FIESTA RTG SAUSAGE WITH MENTAI LAVA 60GR'), ('13010125', 'FIESTA RTG SAUSAGE WITH MENTAI LAVA 60GR'),
('13010510', 'CHAMP CHICK SSG 75 GR'), ('13010510', 'CHAMP CHICK SSG 75 GR'),
('13010513', 'CHAMP CHICK SSG 375 GR'), ('13010513', 'CHAMP CHICK SSG 375 GR'),
('13010514', 'CHAMP CHICK SSG 1000 GR'), ('13010514', 'CHAMP CHICK SSG 1000 GR'),
('13010518', 'CHAMP SSG JUMBO BAKAR 500 GR/PAC'), ('13010518', 'CHAMP SSG JUMBO BAKAR 500 GR/PAC'),
('13010519', 'CHAMP SSG BAKAR MINI 500 GR/PAC-INACT'), ('13010519', 'CHAMP SSG BAKAR MINI 500 GR/PAC-INACT'),
('13010521', 'CHAMP CHICK SSG 150 GR/PAC (NEW)'), ('13010521', 'CHAMP CHICK SSG 150 GR/PAC (NEW)'),
('13010523', 'CHAMP CHICK SSG AYAM MADU 300 GR/PAC'), ('13010523', 'CHAMP CHICK SSG AYAM MADU 300 GR/PAC'),
('13010524', 'CHAMP SSG JUMBO BAKAR 500 GR/PAC (NEW)'), ('13010524', 'CHAMP SSG JUMBO BAKAR 500 GR/PAC (NEW)'),
('13010525', 'CHAMP SSG BAKAR MINI 500 GR/PAC (NEW)'), ('13010525', 'CHAMP SSG BAKAR MINI 500 GR/PAC (NEW)'),
('13010809', 'OKEY CHICK SSG 500GR-INACT'), ('13010809', 'OKEY CHICK SSG 500GR-INACT'),
('13010815', 'OKEY SSG BAKAR JUMBO 500 GR/PAC (NEW)'), ('13010815', 'OKEY SSG BAKAR JUMBO 500 GR/PAC (NEW)'),
('13010816', 'OKEY SSG BAKAR MINI 500 GR/PAC (NEW)'), ('13010816', 'OKEY SSG BAKAR MINI 500 GR/PAC (NEW)'),
('13010817', 'OKEY SSG CHICK KOMBINASI 500 GR (EXTRA1)'), ('13010817', 'OKEY SSG CHICK KOMBINASI 500 GR (EXTRA1)'),
('13010818', 'OKEY SSG CHICK KOMBINASI 1 KG (EXTR$A2)'), ('13010818', 'OKEY SSG CHICK KOMBINASI 1 KG (EXTR$A2)'),
('13012205', 'ASIMO SOSIS AYAM KOMBINASI 375 GR (PAC)'), ('13012205', 'ASIMO SOSIS AYAM KOMBINASI 375 GR (PAC)'),
('13012206', 'ASIMO SOSIS AYAM KOMBINASI 500 GR'), ('13012206', 'ASIMO SOSIS AYAM KOMBINASI 500 GR'),
('13012207', 'ASIMO SOSIS AYAM KOMBINASI 750 GR'), ('13012207', 'ASIMO SOSIS AYAM KOMBINASI 750 GR'),
('13012208', 'ASIMO SOSIS AYAM KOMBINASI 1000 GR'), ('13012208', 'ASIMO SOSIS AYAM KOMBINASI 1000 GR'),
('13012209', 'ASIMO SSG CHICK KOMBINASI 500 GR (EXTRA1)'), ('13012209', 'ASIMO SSG CHICK KOMBINASI 500 GR (EXTRA1)'),
('13012210', 'ASIMO SSG CHICK KOMBINASI 1 KG (EXTR$A2)'), ('13012210', 'ASIMO SSG CHICK KOMBINASI 1 KG (EXTR$A2)'),
('13030101', 'FIESTA CHICK MEAT BALL 300 GR'), ('13030101', 'FIESTA CHICK MEAT BALL 300 GR'),
('13030102', 'FIESTA CHICK MEATBALL 500 GR'), ('13030102', 'FIESTA CHICK MEATBALL 500 GR'),
('13030501', 'CHAMP CHICK MEATBALL 200 GR'), ('13030501', 'CHAMP CHICK MEATBALL 200 GR'),
('13030502', 'CHAMP CHICK MEATBALL 500 GR'), ('13030502', 'CHAMP CHICK MEATBALL 500 GR'),
('13050101', 'FIESTA SCB 250 GR'), ('13050101', 'FIESTA SCB 250 GR'),
('13050105', 'FIESTA CHICKEN SLICE 300 GR'), ('13050105', 'FIESTA CHICKEN SLICE 300 GR'),
('13050106', 'FIESTA BEEF SLICE 300 GR'), ('13050106', 'FIESTA BEEF SLICE 300 GR'),
('13070501', 'CHAMP BEEF SSG SERBAGUNA 150 GR'), ('13070501', 'CHAMP BEEF SSG SERBAGUNA 150 GR'),
('13070502', 'CHAMP BEEF SSG SERBAGUNA 375GR'), ('13070502', 'CHAMP BEEF SSG SERBAGUNA 375GR'),
('13070505', 'CHAMP BEEF SSG GORENG 375 GR'), ('13070505', 'CHAMP BEEF SSG GORENG 375 GR'),
('13070506', 'CHAMP FRANKFURTER SSG 375GR'), ('13070506', 'CHAMP FRANKFURTER SSG 375GR'),
('13100512', 'CHAMP CHICK SSG S/SANTAP ORIG 546GR (CAN)'), ('13100512', 'CHAMP CHICK SSG S/SANTAP ORIG 546GR (CAN)'),
('13110504', 'CHAMP BEEF BALL 500GR'), ('13110504', 'CHAMP BEEF BALL 500GR'),
('13170510', 'CHAMP BEEF BBQ SSG S/SANTAP 546GR (CAN)'), ('13170510', 'CHAMP BEEF BBQ SSG S/SANTAP 546GR (CAN)'),
('15010101', 'FIESTA SHOESTRING 500 GR'), ('15010101', 'FIESTA SHOESTRING 500 GR'),
('15010102', 'FIESTA SHOESTRING 1000 GR'), ('15010102', 'FIESTA SHOESTRING 1000 GR'),
('15010107', 'FIESTA FRENCH F SHOESTRING INSTITUSI 2KG'), ('15010107', 'FIESTA FRENCH F SHOESTRING INSTITUSI 2KG'),
('15020101', 'FIESTA STRAIGHT CUT 500 GR'), ('15020101', 'FIESTA STRAIGHT CUT 500 GR'),
('15020102', 'FIESTA STRAIGHT CUT 1000 GR'), ('15020102', 'FIESTA STRAIGHT CUT 1000 GR'),
('15030101', 'FIESTA CRINKLE CUT 500 GR'), ('15030101', 'FIESTA CRINKLE CUT 500 GR'),
('15030102', 'FIESTA CRINKLE CUT 1000 GR'), ('15030102', 'FIESTA CRINKLE CUT 1000 GR'),
('15040101', 'FIESTA BATTER COATED 500 GR'), ('15040101', 'FIESTA BATTER COATED 500 GR'),
('15040102', 'FIESTA BATTER COATED 1000 GR'), ('15040102', 'FIESTA BATTER COATED 1000 GR'),
('16060103', 'FIESTA CHICK SIOMAY 900GR'), ('16060103', 'FIESTA CHICK SIOMAY 900GR'),
('16060113', 'FIESTA CHICK SIOMAY 180GR (NEW)'), ('16060113', 'FIESTA CHICK SIOMAY 180GR (NEW)'),
('16060114', 'FIESTA GYOZA 180 GR (NEW)'), ('16060114', 'FIESTA GYOZA 180 GR (NEW)'),
('16060119', 'FIESTA RTG SIOMAY 54 GR/PAC'), ('16060119', 'FIESTA RTG SIOMAY 54 GR/PAC'),
('16060120', 'FIESTA KEECHO 400 GR/PAC'), ('16060120', 'FIESTA KEECHO 400 GR/PAC'),
('16060121', 'FIESTA CHICKEN TOFU 400 GR/PAC (NEW)'), ('16060121', 'FIESTA CHICKEN TOFU 400 GR/PAC (NEW)'),
('16060503', 'CHAMP CHICK&FISH SIOMAY 180 GR (NEW)'), ('16060503', 'CHAMP CHICK&FISH SIOMAY 180 GR (NEW)'),
('17200109', 'FIESTA RTS C/TERIYAKI 300GR/PAC'), ('17200109', 'FIESTA RTS C/TERIYAKI 300GR/PAC'),
('17200110', 'FIESTA RTS C/RENDANG 300GR/PAC'), ('17200110', 'FIESTA RTS C/RENDANG 300GR/PAC'),
('17200111', 'FIESTA RTS C/W RUJAK SC 300GR/PAC'), ('17200111', 'FIESTA RTS C/W RUJAK SC 300GR/PAC'),
('17200112', 'FIESTA RTS C/W S/MATAH 300GR/PAC'), ('17200112', 'FIESTA RTS C/W S/MATAH 300GR/PAC'),
('17210106', 'FIESTA RTS B/YAKINIKU 300GR/PAC'), ('17210106', 'FIESTA RTS B/YAKINIKU 300GR/PAC'),
('17210107', 'FIESTA RTS B/RENDANG 300GR/PAC'), ('17210107', 'FIESTA RTS B/RENDANG 300GR/PAC'),
('17210108', 'FIESTA RTS B/BLACKPEPPER 300GR/PAC'), ('17210108', 'FIESTA RTS B/BLACKPEPPER 300GR/PAC'),
('17210109', 'FIESTA RTS B/BULGOGI 300GR/PAC'), ('17210109', 'FIESTA RTS B/BULGOGI 300GR/PAC'),
('18050102', 'FIESTA RTG BAKSO KEJU 60 GR/PAC'), ('18050102', 'FIESTA RTG BAKSO KEJU 60 GR/PAC'),
('18050103', 'FIESTA RTG BAKSO BAKAR BBQ 60 GR/PAC'), ('18050103', 'FIESTA RTG BAKSO BAKAR BBQ 60 GR/PAC'),
('18050104', 'FIESTA RTG BEEF BALL WITH MENTAI LAVA 55GR'), ('18050104', 'FIESTA RTG BEEF BALL WITH MENTAI LAVA 55GR'),
('18050105', 'FIESTA RTG BEEF BALL WITH CHEESE LAVA 55GR'), ('18050105', 'FIESTA RTG BEEF BALL WITH CHEESE LAVA 55GR'),
('20010101', 'FIESTA CRISPY CRUMBS 200 GR'), ('20010101', 'FIESTA CRISPY CRUMBS 200 GR'),
('20010102', 'FIESTA TP ROTI PUTIH 200 GR'), ('20010102', 'FIESTA TP ROTI PUTIH 200 GR'),
('20040101', 'FIESTA RAMEN BEKU 570 GR/PAC'), ('20040101', 'FIESTA RAMEN BEKU 570 GR/PAC'),
('20120102', 'FIESTA T/B AYAM GORENG 80 GR'), ('20120102', 'FIESTA T/B AYAM GORENG 80 GR'),
('20120105', 'FIESTA T/B SERBAGUNA 80 GR'), ('20120105', 'FIESTA T/B SERBAGUNA 80 GR'),
('20120106', 'FIESTA T/B KREMES 80 GR'), ('20120106', 'FIESTA T/B KREMES 80 GR'),
('20120115', 'FIESTA RACIK AYAM GORENG 20 GR/PAC'), ('20120115', 'FIESTA RACIK AYAM GORENG 20 GR/PAC'),
('20120116', 'FIESTA RACIK NASI GORENG 20 GR/PAC'), ('20120116', 'FIESTA RACIK NASI GORENG 20 GR/PAC'),
('21000123', 'FIESTA RICE W/GEPREK CHICKEN 320GR/PAC'), ('21000123', 'FIESTA RICE W/GEPREK CHICKEN 320GR/PAC'),
('21000124', 'FIESTA RICE W/CHICK RUJAK 320 GR/PAC'), ('21000124', 'FIESTA RICE W/CHICK RUJAK 320 GR/PAC'),
('21000125', 'FIESTA RICE W/KOREAN BBQ CHICK 320 GR/PAC'), ('21000125', 'FIESTA RICE W/KOREAN BBQ CHICK 320 GR/PAC'),
('21000126', 'NEW FIESTA CHICK RENDANG W RICE 320GR (PAC)'), ('21000126', 'NEW FIESTA CHICK RENDANG W RICE 320GR (PAC)'),
('21000127', 'NEW FIESTA CHICK TERIYAKI W RICE 320GR (PAC)'), ('21000127', 'NEW FIESTA CHICK TERIYAKI W RICE 320GR (PAC)'),
('21000128', 'NEW FIESTA CHICK TANDORI W RICE 320GR (PAC)'), ('21000128', 'NEW FIESTA CHICK TANDORI W RICE 320GR (PAC)'),
('21000129', 'NEW FIESTA RICE W KARAGE&SSS 320GR (PAC)'), ('21000129', 'NEW FIESTA RICE W KARAGE&SSS 320GR (PAC)'),
('21000130', 'NEW FIESTA RICE W/C CHEESE BULDAK 320GR (PAC)'), ('21000130', 'NEW FIESTA RICE W/C CHEESE BULDAK 320GR (PAC)'),
('21000131', 'NEW FIESTA RICE W CHIC CURRY 320GR (PAC)'), ('21000131', 'NEW FIESTA RICE W CHIC CURRY 320GR (PAC)'),
('21000132', 'NEW FIESTA RICE W CHICK DONBURI 320GR (PAC)'), ('21000132', 'NEW FIESTA RICE W CHICK DONBURI 320GR (PAC)'),
('21000133', 'NEW FIESTA RICE W CHICK SATAY 320GR (PAC)'), ('21000133', 'NEW FIESTA RICE W CHICK SATAY 320GR (PAC)'),
('21000134', 'NEW FIESTA COCONUT RICE W SPICY CHICK 320GR (PAC)'), ('21000134', 'NEW FIESTA COCONUT RICE W SPICY CHICK 320GR (PAC)'),
('21000135', 'NEW FIESTA RICE W POPBITES S/MATAH 320GR (PAC)'), ('21000135', 'NEW FIESTA RICE W POPBITES S/MATAH 320GR (PAC)'),
('21000136', 'NEW FIESTA TUMERIC W POPBITES 320GR (PAC)'), ('21000136', 'NEW FIESTA TUMERIC W POPBITES 320GR (PAC)'),
('21000137', 'FIESTA HAINAMESE CHICKEN RICE 320GR (PAC)'), ('21000137', 'FIESTA HAINAMESE CHICKEN RICE 320GR (PAC)'),
('21010101', 'FIESTA TRUFFLE GYUDON 320 GR/PAC'), ('21010101', 'FIESTA TRUFFLE GYUDON 320 GR/PAC'),
('21010102', 'NEW FIESTA BEEF YAKINIKU W RICE 320GR (PAC)'), ('21010102', 'NEW FIESTA BEEF YAKINIKU W RICE 320GR (PAC)'),
('21010103', 'NEW FIESTA BEEF BULGOGI W RICE 320GR (PAC)'), ('21010103', 'NEW FIESTA BEEF BULGOGI W RICE 320GR (PAC)'),
('21010104', 'NEW FIESTA BEEF RENDANG W RICE 320GR (PAC)'), ('21010104', 'NEW FIESTA BEEF RENDANG W RICE 320GR (PAC)'),
('21010105', 'NEW FIESTA RICE W BEEF BLACKPEPPER 320GR (PAC)'), ('21010105', 'NEW FIESTA RICE W BEEF BLACKPEPPER 320GR (PAC)'),
('21200107', 'NEW FIESTA SPAGHETTI CARBONARA 300GR (PAC)'), ('21200107', 'NEW FIESTA SPAGHETTI CARBONARA 300GR (PAC)'),
('21200108', 'NEW FIESTA SPAGHETTI CHIC BOLOGNESE 320GR (PAC)'), ('21200108', 'NEW FIESTA SPAGHETTI CHIC BOLOGNESE 320GR (PAC)'),
('21200109', 'NEW FIESTA ITALIAN MEATBALL SPAGHETTI 320GR (PAC)'), ('21200109', 'NEW FIESTA ITALIAN MEATBALL SPAGHETTI 320GR (PAC)'),
('21310103', 'NEW FIESTA SCB&S/SSG FRIED RICE 320GR (PAC)'), ('21310103', 'NEW FIESTA SCB&S/SSG FRIED RICE 320GR (PAC)'),
('21500101', 'FIESTA CHICK SSG & C. BALL PIZZA 230GR/PAC'), ('21500101', 'FIESTA CHICK SSG & C. BALL PIZZA 230GR/PAC'),
('21500102', 'FIESTA CHEESY BEEF PIZZA 230GR/PAC'), ('21500102', 'FIESTA CHEESY BEEF PIZZA 230GR/PAC'),
('91000012', 'PHOTOCARD RTG'), ('91000012', 'PHOTOCARD RTG'),
('1188002W', 'PAHA ATAS 25-30 G FZ (*)'), ('1188002W', 'PAHA ATAS 25-30 G FZ (*)'),
('1195008A', 'RTC CHICKEN KALASAN 400 GR (PAC)'), ('1195008A', 'RTC CHICKEN KALASAN 400 GR (PAC)'),
('1195008E', 'RTC CHICKEN TERIYAKI 400 GR (PAC)'), ('1195008E', 'RTC CHICKEN TERIYAKI 400 GR (PAC)'),
('1195008X', 'RTC CHICKEN SPICY 400 GR (PAC)') ('1195008X', 'RTC CHICKEN SPICY 400 GR (PAC)')
ON CONFLICT (no_sku) DO NOTHING; ON CONFLICT (no_sku) DO NOTHING;
+9 -9
View File
@@ -1,9 +1,9 @@
services: services:
db: db:
ports: ports:
- "5432:5432" - "5432:5432"
pipeline-api: pipeline-api:
ports: ports:
- "8090:8090" - "8090:8090"
environment: environment:
- VLLM_SERVER_URL=http://paddleocr-vllm-server:8118/v1 - VLLM_SERVER_URL=http://paddleocr-vllm-server:8118/v1
+146 -146
View File
@@ -1,146 +1,146 @@
name: ai-ocr-pfm-2026 name: ai-ocr-pfm-2026
services: services:
nginx: nginx:
image: nginx:alpine image: nginx:alpine
container_name: paddleocr-nginx container_name: paddleocr-nginx
ports: ports:
- "${APP_PORT:-8000}:80" - "${APP_PORT:-8000}:80"
volumes: volumes:
- ./nginx.conf:/etc/nginx/nginx.conf:ro - ./nginx.conf:/etc/nginx/nginx.conf:ro
depends_on: depends_on:
- vllm-server - vllm-server
- pipeline-api - pipeline-api
- gradio-ui - gradio-ui
- pfm-web-app - pfm-web-app
restart: unless-stopped restart: unless-stopped
vllm-server: vllm-server:
build: build:
context: . context: .
target: vllm-server target: vllm-server
container_name: paddleocr-vllm-server container_name: paddleocr-vllm-server
image: paddleocr-vllm-server:latest image: paddleocr-vllm-server:latest
environment: environment:
- GENAI_HOST=0.0.0.0 - GENAI_HOST=0.0.0.0
- GENAI_PORT=8118 - GENAI_PORT=8118
- GENAI_MODEL=${GENAI_MODEL:-PaddleOCR-VL-1.6-0.9B} - GENAI_MODEL=${GENAI_MODEL:-PaddleOCR-VL-1.6-0.9B}
- GENAI_BACKEND=${GENAI_BACKEND:-vllm} - GENAI_BACKEND=${GENAI_BACKEND:-vllm}
- VLLM_CONFIG=${VLLM_CONFIG:-config/vllm_config.yaml} - VLLM_CONFIG=${VLLM_CONFIG:-config/vllm_config.yaml}
- CUDA_VISIBLE_DEVICES=${CUDA_VISIBLE_DEVICES:-0} - CUDA_VISIBLE_DEVICES=${CUDA_VISIBLE_DEVICES:-0}
- PADDLE_PDX_DISABLE_MODEL_SOURCE_CHECK=True - PADDLE_PDX_DISABLE_MODEL_SOURCE_CHECK=True
# No exposed ports; internal only # No exposed ports; internal only
deploy: deploy:
resources: resources:
reservations: reservations:
devices: devices:
- driver: nvidia - driver: nvidia
count: all count: all
capabilities: [gpu] capabilities: [gpu]
volumes: volumes:
- hf_cache:/root/.cache/huggingface - hf_cache:/root/.cache/huggingface
- paddle_cache:/root/.paddleocr - paddle_cache:/root/.paddleocr
- paddlex_cache:/root/.paddlex - paddlex_cache:/root/.paddlex
- ./config:/app/config - ./config:/app/config
- ./.env:/app/.env:ro - ./.env:/app/.env:ro
restart: unless-stopped restart: unless-stopped
pipeline-api: pipeline-api:
build: build:
context: . context: .
target: pipeline-api target: pipeline-api
container_name: paddleocr-pipeline-api-v10 container_name: paddleocr-pipeline-api-v10
image: paddleocr-pipeline-api:latest image: paddleocr-pipeline-api:latest
environment: environment:
- PIPELINE_CONFIG=${PIPELINE_CONFIG:-config/pipeline_config_vllm.yaml} - PIPELINE_CONFIG=${PIPELINE_CONFIG:-config/pipeline_config_vllm.yaml}
- PIPELINE_HOST=0.0.0.0 - PIPELINE_HOST=0.0.0.0
- PIPELINE_PORT=8090 - PIPELINE_PORT=8090
- PIPELINE_DEVICE=gpu:0 - PIPELINE_DEVICE=gpu:0
- CUDA_VISIBLE_DEVICES=${CUDA_VISIBLE_DEVICES:-0} - CUDA_VISIBLE_DEVICES=${CUDA_VISIBLE_DEVICES:-0}
- VLLM_SERVER_URL=http://paddleocr-pfm-web-app:3000/api/vllm-proxy/v1 - VLLM_SERVER_URL=http://paddleocr-pfm-web-app:3000/api/vllm-proxy/v1
- PADDLE_PDX_DISABLE_MODEL_SOURCE_CHECK=True - PADDLE_PDX_DISABLE_MODEL_SOURCE_CHECK=True
# No exposed ports; internal only # No exposed ports; internal only
deploy: deploy:
resources: resources:
reservations: reservations:
devices: devices:
- driver: nvidia - driver: nvidia
count: all count: all
capabilities: [gpu] capabilities: [gpu]
volumes: volumes:
- paddle_cache:/root/.paddleocr - paddle_cache:/root/.paddleocr
- paddlex_cache:/root/.paddlex - paddlex_cache:/root/.paddlex
- ./config:/app/config - ./config:/app/config
- ./pfm-web-app/public/produk-pfm/models:/app/pfm-web-app/public/produk-pfm/models:ro - ./pfm-web-app/public/produk-pfm/models:/app/pfm-web-app/public/produk-pfm/models:ro
- ./.env:/app/.env:ro - ./.env:/app/.env:ro
depends_on: depends_on:
- vllm-server - vllm-server
restart: unless-stopped restart: unless-stopped
gradio-ui: gradio-ui:
build: build:
context: . context: .
target: gradio-ui target: gradio-ui
container_name: paddleocr-gradio-ui container_name: paddleocr-gradio-ui
image: paddleocr-gradio-ui:latest image: paddleocr-gradio-ui:latest
environment: environment:
- API_URL=http://paddleocr-pipeline-api-v10:8090/layout-parsing - API_URL=http://paddleocr-pipeline-api-v10:8090/layout-parsing
- GRADIO_PORT=7870 - GRADIO_PORT=7870
- GRADIO_MCP_SERVER=True - GRADIO_MCP_SERVER=True
# No exposed ports; internal only # No exposed ports; internal only
depends_on: depends_on:
- pipeline-api - pipeline-api
restart: unless-stopped restart: unless-stopped
pfm-web-app: pfm-web-app:
build: build:
context: . context: .
target: pfm-web-app target: pfm-web-app
container_name: paddleocr-pfm-web-app container_name: paddleocr-pfm-web-app
image: paddleocr-pfm-web-app:latest image: paddleocr-pfm-web-app:latest
command: npm run dev command: npm run dev
environment: environment:
- PIPELINE_URL=http://paddleocr-pipeline-api-v10:8090/layout-parsing - PIPELINE_URL=http://paddleocr-pipeline-api-v10:8090/layout-parsing
- NODE_ENV=development - NODE_ENV=development
- PGHOST=paddleocr-db - PGHOST=paddleocr-db
- PGPORT=5432 - PGPORT=5432
- PGUSER=postgres - PGUSER=postgres
- PGPASSWORD=postgres - PGPASSWORD=postgres
- PGDATABASE=dopfm - PGDATABASE=dopfm
- JWT_SECRET=${JWT_SECRET:-dev-only-insecure-secret-change-me} - JWT_SECRET=${JWT_SECRET:-dev-only-insecure-secret-change-me}
pid: "host" pid: "host"
volumes: volumes:
- ./pfm-web-app:/app - ./pfm-web-app:/app
- /app/node_modules - /app/node_modules
- /app/.next - /app/.next
- ./uploads:/uploads - ./uploads:/uploads
- /var/run/docker.sock:/var/run/docker.sock - /var/run/docker.sock:/var/run/docker.sock
# No exposed ports; internal only # No exposed ports; internal only
depends_on: depends_on:
- pipeline-api - pipeline-api
- db - db
extra_hosts: extra_hosts:
- "host.docker.internal:host-gateway" - "host.docker.internal:host-gateway"
restart: unless-stopped restart: unless-stopped
db: db:
image: postgres:15-alpine image: postgres:15-alpine
container_name: paddleocr-db container_name: paddleocr-db
environment: environment:
- POSTGRES_USER=postgres - POSTGRES_USER=postgres
- POSTGRES_PASSWORD=postgres - POSTGRES_PASSWORD=postgres
- POSTGRES_DB=dopfm - POSTGRES_DB=dopfm
volumes: volumes:
- pgdata:/var/lib/postgresql/data - pgdata:/var/lib/postgresql/data
- ./db/migrations:/docker-entrypoint-initdb.d:ro - ./db/migrations:/docker-entrypoint-initdb.d:ro
restart: unless-stopped restart: unless-stopped
volumes: volumes:
hf_cache: hf_cache:
name: paddleocr_hf_cache name: paddleocr_hf_cache
paddle_cache: paddle_cache:
name: paddleocr_paddle_cache name: paddleocr_paddle_cache
paddlex_cache: paddlex_cache:
name: paddleocr_paddlex_cache name: paddleocr_paddlex_cache
pgdata: pgdata:
name: paddleocr_pgdata name: paddleocr_pgdata
+95 -95
View File
@@ -1,95 +1,95 @@
# Feature List (backend) # Feature List (backend)
Structured log of shipped backend features, updated by the `n`/`next` workflow (see Structured log of shipped backend features, updated by the `n`/`next` workflow (see
[AGENTS.md](../AGENTS.md) Part B) whenever a task in [AGENTS.md](../AGENTS.md) Part B) whenever a task in
[plans/next-enhancements.md](../plans/next-enhancements.md) is marked `[DONE]`. [plans/next-enhancements.md](../plans/next-enhancements.md) is marked `[DONE]`.
Split out 2026-07-08 from root `docs/feature-list.md`'s backend sections — this file Split out 2026-07-08 from root `docs/feature-list.md`'s backend sections — this file
is the sole home for backend feature history going forward. is the sole home for backend feature history going forward.
## Format ## Format
``` ```
## <Section / Module Name> ## <Section / Module Name>
- **<task number>** <feature description> — shipped <date> - **<task number>** <feature description> — shipped <date>
``` ```
--- ---
## Existing Features (pre-kit) ## Existing Features (pre-kit)
Backfilled 2026-07-08 during adoption of this kit — these predate the `e`/`n` Backfilled 2026-07-08 during adoption of this kit — these predate the `e`/`n`
workflow and have no task numbers; see `git log` for real dates/history. workflow and have no task numbers; see `git log` for real dates/history.
### Backend — Next.js API Gateway ### Backend — Next.js API Gateway
- Upload/parse/documents CRUD routes, GPU status endpoint, vLLM proxy, manual-label review tool. - Upload/parse/documents CRUD routes, GPU status endpoint, vLLM proxy, manual-label review tool.
### Backend — OCR Pipeline & Accuracy ### Backend — OCR Pipeline & Accuracy
- PaddleOCR + vLLM classification pipeline with DB layout caching, table column-shift correction, date normalization, and an accuracy regression harness (`pfm-web-app/scripts/accuracy-check.mts`) — **95.10% overall as of 2026-07-08** (target 95% met; see task 2.3 below for the investigation and `CLAUDE.md`). - PaddleOCR + vLLM classification pipeline with DB layout caching, table column-shift correction, date normalization, and an accuracy regression harness (`pfm-web-app/scripts/accuracy-check.mts`) — **95.10% overall as of 2026-07-08** (target 95% met; see task 2.3 below for the investigation and `CLAUDE.md`).
### Backend — Postgres Data Layer ### Backend — Postgres Data Layer
- Schema/init in `pfm-web-app/src/db/init.ts`, served via the canonical root `docker-compose.yml` stack. - Schema/init in `pfm-web-app/src/db/init.ts`, served via the canonical root `docker-compose.yml` stack.
- **3.3** Added a standard `INDEX` on `documents(file_hash)` in `db/init.ts` to accelerate the upload deduplication queries without strictly enforcing uniqueness across different stores. Correspondingly updated the dedup query in `api/v1/documents/upload/route.ts` to scope duplicate detection by `kode_toko`. This fixes a conflict where one store could be incorrectly linked to another store's duplicate receipt image — shipped 2026-07-08. - **3.3** Added a standard `INDEX` on `documents(file_hash)` in `db/init.ts` to accelerate the upload deduplication queries without strictly enforcing uniqueness across different stores. Correspondingly updated the dedup query in `api/v1/documents/upload/route.ts` to scope duplicate detection by `kode_toko`. This fixes a conflict where one store could be incorrectly linked to another store's duplicate receipt image — shipped 2026-07-08.
### DevOps — Docker & Dev Tunnel ### DevOps — Docker & Dev Tunnel
- **4.1 Docker Compose Policy Documented**: Formalized the execution policy in `README.md` and `CLAUDE.md`, explicitly requiring the use of the `docker-compose.demo.yml` override (production build) for all client demonstrations and field testing to bypass the Next.js dev server bottleneck — shipped 2026-07-08. - **4.1 Docker Compose Policy Documented**: Formalized the execution policy in `README.md` and `CLAUDE.md`, explicitly requiring the use of the `docker-compose.demo.yml` override (production build) for all client demonstrations and field testing to bypass the Next.js dev server bottleneck — shipped 2026-07-08.
- **Docker Compose Dependency Gates**: Added strict Docker `healthcheck` gates (`Task 4.2`) blocking the `pfm-web-app` (Next.js) from starting until PostgreSQL and the VLLM models are initialized and fully healthy. - **Docker Compose Dependency Gates**: Added strict Docker `healthcheck` gates (`Task 4.2`) blocking the `pfm-web-app` (Next.js) from starting until PostgreSQL and the VLLM models are initialized and fully healthy.
- **Secure Tunnel Ingress**: Restructured `nginx.conf` and `start-dev-tunnel.ps1` (`Tasks 4.3, 4.5`) to expose a dedicated, restricted port (`8001`) that exclusively routes to `/api/v1/*`. This perfectly secures the development UI (`/scan-pfm`) and legacy routes from public exposure. - **Secure Tunnel Ingress**: Restructured `nginx.conf` and `start-dev-tunnel.ps1` (`Tasks 4.3, 4.5`) to expose a dedicated, restricted port (`8001`) that exclusively routes to `/api/v1/*`. This perfectly secures the development UI (`/scan-pfm`) and legacy routes from public exposure.
- **Dead Config Pruning**: Stripped deprecated and redundant proxy blocks from the Nginx edge router (`Task 4.4`). - **Dead Config Pruning**: Stripped deprecated and redundant proxy blocks from the Nginx edge router (`Task 4.4`).
*(New features shipped via `n`/`next` go below, organized the same way, with task numbers.)* *(New features shipped via `n`/`next` go below, organized the same way, with task numbers.)*
## Backend — Next.js API Gateway ## Backend — Next.js API Gateway
- **1.4** Enforced real 401 auth on `/api/v1/documents/*` (list, PUT-by-id, upload) — the actual production API surface, already fully supported by the Flutter client (real login + `Authorization: Bearer` on every request). Previously none of these three routes rejected a missing/invalid token; upload only optionally read it. Added the pre-existing `getAccountFromAuthHeader()` helper (`utils/auth.ts`) + a 401 guard to all three; `OPTIONS` (CORS preflight) untouched. The original task 1.3 (auth on the *classic* routes) was cancelled instead — those routes are dev-only web UI surface with no login flow, going away in production. Verified via `curl`: 401 with no token, success with a real token from `/api/v1/auth/login` — shipped 2026-07-08. - **1.4** Enforced real 401 auth on `/api/v1/documents/*` (list, PUT-by-id, upload) — the actual production API surface, already fully supported by the Flutter client (real login + `Authorization: Bearer` on every request). Previously none of these three routes rejected a missing/invalid token; upload only optionally read it. Added the pre-existing `getAccountFromAuthHeader()` helper (`utils/auth.ts`) + a 401 guard to all three; `OPTIONS` (CORS preflight) untouched. The original task 1.3 (auth on the *classic* routes) was cancelled instead — those routes are dev-only web UI surface with no login flow, going away in production. Verified via `curl`: 401 with no token, success with a real token from `/api/v1/auth/login` — shipped 2026-07-08.
- **1.5** Implemented per-store data scoping on `/api/v1/documents/*`. Added `kode_toko` column to `documents` table via `db/init.ts` migration. The upload route now binds `kode_toko` to documents upon creation. `GET /api/v1/documents` and `PUT /api/v1/documents/:id` enforce ownership checks (`kode_toko` matching) for `store` role accounts, while `admin` retains global access including legacy unassigned documents — shipped 2026-07-08. - **1.5** Implemented per-store data scoping on `/api/v1/documents/*`. Added `kode_toko` column to `documents` table via `db/init.ts` migration. The upload route now binds `kode_toko` to documents upon creation. `GET /api/v1/documents` and `PUT /api/v1/documents/:id` enforce ownership checks (`kode_toko` matching) for `store` role accounts, while `admin` retains global access including legacy unassigned documents — shipped 2026-07-08.
- **1.6** `GET /api/v1/health` Endpoint: Unauthenticated health probe verifying both PostgreSQL connectivity and Pipeline API HTTP reachability. Returns `HTTP 503` if any core dependency is down — shipped 2026-07-08. - **1.6** `GET /api/v1/health` Endpoint: Unauthenticated health probe verifying both PostgreSQL connectivity and Pipeline API HTTP reachability. Returns `HTTP 503` if any core dependency is down — shipped 2026-07-08.
- **Ad-hoc** Connected `/scan-pfm` page with `/api/scan-pfm` route and enabled auto-trigger scanning on custom file upload, sample selection, thumbnail change, and canvas rotation. Supported both `image` and `image_base64` payload keys — shipped 2026-07-09. - **Ad-hoc** Connected `/scan-pfm` page with `/api/scan-pfm` route and enabled auto-trigger scanning on custom file upload, sample selection, thumbnail change, and canvas rotation. Supported both `image` and `image_base64` payload keys — shipped 2026-07-09.
- **9.1** Added `GET /api/v1/documents/:id` (same 401/403 scoping as `PUT`), returning a single document — including still-unparsed rows — with a new `parseStatus: "pending"|"done"|"failed"` field, so the Flutter poller can move off scanning the entire list every 2s. Added `scan_mode`/`parse_error` columns to `documents` (`db/init.ts`, migrated via `ALTER TABLE ... ADD COLUMN IF NOT EXISTS` for already-running DBs). `scan_mode` is now persisted on upload (`v1/documents/upload/route.ts`) and on the classic `/api/parse` route's upserts (`COALESCE`, same pattern as `kode_toko`), and surfaced as `docType` on every GET response (`utils/document-mapper.ts`, a new shared helper extracted from the list route's inline mapping so list/by-id/dedup all agree) — falling back to the legacy `order_untuk == "PRODUCT SCAN"` sentinel for pre-existing rows with no `scan_mode`. `parse_error` is now recorded when the upload route's *internal* call to `/api/parse` itself fails to complete (network error or the 210s abort firing) — previously this was silently swallowed and the document stayed `parsed=false` forever with no signal, burning the client's full 260s timeout; `/api/parse`'s own existing pipeline-error fallback (`parsed=true` + "Not Found" placeholder) was already fine and is unchanged. Also fixed the dedup branch (a repeat upload of an already-seen file) to return the original document's real current state via the same mapper instead of a hardcoded empty stub. Verified via `docker compose up -d --build` + `curl`: schema migration applied cleanly to the live DB (confirmed via `psql`), DO and Product uploads both correctly persist `scan_mode` and surface it as `docType`, a dedup retry returns real header/items instead of an empty stub, `GET /:id` returns 401 (no token) / 403 (wrong store) / 404 (nonexistent id) / 200 (admin or owning store), and the list endpoint's existing filter/scoping is unchanged — shipped 2026-07-10. - **9.1** Added `GET /api/v1/documents/:id` (same 401/403 scoping as `PUT`), returning a single document — including still-unparsed rows — with a new `parseStatus: "pending"|"done"|"failed"` field, so the Flutter poller can move off scanning the entire list every 2s. Added `scan_mode`/`parse_error` columns to `documents` (`db/init.ts`, migrated via `ALTER TABLE ... ADD COLUMN IF NOT EXISTS` for already-running DBs). `scan_mode` is now persisted on upload (`v1/documents/upload/route.ts`) and on the classic `/api/parse` route's upserts (`COALESCE`, same pattern as `kode_toko`), and surfaced as `docType` on every GET response (`utils/document-mapper.ts`, a new shared helper extracted from the list route's inline mapping so list/by-id/dedup all agree) — falling back to the legacy `order_untuk == "PRODUCT SCAN"` sentinel for pre-existing rows with no `scan_mode`. `parse_error` is now recorded when the upload route's *internal* call to `/api/parse` itself fails to complete (network error or the 210s abort firing) — previously this was silently swallowed and the document stayed `parsed=false` forever with no signal, burning the client's full 260s timeout; `/api/parse`'s own existing pipeline-error fallback (`parsed=true` + "Not Found" placeholder) was already fine and is unchanged. Also fixed the dedup branch (a repeat upload of an already-seen file) to return the original document's real current state via the same mapper instead of a hardcoded empty stub. Verified via `docker compose up -d --build` + `curl`: schema migration applied cleanly to the live DB (confirmed via `psql`), DO and Product uploads both correctly persist `scan_mode` and surface it as `docType`, a dedup retry returns real header/items instead of an empty stub, `GET /:id` returns 401 (no token) / 403 (wrong store) / 404 (nonexistent id) / 200 (admin or owning store), and the list endpoint's existing filter/scoping is unchanged — shipped 2026-07-10.
- **9.3** Added authenticated `POST /api/v1/scan-product`, the v1 equivalent of the classic dev-only `/api/scan-pfm` (unauthenticated, and unreachable off-LAN since task 4.5 restricted the public tunnel to `/api/v1/*`). Extracted the shared classify-and-match logic (Python classifier call + Levenshtein SKU matching against `sku_master`, top-5 scoring) out of `api/scan-pfm/route.ts` into a new `utils/product-scan.ts` (`classifyAndMatchProduct`, plus a `ClassifierError` class that preserves forwarding the classifier's own HTTP status instead of collapsing every failure to 500) so the classic route and the new v1 route share one implementation instead of duplicating it — the classic route's response shape, auth-free behavior, and desktop-only layout-parsing visualization are otherwise unchanged. The new route accepts **either** multipart (`image`/`file` field, matching the v1 upload route's convention) or a JSON `{image_base64}` body, is open to any authenticated account (not admin-gated, since this is what the mobile app itself calls), and wraps the result in the standard `{status, data}` envelope with `classification`, `ocr` (including `extracted_expired_date`), and `possibleMatches`. Verified via `curl` against the live stack with a real product photo: multipart upload and JSON-body variants both return identical, correct top-5 matches; no-token request returns 401; the classic `/api/scan-pfm` route's response (including `layoutParsingResult`) is unchanged post-refactor — shipped 2026-07-10. - **9.3** Added authenticated `POST /api/v1/scan-product`, the v1 equivalent of the classic dev-only `/api/scan-pfm` (unauthenticated, and unreachable off-LAN since task 4.5 restricted the public tunnel to `/api/v1/*`). Extracted the shared classify-and-match logic (Python classifier call + Levenshtein SKU matching against `sku_master`, top-5 scoring) out of `api/scan-pfm/route.ts` into a new `utils/product-scan.ts` (`classifyAndMatchProduct`, plus a `ClassifierError` class that preserves forwarding the classifier's own HTTP status instead of collapsing every failure to 500) so the classic route and the new v1 route share one implementation instead of duplicating it — the classic route's response shape, auth-free behavior, and desktop-only layout-parsing visualization are otherwise unchanged. The new route accepts **either** multipart (`image`/`file` field, matching the v1 upload route's convention) or a JSON `{image_base64}` body, is open to any authenticated account (not admin-gated, since this is what the mobile app itself calls), and wraps the result in the standard `{status, data}` envelope with `classification`, `ocr` (including `extracted_expired_date`), and `possibleMatches`. Verified via `curl` against the live stack with a real product photo: multipart upload and JSON-body variants both return identical, correct top-5 matches; no-token request returns 401; the classic `/api/scan-pfm` route's response (including `layoutParsingResult`) is unchanged post-refactor — shipped 2026-07-10.
- **9.2** Relaxed `GET /api/v1/master/skus` (`master/skus/route.ts`) so any authenticated account can read the SKU master list, not just `admin` — the Flutter product editor needs this and previously had to string-hack its base URL to call the unauthenticated classic `GET /api/skus`, which task 4.5 had already removed from the public tunnel, breaking product scans off-LAN. Changed the guard from a combined `!account || role !== 'admin'` check (403 for both "no token" and "wrong role") to `!account` (correct 401) followed by an unconditional pass-through for any valid account; `POST` (SKU creation) is untouched, still admin-only, per the user's explicit choice between the two options this task flagged as undecided. No response-shape change. Verified via `curl` against the live stack with a real non-admin (`store` role) account's token: `GET` → 200 with real data; no token → 401 (was incorrectly 403 before this fix); the same non-admin token against `POST` → still 403; admin `GET` → still 200. Along the way, hit and resolved a dev-loop issue: the container had the edited file on disk but Turbopack's file watcher wasn't detecting the change over the Windows bind mount, requiring `docker restart paddleocr-pfm-web-app` to pick it up — noted in case it recurs for future edits. With 9.1-9.3 all shipped, Flutter root task 7.1 (moving the product editor onto the v1 surface) is now fully unblocked — shipped 2026-07-10. - **9.2** Relaxed `GET /api/v1/master/skus` (`master/skus/route.ts`) so any authenticated account can read the SKU master list, not just `admin` — the Flutter product editor needs this and previously had to string-hack its base URL to call the unauthenticated classic `GET /api/skus`, which task 4.5 had already removed from the public tunnel, breaking product scans off-LAN. Changed the guard from a combined `!account || role !== 'admin'` check (403 for both "no token" and "wrong role") to `!account` (correct 401) followed by an unconditional pass-through for any valid account; `POST` (SKU creation) is untouched, still admin-only, per the user's explicit choice between the two options this task flagged as undecided. No response-shape change. Verified via `curl` against the live stack with a real non-admin (`store` role) account's token: `GET` → 200 with real data; no token → 401 (was incorrectly 403 before this fix); the same non-admin token against `POST` → still 403; admin `GET` → still 200. Along the way, hit and resolved a dev-loop issue: the container had the edited file on disk but Turbopack's file watcher wasn't detecting the change over the Windows bind mount, requiring `docker restart paddleocr-pfm-web-app` to pick it up — noted in case it recurs for future edits. With 9.1-9.3 all shipped, Flutter root task 7.1 (moving the product editor onto the v1 surface) is now fully unblocked — shipped 2026-07-10.
## Backend — OCR Pipeline & Accuracy ## Backend — OCR Pipeline & Accuracy
- **2.1** Built the Product/SKU scan classifier's model artifacts: `models/dinov2_index.pkl` (118/118 reference photos indexed across 16 SKU classes) and `models/produk-pfm-classifier-26n-100e-2026-07-08.pt` (+ `.onnx` export) — a YOLO classifier fine-tuned 100 epochs, 83.3% top-1 / 90% top-5 validation accuracy on the current (thin, 2-16 photos/class) dataset. Built via a one-off `docker run` from a freshly-rebuilt `pipeline-api` image (bare-metal training isn't viable on Windows — `paddlepaddle-gpu`'s wheel index is Linux-only). `pipeline-api` restarted and confirmed loading both models from logs. Also fixed `scripts/install-pipeline.sh`, which was missing `ultralytics`/`torch` — shipped 2026-07-08. - **2.1** Built the Product/SKU scan classifier's model artifacts: `models/dinov2_index.pkl` (118/118 reference photos indexed across 16 SKU classes) and `models/produk-pfm-classifier-26n-100e-2026-07-08.pt` (+ `.onnx` export) — a YOLO classifier fine-tuned 100 epochs, 83.3% top-1 / 90% top-5 validation accuracy on the current (thin, 2-16 photos/class) dataset. Built via a one-off `docker run` from a freshly-rebuilt `pipeline-api` image (bare-metal training isn't viable on Windows — `paddlepaddle-gpu`'s wheel index is Linux-only). `pipeline-api` restarted and confirmed loading both models from logs. Also fixed `scripts/install-pipeline.sh`, which was missing `ultralytics`/`torch` — shipped 2026-07-08.
- **2.1 (verification pass)** Ran a full browser walkthrough of `/scan-pfm` (classification, top-5, OCR expiry extraction + crop, SKU-master matching, Visual/Spotting Grid, Raw Response — all confirmed working with real data). Found and fixed a real bug: "Save Ground Truth" was returning success but silently writing into the `pfm-web-app` container's ephemeral filesystem instead of the host, because `/sources` wasn't a bind-mounted path in root `docker-compose.yml`. Added `./backend/sources:/sources` to the `pfm-web-app` service, recovered an orphaned entry via `docker cp`, and re-verified the save now persists to `backend/sources/product_manual_labels.json` on the host (confirmed the DO-flow's `manual_labels.json` save was fixed by the same change too) — shipped 2026-07-08. - **2.1 (verification pass)** Ran a full browser walkthrough of `/scan-pfm` (classification, top-5, OCR expiry extraction + crop, SKU-master matching, Visual/Spotting Grid, Raw Response — all confirmed working with real data). Found and fixed a real bug: "Save Ground Truth" was returning success but silently writing into the `pfm-web-app` container's ephemeral filesystem instead of the host, because `/sources` wasn't a bind-mounted path in root `docker-compose.yml`. Added `./backend/sources:/sources` to the `pfm-web-app` service, recovered an orphaned entry via `docker cp`, and re-verified the save now persists to `backend/sources/product_manual_labels.json` on the host (confirmed the DO-flow's `manual_labels.json` save was fixed by the same change too) — shipped 2026-07-08.
- **2.3** Ran the accuracy regression harness and discovered `sources/accuracy_report.md` was badly stale (claimed 75.04%; real current baseline is **95.10% overall, already at/above the 95% target** — added a staleness banner to that file). Root-caused every remaining mismatch by pulling raw OCR text from Postgres (`documents.layout_parsing_result`): the worst field, `plat` (67.6%), is almost entirely the license-plate region being classified as an image/seal by the layout model rather than OCR'd as text — not fixable in `parser.ts`. Found and fixed one genuine parser logic bug along the way: the "global pattern scanning fallback" could duplicate an already-correctly-extracted `noDO` value into a still-missing `noSO` field; fixed by excluding already-assigned values from that fallback's candidate pool (`pfm-web-app/src/utils/parser.ts`). Doesn't change the aggregate score (a wrong value and "Not Found" score the same) but stops a fabricated-looking wrong number from silently reaching the database. All 48 parser unit tests still pass — shipped 2026-07-08. - **2.3** Ran the accuracy regression harness and discovered `sources/accuracy_report.md` was badly stale (claimed 75.04%; real current baseline is **95.10% overall, already at/above the 95% target** — added a staleness banner to that file). Root-caused every remaining mismatch by pulling raw OCR text from Postgres (`documents.layout_parsing_result`): the worst field, `plat` (67.6%), is almost entirely the license-plate region being classified as an image/seal by the layout model rather than OCR'd as text — not fixable in `parser.ts`. Found and fixed one genuine parser logic bug along the way: the "global pattern scanning fallback" could duplicate an already-correctly-extracted `noDO` value into a still-missing `noSO` field; fixed by excluding already-assigned values from that fallback's candidate pool (`pfm-web-app/src/utils/parser.ts`). Doesn't change the aggregate score (a wrong value and "Not Found" score the same) but stops a fabricated-looking wrong number from silently reaching the database. All 48 parser unit tests still pass — shipped 2026-07-08.
- **Ad-hoc** Built custom expiry-date-based auto-rotation algorithm in Python classifier server (`classify_ocr_server.py`). The algorithm calculates the slant angle of the Expiry Date / Batch text line bounding box, automatically rotates the image to make it horizontal, and re-runs YOLO classification + PaddleOCR for maximum accuracy. Enhanced SKU matching database lookup to prioritize exact SKU matches with a score of 1.0, pinning them as the Best Match — shipped 2026-07-09. - **Ad-hoc** Built custom expiry-date-based auto-rotation algorithm in Python classifier server (`classify_ocr_server.py`). The algorithm calculates the slant angle of the Expiry Date / Batch text line bounding box, automatically rotates the image to make it horizontal, and re-runs YOLO classification + PaddleOCR for maximum accuracy. Enhanced SKU matching database lookup to prioritize exact SKU matches with a score of 1.0, pinning them as the Best Match — shipped 2026-07-09.
- **2.5** Retrained the Product/SKU scan classifier's model artifacts against the full current dataset, which had grown to 81 SKU classes / 2,493 photos (up from the original 16 classes / 118 photos the deployed model dated 2026-07-08 was actually trained on — the other 65 classes had photos but no trained weights). Rebuilt `models/dinov2_index.pkl` (now 2,493/2,493 photos indexed) and retrained the YOLO classifier 100 epochs on an RTX 2060 (real elapsed time 54m21s), publishing `models/produk-pfm-classifier-26n-100e-2026-07-14.pt`/`.onnx` at **85.8% top-1 / 94.4% top-5** validation accuracy across all 81 classes (up from 83.3%/90% on the old 16-class model). Along the way, fixed a real train/val split bug in `train_classifier.py`: `split_dataset()` previously shuffled and split individual image files, letting an augmented copy (`photo_aug_2.jpeg`) land in validation while its near-duplicate source stayed in training — inflating val accuracy with memorization instead of measuring generalization; now groups by source photo (stripping `_aug_N`) before shuffling and splitting 80/20. Verified via `docker compose up -d pipeline-api` + `docker logs`: "DINOv2 index loaded with 2493 reference images", "Using classifier weights: .../produk-pfm-classifier-26n-100e-2026-07-14.pt", "YOLO model loaded successfully" — the live service is confirmed serving the new 81-class model, not assumed from the newest-file-by-date fallback logic. Remaining gap toward the program's ±230-SKU target is dataset growth, not a pipeline limitation — shipped 2026-07-14. - **2.5** Retrained the Product/SKU scan classifier's model artifacts against the full current dataset, which had grown to 81 SKU classes / 2,493 photos (up from the original 16 classes / 118 photos the deployed model dated 2026-07-08 was actually trained on — the other 65 classes had photos but no trained weights). Rebuilt `models/dinov2_index.pkl` (now 2,493/2,493 photos indexed) and retrained the YOLO classifier 100 epochs on an RTX 2060 (real elapsed time 54m21s), publishing `models/produk-pfm-classifier-26n-100e-2026-07-14.pt`/`.onnx` at **85.8% top-1 / 94.4% top-5** validation accuracy across all 81 classes (up from 83.3%/90% on the old 16-class model). Along the way, fixed a real train/val split bug in `train_classifier.py`: `split_dataset()` previously shuffled and split individual image files, letting an augmented copy (`photo_aug_2.jpeg`) land in validation while its near-duplicate source stayed in training — inflating val accuracy with memorization instead of measuring generalization; now groups by source photo (stripping `_aug_N`) before shuffling and splitting 80/20. Verified via `docker compose up -d pipeline-api` + `docker logs`: "DINOv2 index loaded with 2493 reference images", "Using classifier weights: .../produk-pfm-classifier-26n-100e-2026-07-14.pt", "YOLO model loaded successfully" — the live service is confirmed serving the new 81-class model, not assumed from the newest-file-by-date fallback logic. Remaining gap toward the program's ±230-SKU target is dataset growth, not a pipeline limitation — shipped 2026-07-14.
## Backend — Postgres Data Layer ## Backend — Postgres Data Layer
- **3.1** Wrapped the `ocr_items` delete-then-reinsert in `/api/parse` and `/api/v1/documents/[id]` PUT inside a DB transaction (`withTransaction` helper, `pfm-web-app/src/db/index.ts`) — a mid-loop insert failure now rolls back to the previous item set instead of leaving a document with a correct header but partial/missing items — shipped 2026-07-08. (Renumbered from root's `7.1` when this file split from root `docs/feature-list.md`.) - **3.1** Wrapped the `ocr_items` delete-then-reinsert in `/api/parse` and `/api/v1/documents/[id]` PUT inside a DB transaction (`withTransaction` helper, `pfm-web-app/src/db/index.ts`) — a mid-loop insert failure now rolls back to the previous item set instead of leaving a document with a correct header but partial/missing items — shipped 2026-07-08. (Renumbered from root's `7.1` when this file split from root `docs/feature-list.md`.)
- **3.2** Hashed `accounts.password` with `bcryptjs` (pure-JS, no native compile step — the `pfm-web-app` Docker image has no build toolchain). `db/init.ts` hashes the seed and idempotently migrates any pre-existing plaintext rows on every startup; `api/v1/auth/login/route.ts` now compares with `bcrypt.compareSync` and cleanly rejects missing credentials with a 401 instead of risking a raw-query edge case. Verified via `psql` (hash format) and `curl` (correct login succeeds, wrong/missing password returns 401) — shipped 2026-07-08. - **3.2** Hashed `accounts.password` with `bcryptjs` (pure-JS, no native compile step — the `pfm-web-app` Docker image has no build toolchain). `db/init.ts` hashes the seed and idempotently migrates any pre-existing plaintext rows on every startup; `api/v1/auth/login/route.ts` now compares with `bcrypt.compareSync` and cleanly rejects missing credentials with a 401 instead of risking a raw-query edge case. Verified via `psql` (hash format) and `curl` (correct login succeeds, wrong/missing password returns 401) — shipped 2026-07-08.
## Docs & Workflow Integrity ## Docs & Workflow Integrity
- **5.1** Fixed stale doc claims in `SKILLS.md` (accuracy baseline pointer) and `CLAUDE.md` (Flutter auth claim and API base URL fallback) — shipped 2026-07-08. - **5.1** Fixed stale doc claims in `SKILLS.md` (accuracy baseline pointer) and `CLAUDE.md` (Flutter auth claim and API base URL fallback) — shipped 2026-07-08.
- **5.2** Refactored `plans/next-enhancements.md` to archive verbose `[DONE]` and `[CANCELLED]` task bodies into one-line stubs. Reduced the file size significantly, strictly enforcing the 256-line threshold rule for maintainability — shipped 2026-07-08. - **5.2** Refactored `plans/next-enhancements.md` to archive verbose `[DONE]` and `[CANCELLED]` task bodies into one-line stubs. Reduced the file size significantly, strictly enforcing the 256-line threshold rule for maintainability — shipped 2026-07-08.
- **5.3** Amended `AGENTS.md` completion checklist with a doc-sync step to ensure architecture changes are synced back to documentation — shipped 2026-07-08. - **5.3** Amended `AGENTS.md` completion checklist with a doc-sync step to ensure architecture changes are synced back to documentation — shipped 2026-07-08.
## Product Scan — Ground Truth Annotation & Accuracy ## Product Scan — Ground Truth Annotation & Accuracy
- **6.1** Built standalone annotation page `manual-label-scan/page.tsx` for ground truth editing. Includes image browser, editable fields (`no_sku`, `nama_item`, `expiry_date`, `notes`), and a "Scan with AI" fill-blanks feature — shipped 2026-07-08. - **6.1** Built standalone annotation page `manual-label-scan/page.tsx` for ground truth editing. Includes image browser, editable fields (`no_sku`, `nama_item`, `expiry_date`, `notes`), and a "Scan with AI" fill-blanks feature — shipped 2026-07-08.
- **6.2** API + storage groundwork for scan annotation. Extended `api/manual-label-scan` with `GET` list mode and `DELETE`. Persisted uploaded scan photos as base64 images into `sources/product-test-images/`. Made the `scan-pfm` quick-save honest by allowing manual correction before save — shipped 2026-07-08. - **6.2** API + storage groundwork for scan annotation. Extended `api/manual-label-scan` with `GET` list mode and `DELETE`. Persisted uploaded scan photos as base64 images into `sources/product-test-images/`. Made the `scan-pfm` quick-save honest by allowing manual correction before save — shipped 2026-07-08.
- **6.3** Built `backend/scripts/accuracy-check-scan.mts` mirroring the DO-harness architecture, measuring overall match rate plus per-field breakdown (`no_sku`, `expiry_date`) against the new stable labels — shipped 2026-07-08. - **6.3** Built `backend/scripts/accuracy-check-scan.mts` mirroring the DO-harness architecture, measuring overall match rate plus per-field breakdown (`no_sku`, `expiry_date`) against the new stable labels — shipped 2026-07-08.
- **6.4** Ported the DO-harness's auto-diff-vs-previous-run reporting into `accuracy-check-scan.mts`: every run now prints a Δ column per field per split (Training/Validation) vs the last `product_accuracy_history.jsonl` entry, and calls out field- and image-level regressions/improvements explicitly. Added classifier method (`dinov2_similarity`/`yolo_classifier`) distribution and average confidence as informational (non-scoring) context. Created the previously-missing `sources/product-test-images/README.md` documenting the validation-photo drop workflow — shipped 2026-07-13, user-directed `n` request to make algorithm tuning self-verifying. - **6.4** Ported the DO-harness's auto-diff-vs-previous-run reporting into `accuracy-check-scan.mts`: every run now prints a Δ column per field per split (Training/Validation) vs the last `product_accuracy_history.jsonl` entry, and calls out field- and image-level regressions/improvements explicitly. Added classifier method (`dinov2_similarity`/`yolo_classifier`) distribution and average confidence as informational (non-scoring) context. Created the previously-missing `sources/product-test-images/README.md` documenting the validation-photo drop workflow — shipped 2026-07-13, user-directed `n` request to make algorithm tuning self-verifying.
### Master Data Management ### Master Data Management
- **8.1 & 8.3 CRUD APIs and Web UI**: Created `/api/v1/master/stores` and `/api/v1/master/skus` endpoints alongside a Next.js Admin page (`/admin/master-data`) to visually manage the core reference data used by the OCR matching engine — shipped 2026-07-08. - **8.1 & 8.3 CRUD APIs and Web UI**: Created `/api/v1/master/stores` and `/api/v1/master/skus` endpoints alongside a Next.js Admin page (`/admin/master-data`) to visually manage the core reference data used by the OCR matching engine — shipped 2026-07-08.
- **8.2 Auto-Provisioning Store Accounts**: Store creation now automatically securely hashes a default password ("123") and creates a paired login account, keeping store configuration perfectly in sync with the `accounts` table — shipped 2026-07-08. - **8.2 Auto-Provisioning Store Accounts**: Store creation now automatically securely hashes a default password ("123") and creates a paired login account, keeping store configuration perfectly in sync with the `accounts` table — shipped 2026-07-08.
## Auth — Store Accounts & Profile-Sourced Metadata ## Auth — Store Accounts & Profile-Sourced Metadata
- **7.1** Seeded one account per store in `db/init.ts` during initialization by assigning `username = kode_toko` and a bcrypt-hashed default password `"123"`. Included `role` and `is_active` schema additions — shipped 2026-07-08. - **7.1** Seeded one account per store in `db/init.ts` during initialization by assigning `username = kode_toko` and a bcrypt-hashed default password `"123"`. Included `role` and `is_active` schema additions — shipped 2026-07-08.
- **7.2** Enhanced authentication routing by modifying `POST /api/v1/auth/login` to perform a `LEFT JOIN` on `store_master`, returning the extended store profile alongside the token. Added a guard to reject login if `is_active = false`. Implemented a new `GET /api/v1/auth/me` endpoint to cleanly re-fetch the profile via token — shipped 2026-07-08. - **7.2** Enhanced authentication routing by modifying `POST /api/v1/auth/login` to perform a `LEFT JOIN` on `store_master`, returning the extended store profile alongside the token. Added a guard to reject login if `is_active = false`. Implemented a new `GET /api/v1/auth/me` endpoint to cleanly re-fetch the profile via token — shipped 2026-07-08.
- **7.3** Created a reproducible `store_master` bootstrap logic in `db/init.ts` that reads from `sources/toko_aktif.json` idempotently on startup. Also correctly seeded the `WH_JOFFICE` head office to resolve the admin account foreign-key setup constraint — shipped 2026-07-08. - **7.3** Created a reproducible `store_master` bootstrap logic in `db/init.ts` that reads from `sources/toko_aktif.json` idempotently on startup. Also correctly seeded the `WH_JOFFICE` head office to resolve the admin account foreign-key setup constraint — shipped 2026-07-08.
## Backend — Document Confirmation Gate & Data Hygiene ## Backend — Document Confirmation Gate & Data Hygiene
- **10.1** Added a `confirmed BOOLEAN NOT NULL DEFAULT TRUE` column to `documents` (`db/init.ts`, `ALTER TABLE ... ADD COLUMN IF NOT EXISTS` — grandfathers every pre-existing row so today's history didn't go empty after migration) and used it to separate "OCR finished" from "user confirmed": previously `GET /api/v1/documents` filtered only on `parsed = true`, which the backend sets synchronously right after upload — before the mobile user ever taps "Simpan & Konfirmasi" in the editor — so a scan captured, previewed, then backed out of (never confirmed) was already sitting in every entitled account's document list with blank/placeholder fields (root cause of `document_card.dart`'s "Staff Toko" fallback text on the Flutter side). `v1/documents/upload/route.ts` now explicitly inserts `confirmed = false` on every new upload; `v1/documents/[id]/route.ts`'s `PUT` handler is the *only* place that flips it to `true` (literally "the user confirmed"); `v1/documents/route.ts` (list) now filters `AND confirmed = true` unconditionally for every account including `admin` (no role special-casing, per explicit user decision); `v1/documents/[id]/route.ts`'s `GET`-by-id handler is deliberately untouched by the new filter so the mobile poller can keep seeing pending/unconfirmed documents mid-flow. `utils/document-mapper.ts`'s shared `DocumentRow`/`mapDocumentRow()` now carries `confirmed` through to all three call sites (list, GET-by-id, upload's dedup-hit branch) from one place. `parse/route.ts`'s own `INSERT ... ON CONFLICT (filename) DO UPDATE` statements (both DO and Product branches) were deliberately left untouched for `confirmed` — in the real mobile flow the upload route's INSERT always runs first, so this upsert always hits the `ON CONFLICT` branch, and since its `SET` clause doesn't mention `confirmed`, Postgres correctly leaves the existing value alone (verified this is correct, not an oversight). Verified live against the running Docker stack: uploaded a real DO photo as a store account without confirming it — absent from that store's list (and from `admin`'s) while `GET /documents/:id` still reported the correct `parseStatus`; `PUT` (confirm) made it appear immediately with the real submitted data; all 13 pre-existing rows carried `confirmed = true` after the migration ran — shipped 2026-07-10. - **10.1** Added a `confirmed BOOLEAN NOT NULL DEFAULT TRUE` column to `documents` (`db/init.ts`, `ALTER TABLE ... ADD COLUMN IF NOT EXISTS` — grandfathers every pre-existing row so today's history didn't go empty after migration) and used it to separate "OCR finished" from "user confirmed": previously `GET /api/v1/documents` filtered only on `parsed = true`, which the backend sets synchronously right after upload — before the mobile user ever taps "Simpan & Konfirmasi" in the editor — so a scan captured, previewed, then backed out of (never confirmed) was already sitting in every entitled account's document list with blank/placeholder fields (root cause of `document_card.dart`'s "Staff Toko" fallback text on the Flutter side). `v1/documents/upload/route.ts` now explicitly inserts `confirmed = false` on every new upload; `v1/documents/[id]/route.ts`'s `PUT` handler is the *only* place that flips it to `true` (literally "the user confirmed"); `v1/documents/route.ts` (list) now filters `AND confirmed = true` unconditionally for every account including `admin` (no role special-casing, per explicit user decision); `v1/documents/[id]/route.ts`'s `GET`-by-id handler is deliberately untouched by the new filter so the mobile poller can keep seeing pending/unconfirmed documents mid-flow. `utils/document-mapper.ts`'s shared `DocumentRow`/`mapDocumentRow()` now carries `confirmed` through to all three call sites (list, GET-by-id, upload's dedup-hit branch) from one place. `parse/route.ts`'s own `INSERT ... ON CONFLICT (filename) DO UPDATE` statements (both DO and Product branches) were deliberately left untouched for `confirmed` — in the real mobile flow the upload route's INSERT always runs first, so this upsert always hits the `ON CONFLICT` branch, and since its `SET` clause doesn't mention `confirmed`, Postgres correctly leaves the existing value alone (verified this is correct, not an oversight). Verified live against the running Docker stack: uploaded a real DO photo as a store account without confirming it — absent from that store's list (and from `admin`'s) while `GET /documents/:id` still reported the correct `parseStatus`; `PUT` (confirm) made it appear immediately with the real submitted data; all 13 pre-existing rows carried `confirmed = true` after the migration ran — shipped 2026-07-10.
- **10.2** Removed the fabricated Product Scan placeholder values `noPO: "PO-PRODUCT-001"`, `noSO: "1002003004"`, `noDO: "DO-PRODUCT-999"` (both the flat keys and the mirrored `header.no_po`/`no_so`/`no_do` sub-object) from `parse/route.ts`'s Product-scan branch, replacing them with empty strings — these are DO-specific concepts that don't apply to a product verification scan, and were never actually read by anything: `pdf_service.dart`'s Product receipt branch never prints them, and `product_editor_submit_logic.dart`'s `_submit()` builds its own `noPo`/`noSo`/`noDo` from the user's PO-link dropdown and batch selection, ignoring the stored values entirely. Same class of issue as the earlier G7 fix (fabricated data presented as if real) — low risk to remove since nothing meaningfully depended on the old values. Scope stayed narrow to exactly these three fields; `nama_driver`/`nama_penerima`'s "PRODUCT SCAN"/"STORE STAFF" placeholders were left alone as a deliberate fixed convention, not a fabricated document number. Verified via `curl`: a freshly-uploaded, unconfirmed Product Scan document's raw `GET /documents/:id` response now returns `no_po`/`no_so`/`no_do` as empty strings instead of the old fake values — shipped 2026-07-10. - **10.2** Removed the fabricated Product Scan placeholder values `noPO: "PO-PRODUCT-001"`, `noSO: "1002003004"`, `noDO: "DO-PRODUCT-999"` (both the flat keys and the mirrored `header.no_po`/`no_so`/`no_do` sub-object) from `parse/route.ts`'s Product-scan branch, replacing them with empty strings — these are DO-specific concepts that don't apply to a product verification scan, and were never actually read by anything: `pdf_service.dart`'s Product receipt branch never prints them, and `product_editor_submit_logic.dart`'s `_submit()` builds its own `noPo`/`noSo`/`noDo` from the user's PO-link dropdown and batch selection, ignoring the stored values entirely. Same class of issue as the earlier G7 fix (fabricated data presented as if real) — low risk to remove since nothing meaningfully depended on the old values. Scope stayed narrow to exactly these three fields; `nama_driver`/`nama_penerima`'s "PRODUCT SCAN"/"STORE STAFF" placeholders were left alone as a deliberate fixed convention, not a fabricated document number. Verified via `curl`: a freshly-uploaded, unconfirmed Product Scan document's raw `GET /documents/:id` response now returns `no_po`/`no_so`/`no_do` as empty strings instead of the old fake values — shipped 2026-07-10.
## Backend — Single-Pass Product Classification ## Backend — Single-Pass Product Classification
- **11.1** Eliminated the duplicate GPU classification pass on Product Scan (gap G3), sourced from user feedback that the review screen took noticeably longer to open than DO Scan's. `api/parse/route.ts`'s Product branch previously had its own separate, poorer inline classify call (kept only `top1_name`/`extracted_sku`), forcing the Flutter editor to re-run the entire classify+OCR pipeline a second time via `POST /api/v1/scan-product` just to get the top-5 candidate list and OCR-extracted expiry date. Now calls the same shared `classifyAndMatchProduct()` (`utils/product-scan.ts`) already used by that v1 route — one GPU call, richer result — and persists it under a new `metadata.productScan` JSONB key (no schema migration), surfaced by `document-mapper.ts` as a top-level `productScan` field on every GET response. Caught and fixed a real regression along the way: delegating to the shared function silently dropped the 90s pipeline timeout the old inline fetch had; added the same bound (`PIPELINE_TIMEOUT_MS`) directly inside `classifyAndMatchProduct()` so both callers — this route and the live `POST /api/v1/scan-product` (which never had the bound either) — are protected. Verified via `curl` with a genuinely fresh image/store combination (proving a real classify pass, not a dedup hit): took 9s, and the immediate `GET /documents/:id` response already contained 5 real `possibleMatches` and the extracted expiry date, before any editor interaction — shipped 2026-07-10. - **11.1** Eliminated the duplicate GPU classification pass on Product Scan (gap G3), sourced from user feedback that the review screen took noticeably longer to open than DO Scan's. `api/parse/route.ts`'s Product branch previously had its own separate, poorer inline classify call (kept only `top1_name`/`extracted_sku`), forcing the Flutter editor to re-run the entire classify+OCR pipeline a second time via `POST /api/v1/scan-product` just to get the top-5 candidate list and OCR-extracted expiry date. Now calls the same shared `classifyAndMatchProduct()` (`utils/product-scan.ts`) already used by that v1 route — one GPU call, richer result — and persists it under a new `metadata.productScan` JSONB key (no schema migration), surfaced by `document-mapper.ts` as a top-level `productScan` field on every GET response. Caught and fixed a real regression along the way: delegating to the shared function silently dropped the 90s pipeline timeout the old inline fetch had; added the same bound (`PIPELINE_TIMEOUT_MS`) directly inside `classifyAndMatchProduct()` so both callers — this route and the live `POST /api/v1/scan-product` (which never had the bound either) — are protected. Verified via `curl` with a genuinely fresh image/store combination (proving a real classify pass, not a dedup hit): took 9s, and the immediate `GET /documents/:id` response already contained 5 real `possibleMatches` and the extracted expiry date, before any editor interaction — shipped 2026-07-10.
File diff suppressed because it is too large. Load diff
+214 -214
View File
@@ -1,214 +1,214 @@
# Product Scan (scan-pfm) — How It Works # Product Scan (scan-pfm) — How It Works
End-to-end reference for the Product/SKU scanning feature: a photo of a Primafood End-to-end reference for the Product/SKU scanning feature: a photo of a Primafood
product package goes in; the SKU class, product name, expiry date, and a ranked product package goes in; the SKU class, product name, expiry date, and a ranked
SKU-master match list come out. Written 2026-07-08 against the live code. Related: SKU-master match list come out. Written 2026-07-08 against the live code. Related:
`plans/next-enhancements.md` §2 (build history) and §6 (ground-truth roadmap); `plans/next-enhancements.md` §2 (build history) and §6 (ground-truth roadmap);
`docs/feature-list.md` tasks 2.1/2.3. `docs/feature-list.md` tasks 2.1/2.3.
## High-level flow ## High-level flow
```mermaid ```mermaid
flowchart LR flowchart LR
A[Browser: /scan-pfm page] -->|"POST /api/scan-pfm {image_base64}"| B[Next.js gateway<br/>pfm-web-app :3000] A[Browser: /scan-pfm page] -->|"POST /api/scan-pfm {image_base64}"| B[Next.js gateway<br/>pfm-web-app :3000]
B -->|"POST :8120/classify-ocr"| C[classify_ocr_server.py<br/>FastAPI, in pipeline-api] B -->|"POST :8120/classify-ocr"| C[classify_ocr_server.py<br/>FastAPI, in pipeline-api]
C --> C1[1. DINOv2 similarity<br/>fallback: YOLO classifier] C --> C1[1. DINOv2 similarity<br/>fallback: YOLO classifier]
C --> C2[2. PaddleOCR + regex<br/>SKU / expiry / name] C --> C2[2. PaddleOCR + regex<br/>SKU / expiry / name]
C -->|"POST localhost:8090/layout-parsing<br/>promptLabel: spotting"| D[PaddleX pipeline<br/>same container] C -->|"POST localhost:8090/layout-parsing<br/>promptLabel: spotting"| D[PaddleX pipeline<br/>same container]
B -->|"POST :8090/layout-parsing"| D B -->|"POST :8090/layout-parsing"| D
B -->|"SELECT sku_master"| E[(Postgres)] B -->|"SELECT sku_master"| E[(Postgres)]
B -->|Levenshtein ranking| A B -->|Levenshtein ranking| A
``` ```
Two processes live in the `paddleocr-pipeline-api` container, both started by Two processes live in the `paddleocr-pipeline-api` container, both started by
`scripts/serve-pipeline.sh`: the PaddleX layout-parsing pipeline on **:8090** `scripts/serve-pipeline.sh`: the PaddleX layout-parsing pipeline on **:8090**
(shared with the DO flow; VL recognition goes out to the vLLM server on :8118) and (shared with the DO flow; VL recognition goes out to the vLLM server on :8118) and
`config/classify_ocr_server.py` on **:8120** (product scan only). The gateway `config/classify_ocr_server.py` on **:8120** (product scan only). The gateway
reaches them via Docker DNS (`CLASSIFIER_SERVER_URL`, `PIPELINE_URL` in root reaches them via Docker DNS (`CLASSIFIER_SERVER_URL`, `PIPELINE_URL` in root
`docker-compose.yml:87-88`); nginx (:8000) proxies `/scan-pfm` to the Next.js app. `docker-compose.yml:87-88`); nginx (:8000) proxies `/scan-pfm` to the Next.js app.
## Request walkthrough ## Request walkthrough
1. **Page** (`pfm-web-app/src/app/scan-pfm/page.tsx`, desktop-only test UI): pick a 1. **Page** (`pfm-web-app/src/app/scan-pfm/page.tsx`, desktop-only test UI): pick a
sample from the gallery (`GET /api/produk-pfm`) or upload/rotate a photo (rotation sample from the gallery (`GET /api/produk-pfm`) or upload/rotate a photo (rotation
is done client-side on a canvas), then send it as a base64 data-URL. is done client-side on a canvas), then send it as a base64 data-URL.
2. **Gateway** (`api/scan-pfm/route.ts`): 2. **Gateway** (`api/scan-pfm/route.ts`):
- forwards `{image_base64}` to the classifier server (`/classify-ocr`); - forwards `{image_base64}` to the classifier server (`/classify-ocr`);
- separately calls the layout-parsing pipeline with `useLayoutDetection: true` - separately calls the layout-parsing pipeline with `useLayoutDetection: true`
for the Visual Grid tab's output images (failure here is non-fatal — logged, for the Visual Grid tab's output images (failure here is non-fatal — logged,
`layoutParsingResult` returns `null`); `layoutParsingResult` returns `null`);
- loads the full `sku_master` table and ranks every SKU by **Levenshtein - loads the full `sku_master` table and ranks every SKU by **Levenshtein
similarity between `nama_item` and the classifier's `top1_name`** similarity between `nama_item` and the classifier's `top1_name`**
(lowercased, alphanumerics only). Top 5 with score > 0.1 are returned; (lowercased, alphanumerics only). Top 5 with score > 0.1 are returned;
rank 1 gets `isBestMatch: true`. Note: `ocr.extracted_sku` and rank 1 gets `isBestMatch: true`. Note: `ocr.extracted_sku` and
`ocr.extracted_product_name` are read but **not used** in this ranking — `ocr.extracted_product_name` are read but **not used** in this ranking —
see Future recommendations. see Future recommendations.
3. **Classifier server** (`config/classify_ocr_server.py`) does classification, 3. **Classifier server** (`config/classify_ocr_server.py`) does classification,
OCR extraction, and visualization — detailed below — and returns OCR extraction, and visualization — detailed below — and returns
`{classification, ocr}`. `{classification, ocr}`.
4. **Page renders** four tabs: Summary (classification card + top-5 override 4. **Page renders** four tabs: Summary (classification card + top-5 override
"Use" buttons + OCR fields + SKU matches), Visual Grid, Spotting Grid, Raw "Use" buttons + OCR fields + SKU matches), Visual Grid, Spotting Grid, Raw
Response (JSON). "Save Ground Truth" posts to `/api/manual-label-scan`. Response (JSON). "Save Ground Truth" posts to `/api/manual-label-scan`.
## Stage 1 — classification (which product is this?) ## Stage 1 — classification (which product is this?)
**Primary: DINOv2 similarity search** (`method: "dinov2_similarity"`). At startup **Primary: DINOv2 similarity search** (`method: "dinov2_similarity"`). At startup
the server loads `dinov2_vits14` **from `torch.hub` (network fetch on first run)** the server loads `dinov2_vits14` **from `torch.hub` (network fetch on first run)**
plus `models/dinov2_index.pkl` — precomputed L2-normalized 384-dim embeddings of plus `models/dinov2_index.pkl` — precomputed L2-normalized 384-dim embeddings of
all 118 reference photos across 16 SKU class folders. Per request: embed the query all 118 reference photos across 16 SKU class folders. Per request: embed the query
image (resize 224², ImageNet normalization), dot-product against all reference image (resize 224², ImageNet normalization), dot-product against all reference
embeddings (= cosine similarity), then aggregate **per class = max similarity of embeddings (= cosine similarity), then aggregate **per class = max similarity of
any reference photo in that class**. Classes sorted by similarity become any reference photo in that class**. Classes sorted by similarity become
`all_probabilities`. Caveat: these "confidences" are cosine similarities, **not `all_probabilities`. Caveat: these "confidences" are cosine similarities, **not
probabilities** — they don't sum to 1 and are typically all high (0.4–0.9); probabilities** — they don't sum to 1 and are typically all high (0.4–0.9);
compare relatively, not against an absolute threshold. compare relatively, not against an absolute threshold.
**Fallback: YOLO classifier** (`method: "yolo_classifier"`) — only when DINOv2 is **Fallback: YOLO classifier** (`method: "yolo_classifier"`) — only when DINOv2 is
unavailable (no index/model) or throws. A fine-tuned `yolo26n-cls` checkpoint; unavailable (no index/model) or throws. A fine-tuned `yolo26n-cls` checkpoint;
its `all_probabilities` are real softmax probabilities. Weights are its `all_probabilities` are real softmax probabilities. Weights are
**auto-discovered**: `CLASSIFIER_MODEL_PATH` env wins; otherwise the newest **auto-discovered**: `CLASSIFIER_MODEL_PATH` env wins; otherwise the newest
`produk-pfm-classifier-26n-*e-*.pt` in `models/` by (date-in-filename, mtime) — `produk-pfm-classifier-26n-*e-*.pt` in `models/` by (date-in-filename, mtime) —
so retraining just drops a new dated file, no config change. so retraining just drops a new dated file, no config change.
If both are unavailable, `classification` carries an `error` field instead. If both are unavailable, `classification` carries an `error` field instead.
## Stage 2 — OCR extraction (SKU, expiry date, product name) ## Stage 2 — OCR extraction (SKU, expiry date, product name)
PaddleOCR (`lang='en'`, textline orientation on) produces `rec_texts` lines + PaddleOCR (`lang='en'`, textline orientation on) produces `rec_texts` lines +
`rec_polys` boxes. Three extractors run over the lines: `rec_polys` boxes. Three extractors run over the lines:
- **SKU** (`extract_sku`): first 8-digit number anywhere; else first 7–9 digit - **SKU** (`extract_sku`): first 8-digit number anywhere; else first 7–9 digit
number. (Primafood SKUs are 8 digits, printed near the label top.) number. (Primafood SKUs are 8 digits, printed near the label top.)
- **Expiry date** (`extract_expired_date`): each line is first noise-cleaned - **Expiry date** (`extract_expired_date`): each line is first noise-cleaned
(`clean_date_line`: `1)`→`0`, `()`→`0`, `B8/8B/88`→`BB` before digits, o→0, (`clean_date_line`: `1)`→`0`, `()`→`0`, `B8/8B/88`→`BB` before digits, o→0,
I/l/|→1, S→5, Z→2, B→8 when digit-flanked, plus `012`/`112` month-misread I/l/|→1, S→5, Z→2, B→8 when digit-flanked, plus `012`/`112` month-misread
repairs), then a **6-level priority cascade** runs: (1) BB/EXP-keyword line repairs), then a **6-level priority cascade** runs: (1) BB/EXP-keyword line
with compact `DDMMYYYY`; (2) keyword line with spaced `DD MM YYYY`; (3) with compact `DDMMYYYY`; (2) keyword line with spaced `DD MM YYYY`; (3)
keyword + 6–8 digit run; (3.5) keyword line, lenient noisy match; (4) any line keyword + 6–8 digit run; (3.5) keyword line, lenient noisy match; (4) any line
spaced date; (5) any line compact `DDMMYYYY` — skipping lines that look like a spaced date; (5) any line compact `DDMMYYYY` — skipping lines that look like a
SKU-on-product-name; (6) legacy formats (slashes, `05 MAR 2027`). Recognized SKU-on-product-name; (6) legacy formats (slashes, `05 MAR 2027`). Recognized
keywords: `EXP`, `EXPIRED`, `TGL`, `EXPIRY`, `BBD`, `BEST BEFORE`, `BB`, keywords: `EXP`, `EXPIRED`, `TGL`, `EXPIRY`, `BBD`, `BEST BEFORE`, `BB`,
`BAIK DIGUNAKAN`. Output normalized to `DD/MM/YYYY`. `BAIK DIGUNAKAN`. Output normalized to `DD/MM/YYYY`.
- **Product name** (`extract_product_name`): longest line containing a brand/ - **Product name** (`extract_product_name`): longest line containing a brand/
product keyword (FIESTA, CHAMP, OKEY, AKUMO, ASIMO, NUGGET, SOSIS, …) after product keyword (FIESTA, CHAMP, OKEY, AKUMO, ASIMO, NUGGET, SOSIS, …) after
stripping SKU digits and date fragments; falls back to the classifier's stripping SKU digits and date fragments; falls back to the classifier's
`top1_name`, then the longest non-numeric line, then `"Unknown Product"`. `top1_name`, then the longest non-numeric line, then `"Unknown Product"`.
Visualization artifacts built server-side: `vis_image_base64` (all OCR boxes Visualization artifacts built server-side: `vis_image_base64` (all OCR boxes
drawn teal `TEXT`, the expiry line amber `EXP`, on the orientation-corrected drawn teal `TEXT`, the expiry line amber `EXP`, on the orientation-corrected
image so boxes align), `expired_date_crop_base64` (padded crop of the expiry image so boxes align), `expired_date_crop_base64` (padded crop of the expiry
line for eyeball verification — `find_expired_crop_index` prefers the box whose line for eyeball verification — `find_expired_crop_index` prefers the box whose
digits actually contain the date), and `spotting_image_base64` (a second digits actually contain the date), and `spotting_image_base64` (a second
pipeline call with `promptLabel: "spotting"`, no layout detection). pipeline call with `promptLabel: "spotting"`, no layout detection).
## Endpoint reference ## Endpoint reference
| Endpoint | Where | Purpose | | Endpoint | Where | Purpose |
|---|---|---| |---|---|---|
| `POST /api/scan-pfm` | gateway | Main scan. Body `{image_base64}` (data-URL ok). Returns `{classification, ocr, possibleMatches[], layoutParsingResult}` | | `POST /api/scan-pfm` | gateway | Main scan. Body `{image_base64}` (data-URL ok). Returns `{classification, ocr, possibleMatches[], layoutParsingResult}` |
| `POST http://paddleocr-pipeline-api:8120/classify-ocr` | classifier server | Internal. Body `{image_base64}`. Returns `{classification: {top1_name, top1_confidence, all_probabilities[], method}, ocr: {text_lines[], extracted_product_name, extracted_sku, extracted_expired_date, expired_line_index, expired_source_line, expired_date_crop_base64, vis_image_base64, spotting_image_base64}}` | | `POST http://paddleocr-pipeline-api:8120/classify-ocr` | classifier server | Internal. Body `{image_base64}`. Returns `{classification: {top1_name, top1_confidence, all_probabilities[], method}, ocr: {text_lines[], extracted_product_name, extracted_sku, extracted_expired_date, expired_line_index, expired_source_line, expired_date_crop_base64, vis_image_base64, spotting_image_base64}}` |
| `GET /api/produk-pfm` | gateway | Gallery: SKU folders under `public/produk-pfm/foto-kemasan-v2/` with image + thumb URLs | | `GET /api/produk-pfm` | gateway | Gallery: SKU folders under `public/produk-pfm/foto-kemasan-v2/` with image + thumb URLs |
| `GET/POST /api/manual-label-scan` | gateway | Ground-truth read/upsert to `sources/product_manual_labels.json` (host-visible via the `./backend/sources:/sources` mount) | | `GET/POST /api/manual-label-scan` | gateway | Ground-truth read/upsert to `sources/product_manual_labels.json` (host-visible via the `./backend/sources:/sources` mount) |
| `POST :8090/layout-parsing` | pipeline | Shared PaddleX pipeline; used here for Visual Grid images and (with `promptLabel: "spotting"`) the Spotting Grid | | `POST :8090/layout-parsing` | pipeline | Shared PaddleX pipeline; used here for Visual Grid images and (with `promptLabel: "spotting"`) the Spotting Grid |
| `/scan-pfm` | nginx :8000 | Proxies the page to Next.js :3000 | | `/scan-pfm` | nginx :8000 | Proxies the page to Next.js :3000 |
`possibleMatches[]` items: `{no_sku, nama_item, score, yoloSimilarity, isBestMatch}` — `possibleMatches[]` items: `{no_sku, nama_item, score, yoloSimilarity, isBestMatch}` —
`score` currently equals `yoloSimilarity` (name-vs-name Levenshtein, 0..1). `score` currently equals `yoloSimilarity` (name-vs-name Levenshtein, 0..1).
## Model artifacts & retraining ## Model artifacts & retraining
| File (`pfm-web-app/public/produk-pfm/`) | What | | File (`pfm-web-app/public/produk-pfm/`) | What |
|---|---| |---|---|
| `foto-kemasan-v2/<SKU or class>/…` | Reference photo dataset — 81 classes, 2,493 photos (target ~230 SKU) | | `foto-kemasan-v2/<SKU or class>/…` | Reference photo dataset — 81 classes, 2,493 photos (target ~230 SKU) |
| `models/dinov2_index.pkl` | DINOv2 embeddings + metadata (rebuild after adding photos) — currently indexes all 2,493 photos across 81 classes | | `models/dinov2_index.pkl` | DINOv2 embeddings + metadata (rebuild after adding photos) — currently indexes all 2,493 photos across 81 classes |
| `models/produk-pfm-classifier-26n-100e-2026-07-14.pt` / `.onnx` | Fine-tuned YOLO classifier (85.8% top-1 / 94.4% top-5 val across all 81 classes; retrained 2026-07-14, 54m21s on an RTX 2060, up from the prior 2026-07-08 model's 83.3%/90% on only 16 classes) | | `models/produk-pfm-classifier-26n-100e-2026-07-14.pt` / `.onnx` | Fine-tuned YOLO classifier (85.8% top-1 / 94.4% top-5 val across all 81 classes; retrained 2026-07-14, 54m21s on an RTX 2060, up from the prior 2026-07-08 model's 83.3%/90% on only 16 classes) |
| `index_dinov2.py` | Rebuilds the pickle index from `foto-kemasan-v2/` | | `index_dinov2.py` | Rebuilds the pickle index from `foto-kemasan-v2/` |
| `train_classifier.py` | Splits 80/20 into `yolo_dataset/`, fine-tunes `yolo26n-cls.pt` (default 100 epochs, `--imgsz 224`), writes a dated checkpoint | | `train_classifier.py` | Splits 80/20 into `yolo_dataset/`, fine-tunes `yolo26n-cls.pt` (default 100 epochs, `--imgsz 224`), writes a dated checkpoint |
**Retraining procedure (Windows host — bare-metal doesn't work here, **Retraining procedure (Windows host — bare-metal doesn't work here,
`paddlepaddle-gpu` wheels are Linux-only):** add photos to `foto-kemasan-v2/`, `paddlepaddle-gpu` wheels are Linux-only):** add photos to `foto-kemasan-v2/`,
`docker compose build pipeline-api` from the **repo root**, run a one-off `docker compose build pipeline-api` from the **repo root**, run a one-off
`docker run --gpus all` from that image with `models/` mounted **writable** (the `docker run --gpus all` from that image with `models/` mounted **writable** (the
live service mounts it `:ro`), run `index_dinov2.py` then live service mounts it `:ro`), run `index_dinov2.py` then
`train_classifier.py train --imgsz 224`, then `docker compose restart `train_classifier.py train --imgsz 224`, then `docker compose restart
pipeline-api`. From Git Bash prefix `MSYS_NO_PATHCONV=1` or `/app/...` arguments pipeline-api`. From Git Bash prefix `MSYS_NO_PATHCONV=1` or `/app/...` arguments
get mangled. Verify in `docker logs`: "DINOv2 index loaded with N reference get mangled. Verify in `docker logs`: "DINOv2 index loaded with N reference
images", "Using classifier weights: <new dated file>". Full worked example: images", "Using classifier weights: <new dated file>". Full worked example:
`plans/next-enhancements.md` task 2.1. `plans/next-enhancements.md` task 2.1.
## Accuracy regression harness ## Accuracy regression harness
`backend/scripts/accuracy-check-scan.mts` — mirrors the DO-flow's `backend/scripts/accuracy-check-scan.mts` — mirrors the DO-flow's
`pfm-web-app/scripts/accuracy-check.mts`. Hits the live `/api/scan-pfm` for `pfm-web-app/scripts/accuracy-check.mts`. Hits the live `/api/scan-pfm` for
every labeled image in `sources/product_manual_labels.json`, checks 3 fields every labeled image in `sources/product_manual_labels.json`, checks 3 fields
(`no_sku`, `nama_item`, `expiry_date`) against ground truth, and splits into: (`no_sku`, `nama_item`, `expiry_date`) against ground truth, and splits into:
- **Training Set** — gallery photos under `foto-kemasan-v2/` (the classifier's - **Training Set** — gallery photos under `foto-kemasan-v2/` (the classifier's
own reference images; scores here measure memorization, not generalization). own reference images; scores here measure memorization, not generalization).
- **Validation Set** — flat filenames, scored from the frozen - **Validation Set** — flat filenames, scored from the frozen
`sources/product-test-images-fixed/` snapshot (renamed `<index> <no_sku>.<ext>`, `sources/product-test-images-fixed/` snapshot (renamed `<index> <no_sku>.<ext>`,
built by `scripts/freeze-validation-set.mjs`) so a rerun always grades the built by `scripts/freeze-validation-set.mjs`) so a rerun always grades the
same 79 images regardless of what's since been dropped into the live-intake same 79 images regardless of what's since been dropped into the live-intake
`sources/product-test-images/` folder. See each folder's `README.md` — the `sources/product-test-images/` folder. See each folder's `README.md` — the
live folder documents the drop-photo → label → re-run-freeze-script workflow live folder documents the drop-photo → label → re-run-freeze-script workflow
via `/manual-label-scan`; the fixed folder documents the freeze/promote step via `/manual-label-scan`; the fixed folder documents the freeze/promote step
and flags 5 SKUs (12010801, 12012504, 12130504, 13050101, 15040102) whose and flags 5 SKUs (12010801, 12012504, 12130504, 13050101, 15040102) whose
only available photo was already used to train the classifier, so their only available photo was already used to train the classifier, so their
scores aren't a clean held-out result. scores aren't a clean held-out result.
Every run appends to `sources/product_accuracy_history.jsonl` and **auto-diffs Every run appends to `sources/product_accuracy_history.jsonl` and **auto-diffs
against the previous run**: the printed summary shows a Δ column per field per against the previous run**: the printed summary shows a Δ column per field per
split, flags field/image-level regressions and improvements, and reports split, flags field/image-level regressions and improvements, and reports
classifier method (`dinov2_similarity`/`yolo_classifier`) distribution + classifier method (`dinov2_similarity`/`yolo_classifier`) distribution +
average confidence as informational context (not scored pass/fail, since average confidence as informational context (not scored pass/fail, since
DINOv2's "confidence" is a raw cosine similarity, not a calibrated DINOv2's "confidence" is a raw cosine similarity, not a calibrated
probability — see Stage 1 above). This is what makes it safe to tune probability — see Stage 1 above). This is what makes it safe to tune
`classify_ocr_server.py` and immediately see whether a change helped or hurt. `classify_ocr_server.py` and immediately see whether a change helped or hurt.
```bash ```bash
node scripts/accuracy-check-scan.mts # from backend/ node scripts/accuracy-check-scan.mts # from backend/
``` ```
## Operational notes ## Operational notes
- **Env vars**: `CLASSIFIER_SERVER_URL`, `PIPELINE_URL` (gateway, set in compose); - **Env vars**: `CLASSIFIER_SERVER_URL`, `PIPELINE_URL` (gateway, set in compose);
`CLASSIFIER_MODELS_DIR`, `CLASSIFIER_MODEL_PATH` (classifier server overrides). `CLASSIFIER_MODELS_DIR`, `CLASSIFIER_MODEL_PATH` (classifier server overrides).
The gateway's in-code default `PIPELINE_URL` (`localhost:7871`) is stale — the The gateway's in-code default `PIPELINE_URL` (`localhost:7871`) is stale — the
compose env always overrides it in Docker. compose env always overrides it in Docker.
- **Startup order/health**: the classifier server loads DINOv2 (torch.hub → - **Startup order/health**: the classifier server loads DINOv2 (torch.hub →
needs network/cache), YOLO, and PaddleOCR at import time; until done, :8120 needs network/cache), YOLO, and PaddleOCR at import time; until done, :8120
refuses connections and `/api/scan-pfm` 500s. No healthcheck exists yet (plan refuses connections and `/api/scan-pfm` 500s. No healthcheck exists yet (plan
task 4.2 / 1.6). task 4.2 / 1.6).
- **GPU**: DINOv2 + YOLO + PaddleOCR share the container/GPU with the PaddleX - **GPU**: DINOv2 + YOLO + PaddleOCR share the container/GPU with the PaddleX
pipeline; all are small (ViT-S/14, nano YOLO) next to the vLLM server's pipeline; all are small (ViT-S/14, nano YOLO) next to the vLLM server's
footprint, but they do add VRAM on the same `PIPELINE_DEVICE`. footprint, but they do add VRAM on the same `PIPELINE_DEVICE`.
- **Failure isolation**: layout-vis and spotting calls are best-effort - **Failure isolation**: layout-vis and spotting calls are best-effort
(`null`/absent on failure); classification and OCR errors surface as `error` (`null`/absent on failure); classification and OCR errors surface as `error`
fields inside their sections rather than failing the whole scan. fields inside their sections rather than failing the whole scan.
## Known gaps & future recommendations ## Known gaps & future recommendations
Tracked ones (see `plans/next-enhancements.md`): Tracked ones (see `plans/next-enhancements.md`):
- **Dataset thinness**: 2–16 photos/class caps both classifiers; every new real - **Dataset thinness**: 2–16 photos/class caps both classifiers; every new real
photo (especially non-studio, in-warehouse shots) matters. The harness above photo (especially non-studio, in-warehouse shots) matters. The harness above
already reports gallery (training) vs. held-out (validation) accuracy already reports gallery (training) vs. held-out (validation) accuracy
separately, and as of 2026-07-14 the Validation Set has 79 labeled images separately, and as of 2026-07-14 the Validation Set has 79 labeled images
(74 genuinely held out, 5 flagged trained-on — see above) — the first real (74 genuinely held out, 5 flagged trained-on — see above) — the first real
(non-zero) Validation Set numbers. (non-zero) Validation Set numbers.
Additional recommendations (not yet tasks — promote via `e`/`n` when wanted): Additional recommendations (not yet tasks — promote via `e`/`n` when wanted):
1. ~~Use `extracted_sku` in match ranking.~~ **Done** — `product-scan.ts`'s 1. ~~Use `extracted_sku` in match ranking.~~ **Done** — `product-scan.ts`'s
`classifyAndMatchProduct` already pins rank 1 to an exact `no_sku` match `classifyAndMatchProduct` already pins rank 1 to an exact `no_sku` match
(score forced to 1.0) before falling back to name similarity. (score forced to 1.0) before falling back to name similarity.
2. **Fuse DINOv2 and YOLO instead of primary/fallback** (e.g. agreement boosts 2. **Fuse DINOv2 and YOLO instead of primary/fallback** (e.g. agreement boosts
confidence; disagreement flags for review) — cheap, both already load. confidence; disagreement flags for review) — cheap, both already load.
3. **"Not a known product" handling**: DINOv2 always returns *some* class; add a 3. **"Not a known product" handling**: DINOv2 always returns *some* class; add a
minimum-similarity threshold below which the response says unknown rather minimum-similarity threshold below which the response says unknown rather
than confidently misclassifying a foreign package. than confidently misclassifying a foreign package.
4. **Pin the DINOv2 backbone offline** (vendor the weights or pre-bake the 4. **Pin the DINOv2 backbone offline** (vendor the weights or pre-bake the
torch.hub cache into the image) — startup currently depends on an internet torch.hub cache into the image) — startup currently depends on an internet
fetch on cold cache, bad for on-prem deploys. fetch on cold cache, bad for on-prem deploys.
5. **Batch/lot number extraction** — explicitly out of scope so far (plan §2 5. **Batch/lot number extraction** — explicitly out of scope so far (plan §2
note); if requested, follow the expiry-date regex-cascade pattern. note); if requested, follow the expiry-date regex-cascade pattern.
6. **Mobile**: no web mobile page by design (task 2.2 cancelled) — real mobile 6. **Mobile**: no web mobile page by design (task 2.2 cancelled) — real mobile
scanning should go through the Flutter app calling `POST /api/scan-pfm` scanning should go through the Flutter app calling `POST /api/scan-pfm`
(would need an authenticated `/api/v1` variant; the classic route has no auth). (would need an authenticated `/api/v1` variant; the classic route has no auth).
+138 -138
View File
@@ -1,138 +1,138 @@
# vLLM Service — Full Reference # vLLM Service — Full Reference
Detail split out of `../AGENTS.md` (2026-07-08, to keep that file under the Detail split out of `../AGENTS.md` (2026-07-08, to keep that file under the
Agents Settings Kit's 256-line threshold once the `e`/`n` workflow was appended Agents Settings Kit's 256-line threshold once the `e`/`n` workflow was appended
to it). `AGENTS.md` keeps the short version — architecture, quick start, the to it). `AGENTS.md` keeps the short version — architecture, quick start, the
env var table, file map — and links here for everything else. env var table, file map — and links here for everything else.
## Issue recording — naming and template ## Issue recording — naming and template
``` ```
issues/{NN}-{slug}.md issues/{NN}-{slug}.md
``` ```
| Part | Rule | Example | | Part | Rule | Example |
|------|------|---------| |------|------|---------|
| `{NN}` | Two-digit running number (`01`, `02`, …). Increment from the highest existing file. | `03` | | `{NN}` | Two-digit running number (`01`, `02`, …). Increment from the highest existing file. | `03` |
| `{slug}` | Lowercase kebab-case summary of the problem | `gpu-memory-startup-failure` | | `{slug}` | Lowercase kebab-case summary of the problem | `gpu-memory-startup-failure` |
Full example: `issues/04-gpu-memory-startup-failure.md` Full example: `issues/04-gpu-memory-startup-failure.md`
### File template ### File template
```markdown ```markdown
# Issue {NN}: {Short title} # Issue {NN}: {Short title}
## Problem ## Problem
What failed, with exact error message or symptom. What failed, with exact error message or symptom.
## Context ## Context
Environment, command run, relevant config (`.env`, `config/vllm_config.yaml`). Environment, command run, relevant config (`.env`, `config/vllm_config.yaml`).
## Solution ## Solution
What fixed it, or current workaround / open status. What fixed it, or current workaround / open status.
## References ## References
Links, related issue files, or AGENTS.md sections. Links, related issue files, or AGENTS.md sections.
``` ```
Check `issues/` for the next number: Check `issues/` for the next number:
```bash ```bash
ls issues/*.md 2>/dev/null | sort ls issues/*.md 2>/dev/null | sort
``` ```
## Client usage ## Client usage
After the server is running: After the server is running:
```bash ```bash
# CLI # CLI
uv run paddleocr doc_parser \ uv run paddleocr doc_parser \
--input https://paddle-model-ecology.bj.bcebos.com/paddlex/imgs/demo_image/paddleocr_vl_demo.png \ --input https://paddle-model-ecology.bj.bcebos.com/paddlex/imgs/demo_image/paddleocr_vl_demo.png \
--vl_rec_backend vllm-server \ --vl_rec_backend vllm-server \
--vl_rec_server_url http://localhost:8118/v1 --vl_rec_server_url http://localhost:8118/v1
``` ```
```python ```python
from paddleocr import PaddleOCRVL from paddleocr import PaddleOCRVL
pipeline = PaddleOCRVL( pipeline = PaddleOCRVL(
vl_rec_backend="vllm-server", vl_rec_backend="vllm-server",
vl_rec_server_url="http://127.0.0.1:8118/v1", vl_rec_server_url="http://127.0.0.1:8118/v1",
) )
output = pipeline.predict("path/to/image.png") output = pipeline.predict("path/to/image.png")
``` ```
Note: The full PaddleOCR-VL client should run in a **separate** environment if it needs PaddlePaddle GPU + Transformers. This repo is the isolated vLLM server only. Note: The full PaddleOCR-VL client should run in a **separate** environment if it needs PaddlePaddle GPU + Transformers. This repo is the isolated vLLM server only.
## Tuning vLLM ## Tuning vLLM
Edit `config/vllm_config.yaml`: Edit `config/vllm_config.yaml`:
```yaml ```yaml
gpu-memory-utilization: 0.8 gpu-memory-utilization: 0.8
max-num-seqs: 128 max-num-seqs: 128
``` ```
Reference: [PaddleOCR-VL vLLM parameter tuning](https://www.paddleocr.ai/latest/en/version3.x/pipeline_usage/PaddleOCR-VL.html#331-server-side-parameter-adjustment) Reference: [PaddleOCR-VL vLLM parameter tuning](https://www.paddleocr.ai/latest/en/version3.x/pipeline_usage/PaddleOCR-VL.html#331-server-side-parameter-adjustment)
## Troubleshooting ## Troubleshooting
See `issues/` for full write-ups. Quick pointers: See `issues/` for full write-ups. Quick pointers:
| Symptom | Issue file | | Symptom | Issue file |
|---------|------------| |---------|------------|
| `paddleocr install_genai_server_deps` / `No module named pip` | [01-genai-server-deps-pip-in-uv-venv.md](../issues/01-genai-server-deps-pip-in-uv-venv.md) | | `paddleocr install_genai_server_deps` / `No module named pip` | [01-genai-server-deps-pip-in-uv-venv.md](../issues/01-genai-server-deps-pip-in-uv-venv.md) |
| flash-attn wheel incompatible with Python version | [02-flash-attn-wheel-python-version-mismatch.md](../issues/02-flash-attn-wheel-python-version-mismatch.md) | | flash-attn wheel incompatible with Python version | [02-flash-attn-wheel-python-version-mismatch.md](../issues/02-flash-attn-wheel-python-version-mismatch.md) |
| `uv pip` targets wrong venv from another project | [03-active-virtual-env-from-other-project.md](../issues/03-active-virtual-env-from-other-project.md) | | `uv pip` targets wrong venv from another project | [03-active-virtual-env-from-other-project.md](../issues/03-active-virtual-env-from-other-project.md) |
| Free memory below `gpu-memory-utilization` on startup | [04-gpu-memory-startup-failure.md](../issues/04-gpu-memory-startup-failure.md) | | Free memory below `gpu-memory-utilization` on startup | [04-gpu-memory-startup-failure.md](../issues/04-gpu-memory-startup-failure.md) |
| `TokenizersBackend has no attribute all_special_tokens_extended` | [05-transformers-tokenizers-incompatibility.md](../issues/05-transformers-tokenizers-incompatibility.md) | | `TokenizersBackend has no attribute all_special_tokens_extended` | [05-transformers-tokenizers-incompatibility.md](../issues/05-transformers-tokenizers-incompatibility.md) |
| Extracted images not shown in Gradio demo (raw base64 in markdown) | [06-extracted-images-raw-base64-not-displayed.md](../issues/06-extracted-images-raw-base64-not-displayed.md) | | Extracted images not shown in Gradio demo (raw base64 in markdown) | [06-extracted-images-raw-base64-not-displayed.md](../issues/06-extracted-images-raw-base64-not-displayed.md) |
### flash-attn build failures ### flash-attn build failures
Install the prebuilt wheel after `uv sync` (see `scripts/install.sh`): Install the prebuilt wheel after `uv sync` (see `scripts/install.sh`):
```bash ```bash
FLASH_ATTN_WHEEL="https://github.com/mjun0812/flash-attention-prebuild-wheels/releases/download/v0.3.14/flash_attn-2.8.2+cu128torch2.8-cp312-cp312-linux_x86_64.whl" \ FLASH_ATTN_WHEEL="https://github.com/mjun0812/flash-attention-prebuild-wheels/releases/download/v0.3.14/flash_attn-2.8.2+cu128torch2.8-cp312-cp312-linux_x86_64.whl" \
./scripts/install.sh ./scripts/install.sh
``` ```
Pick the wheel matching your Python and CUDA versions from [flash-attention prebuild wheels](https://mjunya.com/flash-attention-prebuild-wheels/). Details: [02-flash-attn-wheel-python-version-mismatch.md](../issues/02-flash-attn-wheel-python-version-mismatch.md). Pick the wheel matching your Python and CUDA versions from [flash-attention prebuild wheels](https://mjunya.com/flash-attention-prebuild-wheels/). Details: [02-flash-attn-wheel-python-version-mismatch.md](../issues/02-flash-attn-wheel-python-version-mismatch.md).
Note: `paddleocr install_genai_server_deps` uses `pip` internally and is incompatible with uv-managed venvs. See [01-genai-server-deps-pip-in-uv-venv.md](../issues/01-genai-server-deps-pip-in-uv-venv.md). This repo installs the vLLM stack via `uv sync` + `uv pip`. Note: `paddleocr install_genai_server_deps` uses `pip` internally and is incompatible with uv-managed venvs. See [01-genai-server-deps-pip-in-uv-venv.md](../issues/01-genai-server-deps-pip-in-uv-venv.md). This repo installs the vLLM stack via `uv sync` + `uv pip`.
### `TokenizersBackend has no attribute all_special_tokens_extended` ### `TokenizersBackend has no attribute all_special_tokens_extended`
Pin transformers (already in `pyproject.toml`): Pin transformers (already in `pyproject.toml`):
```bash ```bash
uv pip install "transformers==4.57.6" uv pip install "transformers==4.57.6"
``` ```
See [05-transformers-tokenizers-incompatibility.md](../issues/05-transformers-tokenizers-incompatibility.md). See [05-transformers-tokenizers-incompatibility.md](../issues/05-transformers-tokenizers-incompatibility.md).
### Do not install `paddlepaddle-gpu` in this venv ### Do not install `paddlepaddle-gpu` in this venv
vLLM and PaddlePaddle GPU conflict. This server env uses `paddleocr[doc-parser]` without Paddle GPU. vLLM and PaddlePaddle GPU conflict. This server env uses `paddleocr[doc-parser]` without Paddle GPU.
### GPU memory on startup ### GPU memory on startup
If vLLM reports free memory below `gpu-memory-utilization`, either: If vLLM reports free memory below `gpu-memory-utilization`, either:
- Set `CUDA_VISIBLE_DEVICES` to a less-busy GPU - Set `CUDA_VISIBLE_DEVICES` to a less-busy GPU
- Lower `gpu-memory-utilization` in `config/vllm_config.yaml` (e.g. `0.75` or `0.7`) - Lower `gpu-memory-utilization` in `config/vllm_config.yaml` (e.g. `0.75` or `0.7`)
See [04-gpu-memory-startup-failure.md](../issues/04-gpu-memory-startup-failure.md). See [04-gpu-memory-startup-failure.md](../issues/04-gpu-memory-startup-failure.md).
### Health check ### Health check
```bash ```bash
curl -s http://localhost:8118/v1/models | jq . curl -s http://localhost:8118/v1/models | jq .
``` ```
## References ## References
- [PaddleOCR-VL usage tutorial](https://www.paddleocr.ai/latest/en/version3.x/pipeline_usage/PaddleOCR-VL.html) - [PaddleOCR-VL usage tutorial](https://www.paddleocr.ai/latest/en/version3.x/pipeline_usage/PaddleOCR-VL.html)
- [PaddleOCR genai_server FAQ](https://github.com/PaddlePaddle/PaddleOCR/discussions/16822) - [PaddleOCR genai_server FAQ](https://github.com/PaddlePaddle/PaddleOCR/discussions/16822)
+206 -206
View File
@@ -1,206 +1,206 @@
import json import json
import os import os
import pandas as pd import pandas as pd
from openpyxl.styles import Font, PatternFill, Alignment, Border, Side from openpyxl.styles import Font, PatternFill, Alignment, Border, Side
from openpyxl.utils import get_column_letter from openpyxl.utils import get_column_letter
def main(): def main():
jsonl_file = "/tmp/test_images_results.jsonl" jsonl_file = "/tmp/test_images_results.jsonl"
xlsx_file = "/tmp/test_images_report.xlsx" xlsx_file = "/tmp/test_images_report.xlsx"
if not os.path.exists(jsonl_file): if not os.path.exists(jsonl_file):
print(f"Error: JSONL file not found at {jsonl_file}") print(f"Error: JSONL file not found at {jsonl_file}")
return return
documents = [] documents = []
items = [] items = []
with open(jsonl_file, "r") as f: with open(jsonl_file, "r") as f:
for idx, line in enumerate(f): for idx, line in enumerate(f):
if not line.strip(): if not line.strip():
continue continue
try: try:
data = json.loads(line) data = json.loads(line)
except Exception as e: except Exception as e:
print(f"Skipping line due to parse error: {e}") print(f"Skipping line due to parse error: {e}")
continue continue
filename = data.get("filename", "N/A") filename = data.get("filename", "N/A")
status = data.get("status", "N/A") status = data.get("status", "N/A")
tilt = data.get("tilt", "N/A") tilt = data.get("tilt", "N/A")
unwarped = data.get("unwarped", "N/A") unwarped = data.get("unwarped", "N/A")
metadata = data.get("metadata", {}) metadata = data.get("metadata", {})
no_po = metadata.get("noPO", "N/A") no_po = metadata.get("noPO", "N/A")
no_so = metadata.get("noSO", "N/A") no_so = metadata.get("noSO", "N/A")
no_do = metadata.get("noDO", "N/A") no_do = metadata.get("noDO", "N/A")
tanggal = metadata.get("tanggal", "N/A") tanggal = metadata.get("tanggal", "N/A")
customer = metadata.get("customerInfo", "N/A") customer = metadata.get("customerInfo", "N/A")
store = metadata.get("orderUntuk", "N/A") store = metadata.get("orderUntuk", "N/A")
alamat = metadata.get("alamat", "N/A") alamat = metadata.get("alamat", "N/A")
plat = metadata.get("platTruk", "N/A") plat = metadata.get("platTruk", "N/A")
items_list = data.get("items", []) items_list = data.get("items", [])
# Add to document list # Add to document list
documents.append({ documents.append({
"No": idx + 1, "No": idx + 1,
"Filename": filename, "Filename": filename,
"Status": status, "Status": status,
"Tilt (Degrees)": tilt, "Tilt (Degrees)": tilt,
"Auto-Rotated/Unwarped": unwarped, "Auto-Rotated/Unwarped": unwarped,
"PO Number": no_po, "PO Number": no_po,
"SO Number": no_so, "SO Number": no_so,
"DO Number": no_do, "DO Number": no_do,
"Date": tanggal, "Date": tanggal,
"Customer": customer, "Customer": customer,
"Store Match": store, "Store Match": store,
"Alamat": alamat, "Alamat": alamat,
"Plat Nomor": plat, "Plat Nomor": plat,
"Items Count": len(items_list) "Items Count": len(items_list)
}) })
# Add items to items list # Add items to items list
for item in items_list: for item in items_list:
items.append({ items.append({
"Filename": filename, "Filename": filename,
"Kode Barang (SKU)": item.get("kodeBarang", "N/A"), "Kode Barang (SKU)": item.get("kodeBarang", "N/A"),
"Nama Barang": item.get("namaBarang", "N/A"), "Nama Barang": item.get("namaBarang", "N/A"),
"Banyak (Qty)": item.get("banyak", ""), "Banyak (Qty)": item.get("banyak", ""),
"Jumlah (Unit)": item.get("jumlah", "") "Jumlah (Unit)": item.get("jumlah", "")
}) })
df_docs = pd.DataFrame(documents) df_docs = pd.DataFrame(documents)
df_items = pd.DataFrame(items) df_items = pd.DataFrame(items)
# Style definitions # Style definitions
font_family = "Segoe UI" font_family = "Segoe UI"
header_font = Font(name=font_family, size=11, bold=True, color="FFFFFF") header_font = Font(name=font_family, size=11, bold=True, color="FFFFFF")
regular_font = Font(name=font_family, size=10) regular_font = Font(name=font_family, size=10)
bold_font = Font(name=font_family, size=10, bold=True) bold_font = Font(name=font_family, size=10, bold=True)
# Fill colors # Fill colors
header_fill = PatternFill(start_color="1F4E78", end_color="1F4E78", fill_type="solid") # Dark Blue header_fill = PatternFill(start_color="1F4E78", end_color="1F4E78", fill_type="solid") # Dark Blue
zebra_fill = PatternFill(start_color="F2F5F8", end_color="F2F5F8", fill_type="solid") # Very light blue-gray zebra_fill = PatternFill(start_color="F2F5F8", end_color="F2F5F8", fill_type="solid") # Very light blue-gray
success_fill = PatternFill(start_color="E2EFDA", end_color="E2EFDA", fill_type="solid") # Light green success_fill = PatternFill(start_color="E2EFDA", end_color="E2EFDA", fill_type="solid") # Light green
error_fill = PatternFill(start_color="FCE4D6", end_color="FCE4D6", fill_type="solid") # Light orange error_fill = PatternFill(start_color="FCE4D6", end_color="FCE4D6", fill_type="solid") # Light orange
# Alignments # Alignments
center_align = Alignment(horizontal="center", vertical="center") center_align = Alignment(horizontal="center", vertical="center")
left_align = Alignment(horizontal="left", vertical="center") left_align = Alignment(horizontal="left", vertical="center")
right_align = Alignment(horizontal="right", vertical="center") right_align = Alignment(horizontal="right", vertical="center")
# Borders # Borders
thin_side = Side(border_style="thin", color="D9D9D9") thin_side = Side(border_style="thin", color="D9D9D9")
thick_bottom = Side(border_style="medium", color="1F4E78") thick_bottom = Side(border_style="medium", color="1F4E78")
cell_border = Border(left=thin_side, right=thin_side, top=thin_side, bottom=thin_side) cell_border = Border(left=thin_side, right=thin_side, top=thin_side, bottom=thin_side)
with pd.ExcelWriter(xlsx_file, engine='openpyxl') as writer: with pd.ExcelWriter(xlsx_file, engine='openpyxl') as writer:
df_docs.to_excel(writer, sheet_name='Document Summary', index=False) df_docs.to_excel(writer, sheet_name='Document Summary', index=False)
df_items.to_excel(writer, sheet_name='Parsed Items', index=False) df_items.to_excel(writer, sheet_name='Parsed Items', index=False)
workbook = writer.book workbook = writer.book
# 1. Style Document Summary Sheet # 1. Style Document Summary Sheet
sheet1 = workbook['Document Summary'] sheet1 = workbook['Document Summary']
sheet1.views.sheetView[0].showGridLines = True sheet1.views.sheetView[0].showGridLines = True
# Style Header Row # Style Header Row
for col_idx in range(1, len(df_docs.columns) + 1): for col_idx in range(1, len(df_docs.columns) + 1):
cell = sheet1.cell(row=1, column=col_idx) cell = sheet1.cell(row=1, column=col_idx)
cell.font = header_font cell.font = header_font
cell.fill = header_fill cell.fill = header_fill
cell.alignment = center_align cell.alignment = center_align
cell.border = Border(left=thin_side, right=thin_side, top=thin_side, bottom=thick_bottom) cell.border = Border(left=thin_side, right=thin_side, top=thin_side, bottom=thick_bottom)
# Style Data Rows # Style Data Rows
for row_idx in range(2, len(df_docs) + 2): for row_idx in range(2, len(df_docs) + 2):
# Check status for color coding # Check status for color coding
status_val = sheet1.cell(row=row_idx, column=3).value status_val = sheet1.cell(row=row_idx, column=3).value
row_fill = success_fill if status_val == "Success" else (error_fill if status_val == "Failed" or status_val == "Error" else None) row_fill = success_fill if status_val == "Success" else (error_fill if status_val == "Failed" or status_val == "Error" else None)
# Apply Zebra stripe if no status color # Apply Zebra stripe if no status color
if not row_fill and row_idx % 2 == 0: if not row_fill and row_idx % 2 == 0:
row_fill = zebra_fill row_fill = zebra_fill
for col_idx in range(1, len(df_docs.columns) + 1): for col_idx in range(1, len(df_docs.columns) + 1):
cell = sheet1.cell(row=row_idx, column=col_idx) cell = sheet1.cell(row=row_idx, column=col_idx)
cell.font = regular_font cell.font = regular_font
cell.border = cell_border cell.border = cell_border
# Apply alignments based on column content # Apply alignments based on column content
if col_idx in [1, 3, 4, 5, 9, 13, 14]: # No, Status, Tilt, Auto-rotated, Date, Plat, Items Count if col_idx in [1, 3, 4, 5, 9, 13, 14]: # No, Status, Tilt, Auto-rotated, Date, Plat, Items Count
cell.alignment = center_align cell.alignment = center_align
else: else:
cell.alignment = left_align cell.alignment = left_align
if row_fill: if row_fill:
cell.fill = row_fill cell.fill = row_fill
# Format tilt with degree symbol # Format tilt with degree symbol
if col_idx == 4 and cell.value != "N/A" and cell.value is not None: if col_idx == 4 and cell.value != "N/A" and cell.value is not None:
try: try:
cell.value = float(cell.value) cell.value = float(cell.value)
cell.number_format = '0.00"°"' cell.number_format = '0.00"°"'
except ValueError: except ValueError:
pass pass
# Auto-adjust column width for Sheet 1 # Auto-adjust column width for Sheet 1
for col in sheet1.columns: for col in sheet1.columns:
max_len = 0 max_len = 0
for cell in col: for cell in col:
val_str = str(cell.value or '') val_str = str(cell.value or '')
# Exclude long text like Alamat from width sizing # Exclude long text like Alamat from width sizing
if cell.column in [12]: # Alamat if cell.column in [12]: # Alamat
max_len = max(max_len, min(len(val_str), 30)) max_len = max(max_len, min(len(val_str), 30))
else: else:
max_len = max(max_len, len(val_str)) max_len = max(max_len, len(val_str))
col_letter = get_column_letter(col[0].column) col_letter = get_column_letter(col[0].column)
sheet1.column_dimensions[col_letter].width = max(max_len + 3, 10) sheet1.column_dimensions[col_letter].width = max(max_len + 3, 10)
sheet1.row_dimensions[1].height = 25 sheet1.row_dimensions[1].height = 25
for r in range(2, len(df_docs) + 2): for r in range(2, len(df_docs) + 2):
sheet1.row_dimensions[r].height = 20 sheet1.row_dimensions[r].height = 20
# 2. Style Parsed Items Sheet # 2. Style Parsed Items Sheet
sheet2 = workbook['Parsed Items'] sheet2 = workbook['Parsed Items']
sheet2.views.sheetView[0].showGridLines = True sheet2.views.sheetView[0].showGridLines = True
# Style Header Row # Style Header Row
for col_idx in range(1, len(df_items.columns) + 1): for col_idx in range(1, len(df_items.columns) + 1):
cell = sheet2.cell(row=1, column=col_idx) cell = sheet2.cell(row=1, column=col_idx)
cell.font = header_font cell.font = header_font
cell.fill = header_fill cell.fill = header_fill
cell.alignment = center_align cell.alignment = center_align
cell.border = Border(left=thin_side, right=thin_side, top=thin_side, bottom=thick_bottom) cell.border = Border(left=thin_side, right=thin_side, top=thin_side, bottom=thick_bottom)
# Style Data Rows # Style Data Rows
for row_idx in range(2, len(df_items) + 2): for row_idx in range(2, len(df_items) + 2):
row_fill = zebra_fill if row_idx % 2 == 0 else None row_fill = zebra_fill if row_idx % 2 == 0 else None
for col_idx in range(1, len(df_items.columns) + 1): for col_idx in range(1, len(df_items.columns) + 1):
cell = sheet2.cell(row=row_idx, column=col_idx) cell = sheet2.cell(row=row_idx, column=col_idx)
cell.font = regular_font cell.font = regular_font
cell.border = cell_border cell.border = cell_border
# Alignments # Alignments
if col_idx in [2, 4, 5]: # SKU, Qty, Unit if col_idx in [2, 4, 5]: # SKU, Qty, Unit
cell.alignment = center_align cell.alignment = center_align
else: else:
cell.alignment = left_align cell.alignment = left_align
if row_fill: if row_fill:
cell.fill = row_fill cell.fill = row_fill
# Auto-adjust column width for Sheet 2 # Auto-adjust column width for Sheet 2
for col in sheet2.columns: for col in sheet2.columns:
max_len = max(len(str(cell.value or '')) for cell in col) max_len = max(len(str(cell.value or '')) for cell in col)
col_letter = get_column_letter(col[0].column) col_letter = get_column_letter(col[0].column)
sheet2.column_dimensions[col_letter].width = max(max_len + 3, 10) sheet2.column_dimensions[col_letter].width = max(max_len + 3, 10)
sheet2.row_dimensions[1].height = 25 sheet2.row_dimensions[1].height = 25
for r in range(2, len(df_items) + 2): for r in range(2, len(df_items) + 2):
sheet2.row_dimensions[r].height = 20 sheet2.row_dimensions[r].height = 20
print("Premium Excel report generated successfully!") print("Premium Excel report generated successfully!")
if __name__ == "__main__": if __name__ == "__main__":
main() main()
+151 -151
View File
@@ -1,151 +1,151 @@
events { events {
worker_connections 1024; worker_connections 1024;
} }
http { http {
include /etc/nginx/mime.types; include /etc/nginx/mime.types;
default_type application/octet-stream; default_type application/octet-stream;
sendfile on; sendfile on;
keepalive_timeout 65; keepalive_timeout 65;
map $http_x_forwarded_proto $proxy_x_forwarded_proto { map $http_x_forwarded_proto $proxy_x_forwarded_proto {
default $http_x_forwarded_proto; default $http_x_forwarded_proto;
'' $scheme; '' $scheme;
} }
server { server {
listen 80; listen 80;
server_name _; # accept any host — tunnel URLs, IPs, custom domains server_name _; # accept any host — tunnel URLs, IPs, custom domains
# Disable body size limit for large image/pdf base64 payloads # Disable body size limit for large image/pdf base64 payloads
client_max_body_size 0; client_max_body_size 0;
# Route to Next.js API Gateway (default root) # Route to Next.js API Gateway (default root)
location / { location / {
proxy_pass http://paddleocr-pfm-web-app:3000; proxy_pass http://paddleocr-pfm-web-app:3000;
proxy_http_version 1.1; proxy_http_version 1.1;
proxy_set_header Upgrade $http_upgrade; proxy_set_header Upgrade $http_upgrade;
proxy_set_header Connection "upgrade"; proxy_set_header Connection "upgrade";
proxy_set_header Host $http_host; proxy_set_header Host $http_host;
proxy_set_header X-Real-IP $remote_addr; proxy_set_header X-Real-IP $remote_addr;
proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for; proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for;
proxy_set_header X-Forwarded-Proto $proxy_x_forwarded_proto; proxy_set_header X-Forwarded-Proto $proxy_x_forwarded_proto;
proxy_read_timeout 300s; proxy_read_timeout 300s;
proxy_send_timeout 300s; proxy_send_timeout 300s;
} }
location /history { location /history {
proxy_pass http://paddleocr-pfm-web-app:3000; proxy_pass http://paddleocr-pfm-web-app:3000;
proxy_http_version 1.1; proxy_http_version 1.1;
proxy_set_header Host $http_host; proxy_set_header Host $http_host;
proxy_set_header X-Real-IP $remote_addr; proxy_set_header X-Real-IP $remote_addr;
proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for; proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for;
proxy_set_header X-Forwarded-Proto $proxy_x_forwarded_proto; proxy_set_header X-Forwarded-Proto $proxy_x_forwarded_proto;
proxy_read_timeout 300s; proxy_read_timeout 300s;
proxy_send_timeout 300s; proxy_send_timeout 300s;
} }
location /arena { location /arena {
proxy_pass http://paddleocr-pfm-web-app:3000; proxy_pass http://paddleocr-pfm-web-app:3000;
proxy_http_version 1.1; proxy_http_version 1.1;
proxy_set_header Host $http_host; proxy_set_header Host $http_host;
proxy_set_header X-Real-IP $remote_addr; proxy_set_header X-Real-IP $remote_addr;
proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for; proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for;
proxy_set_header X-Forwarded-Proto $proxy_x_forwarded_proto; proxy_set_header X-Forwarded-Proto $proxy_x_forwarded_proto;
proxy_read_timeout 300s; proxy_read_timeout 300s;
proxy_send_timeout 300s; proxy_send_timeout 300s;
} }
location /gpu { location /gpu {
proxy_pass http://paddleocr-pfm-web-app:3000; proxy_pass http://paddleocr-pfm-web-app:3000;
proxy_http_version 1.1; proxy_http_version 1.1;
proxy_set_header Host $http_host; proxy_set_header Host $http_host;
proxy_set_header X-Real-IP $remote_addr; proxy_set_header X-Real-IP $remote_addr;
proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for; proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for;
proxy_set_header X-Forwarded-Proto $proxy_x_forwarded_proto; proxy_set_header X-Forwarded-Proto $proxy_x_forwarded_proto;
proxy_read_timeout 300s; proxy_read_timeout 300s;
proxy_send_timeout 300s; proxy_send_timeout 300s;
} }
location /api { location /api {
proxy_pass http://paddleocr-pfm-web-app:3000; proxy_pass http://paddleocr-pfm-web-app:3000;
proxy_http_version 1.1; proxy_http_version 1.1;
proxy_set_header Host $http_host; proxy_set_header Host $http_host;
proxy_set_header X-Real-IP $remote_addr; proxy_set_header X-Real-IP $remote_addr;
proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for; proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for;
proxy_set_header X-Forwarded-Proto $proxy_x_forwarded_proto; proxy_set_header X-Forwarded-Proto $proxy_x_forwarded_proto;
proxy_read_timeout 300s; proxy_read_timeout 300s;
proxy_send_timeout 300s; proxy_send_timeout 300s;
} }
location /_next { location /_next {
proxy_pass http://paddleocr-pfm-web-app:3000/_next; proxy_pass http://paddleocr-pfm-web-app:3000/_next;
proxy_http_version 1.1; proxy_http_version 1.1;
proxy_set_header Upgrade $http_upgrade; proxy_set_header Upgrade $http_upgrade;
proxy_set_header Connection "upgrade"; proxy_set_header Connection "upgrade";
proxy_set_header Host $http_host; proxy_set_header Host $http_host;
proxy_set_header X-Real-IP $remote_addr; proxy_set_header X-Real-IP $remote_addr;
proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for; proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for;
proxy_set_header X-Forwarded-Proto $proxy_x_forwarded_proto; proxy_set_header X-Forwarded-Proto $proxy_x_forwarded_proto;
} }
# Route to Pipeline API # Route to Pipeline API
location /layout-parsing { location /layout-parsing {
proxy_pass http://paddleocr-pipeline-api:8090/layout-parsing; proxy_pass http://paddleocr-pipeline-api:8090/layout-parsing;
proxy_http_version 1.1; proxy_http_version 1.1;
proxy_set_header Host $http_host; proxy_set_header Host $http_host;
proxy_set_header X-Real-IP $remote_addr; proxy_set_header X-Real-IP $remote_addr;
proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for; proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for;
proxy_set_header X-Forwarded-Proto $proxy_x_forwarded_proto; proxy_set_header X-Forwarded-Proto $proxy_x_forwarded_proto;
proxy_read_timeout 300s; proxy_read_timeout 300s;
proxy_send_timeout 300s; proxy_send_timeout 300s;
} }
location /health { location /health {
proxy_pass http://paddleocr-pipeline-api:8090/health; proxy_pass http://paddleocr-pipeline-api:8090/health;
proxy_http_version 1.1; proxy_http_version 1.1;
proxy_set_header Host $http_host; proxy_set_header Host $http_host;
proxy_set_header X-Real-IP $remote_addr; proxy_set_header X-Real-IP $remote_addr;
proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for; proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for;
proxy_set_header X-Forwarded-Proto $proxy_x_forwarded_proto; proxy_set_header X-Forwarded-Proto $proxy_x_forwarded_proto;
} }
# Route to vLLM Server API (v1) # Route to vLLM Server API (v1)
location /v1 { location /v1 {
proxy_pass http://paddleocr-vllm-server:8118/v1; proxy_pass http://paddleocr-vllm-server:8118/v1;
proxy_http_version 1.1; proxy_http_version 1.1;
proxy_set_header Host $http_host; proxy_set_header Host $http_host;
proxy_set_header X-Real-IP $remote_addr; proxy_set_header X-Real-IP $remote_addr;
proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for; proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for;
proxy_set_header X-Forwarded-Proto $proxy_x_forwarded_proto; proxy_set_header X-Forwarded-Proto $proxy_x_forwarded_proto;
proxy_read_timeout 300s; proxy_read_timeout 300s;
proxy_send_timeout 300s; proxy_send_timeout 300s;
} }
} }
server { server {
listen 8001; listen 8001;
server_name _; server_name _;
client_max_body_size 0; client_max_body_size 0;
# Secure public endpoint — only allow API v1 surface # Secure public endpoint — only allow API v1 surface
location /api/v1/ { location /api/v1/ {
proxy_pass http://paddleocr-pfm-web-app:3000/api/v1/; proxy_pass http://paddleocr-pfm-web-app:3000/api/v1/;
proxy_http_version 1.1; proxy_http_version 1.1;
proxy_set_header Host $http_host; proxy_set_header Host $http_host;
proxy_set_header X-Real-IP $remote_addr; proxy_set_header X-Real-IP $remote_addr;
proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for; proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for;
proxy_set_header X-Forwarded-Proto $proxy_x_forwarded_proto; proxy_set_header X-Forwarded-Proto $proxy_x_forwarded_proto;
proxy_read_timeout 300s; proxy_read_timeout 300s;
proxy_send_timeout 300s; proxy_send_timeout 300s;
} }
# Deny everything else # Deny everything else
location / { location / {
return 404; return 404;
} }
} }
} }
+41 -41
View File
@@ -1,41 +1,41 @@
# See https://help.github.com/articles/ignoring-files/ for more about ignoring files. # See https://help.github.com/articles/ignoring-files/ for more about ignoring files.
# dependencies # dependencies
/node_modules /node_modules
/.pnp /.pnp
.pnp.* .pnp.*
.yarn/* .yarn/*
!.yarn/patches !.yarn/patches
!.yarn/plugins !.yarn/plugins
!.yarn/releases !.yarn/releases
!.yarn/versions !.yarn/versions
# testing # testing
/coverage /coverage
# next.js # next.js
/.next/ /.next/
/out/ /out/
# production # production
/build /build
# misc # misc
.DS_Store .DS_Store
*.pem *.pem
# debug # debug
npm-debug.log* npm-debug.log*
yarn-debug.log* yarn-debug.log*
yarn-error.log* yarn-error.log*
.pnpm-debug.log* .pnpm-debug.log*
# env files (can opt-in for committing if needed) # env files (can opt-in for committing if needed)
.env* .env*
# vercel # vercel
.vercel .vercel
# typescript # typescript
*.tsbuildinfo *.tsbuildinfo
next-env.d.ts next-env.d.ts
+5 -5
View File
@@ -1,5 +1,5 @@
<!-- BEGIN:nextjs-agent-rules --> <!-- BEGIN:nextjs-agent-rules -->
# This is NOT the Next.js you know # This is NOT the Next.js you know
This version has breaking changes — APIs, conventions, and file structure may all differ from your training data. Read the relevant guide in `node_modules/next/dist/docs/` before writing any code. Heed deprecation notices. This version has breaking changes — APIs, conventions, and file structure may all differ from your training data. Read the relevant guide in `node_modules/next/dist/docs/` before writing any code. Heed deprecation notices.
<!-- END:nextjs-agent-rules --> <!-- END:nextjs-agent-rules -->
+1 -1
View File
@@ -1 +1 @@
@AGENTS.md @AGENTS.md
+36 -36
View File
@@ -1,36 +1,36 @@
This is a [Next.js](https://nextjs.org) project bootstrapped with [`create-next-app`](https://nextjs.org/docs/app/api-reference/cli/create-next-app). This is a [Next.js](https://nextjs.org) project bootstrapped with [`create-next-app`](https://nextjs.org/docs/app/api-reference/cli/create-next-app).
## Getting Started ## Getting Started
First, run the development server: First, run the development server:
```bash ```bash
npm run dev npm run dev
# or # or
yarn dev yarn dev
# or # or
pnpm dev pnpm dev
# or # or
bun dev bun dev
``` ```
Open [http://localhost:3000](http://localhost:3000) with your browser to see the result. Open [http://localhost:3000](http://localhost:3000) with your browser to see the result.
You can start editing the page by modifying `app/page.tsx`. The page auto-updates as you edit the file. You can start editing the page by modifying `app/page.tsx`. The page auto-updates as you edit the file.
This project uses [`next/font`](https://nextjs.org/docs/app/building-your-application/optimizing/fonts) to automatically optimize and load [Geist](https://vercel.com/font), a new font family for Vercel. This project uses [`next/font`](https://nextjs.org/docs/app/building-your-application/optimizing/fonts) to automatically optimize and load [Geist](https://vercel.com/font), a new font family for Vercel.
## Learn More ## Learn More
To learn more about Next.js, take a look at the following resources: To learn more about Next.js, take a look at the following resources:
- [Next.js Documentation](https://nextjs.org/docs) - learn about Next.js features and API. - [Next.js Documentation](https://nextjs.org/docs) - learn about Next.js features and API.
- [Learn Next.js](https://nextjs.org/learn) - an interactive Next.js tutorial. - [Learn Next.js](https://nextjs.org/learn) - an interactive Next.js tutorial.
You can check out [the Next.js GitHub repository](https://github.com/vercel/next.js) - your feedback and contributions are welcome! You can check out [the Next.js GitHub repository](https://github.com/vercel/next.js) - your feedback and contributions are welcome!
## Deploy on Vercel ## Deploy on Vercel
The easiest way to deploy your Next.js app is to use the [Vercel Platform](https://vercel.com/new?utm_medium=default-template&filter=next.js&utm_source=create-next-app&utm_campaign=create-next-app-readme) from the creators of Next.js. The easiest way to deploy your Next.js app is to use the [Vercel Platform](https://vercel.com/new?utm_medium=default-template&filter=next.js&utm_source=create-next-app&utm_campaign=create-next-app-readme) from the creators of Next.js.
Check out our [Next.js deployment documentation](https://nextjs.org/docs/app/building-your-application/deploying) for more details. Check out our [Next.js deployment documentation](https://nextjs.org/docs/app/building-your-application/deploying) for more details.
+103 -103
View File
@@ -1,103 +1,103 @@
const puppeteer = require('puppeteer'); const puppeteer = require('puppeteer');
const fs = require('fs'); const fs = require('fs');
(async () => { (async () => {
const browser = await puppeteer.launch({ const browser = await puppeteer.launch({
headless: "new", headless: "new",
args: ['--no-sandbox', '--disable-setuid-sandbox'] args: ['--no-sandbox', '--disable-setuid-sandbox']
}); });
const page = await browser.newPage(); const page = await browser.newPage();
await page.setViewport({ width: 1280, height: 800 }); await page.setViewport({ width: 1280, height: 800 });
console.log("Navigating to login page..."); console.log("Navigating to login page...");
await page.goto('http://localhost:3000/admin/master-data', { waitUntil: 'networkidle2' }); await page.goto('http://localhost:3000/admin/master-data', { waitUntil: 'networkidle2' });
console.log("Filling login form..."); console.log("Filling login form...");
await page.type('input[type="text"]', 'admin'); await page.type('input[type="text"]', 'admin');
await page.type('input[type="password"]', 'password'); await page.type('input[type="password"]', 'password');
await page.screenshot({ path: 'test_step1_login_filled.png' }); await page.screenshot({ path: 'test_step1_login_filled.png' });
console.log("Clicking login..."); console.log("Clicking login...");
await Promise.all([ await Promise.all([
page.click('button[type="submit"]'), page.click('button[type="submit"]'),
page.waitForNavigation({ waitUntil: 'networkidle0' }).catch(e => console.log('Navigation wait timeout/catch')) page.waitForNavigation({ waitUntil: 'networkidle0' }).catch(e => console.log('Navigation wait timeout/catch'))
]); ]);
// Wait a bit for React to render the stores table // Wait a bit for React to render the stores table
await new Promise(resolve => setTimeout(resolve, 2000)); await new Promise(resolve => setTimeout(resolve, 2000));
await page.screenshot({ path: 'test_step2_after_login.png' }); await page.screenshot({ path: 'test_step2_after_login.png' });
// Add store // Add store
console.log("Clicking Add Store..."); console.log("Clicking Add Store...");
await page.evaluate(() => { await page.evaluate(() => {
const btns = Array.from(document.querySelectorAll('button')); const btns = Array.from(document.querySelectorAll('button'));
const addBtn = btns.find(b => b.textContent.includes('Add Store')); const addBtn = btns.find(b => b.textContent.includes('Add Store'));
if (addBtn) addBtn.click(); if (addBtn) addBtn.click();
}); });
await new Promise(resolve => setTimeout(resolve, 500)); await new Promise(resolve => setTimeout(resolve, 500));
console.log("Filling new store form..."); console.log("Filling new store form...");
const inputs = await page.$$('input[placeholder]'); const inputs = await page.$$('input[placeholder]');
for (const input of inputs) { for (const input of inputs) {
const placeholder = await input.evaluate(el => el.getAttribute('placeholder')); const placeholder = await input.evaluate(el => el.getAttribute('placeholder'));
if (placeholder === 'Kode Toko') await input.type('TEST99'); if (placeholder === 'Kode Toko') await input.type('TEST99');
if (placeholder === 'Nama Toko') await input.type('Toko Test 99'); if (placeholder === 'Nama Toko') await input.type('Toko Test 99');
if (placeholder === 'Alamat') await input.type('Alamat Test'); if (placeholder === 'Alamat') await input.type('Alamat Test');
} }
await page.screenshot({ path: 'test_step3_store_filled.png' }); await page.screenshot({ path: 'test_step3_store_filled.png' });
console.log("Saving store..."); console.log("Saving store...");
await page.evaluate(() => { await page.evaluate(() => {
const btns = Array.from(document.querySelectorAll('button')); const btns = Array.from(document.querySelectorAll('button'));
const saveBtn = btns.find(b => b.textContent === 'Save'); const saveBtn = btns.find(b => b.textContent === 'Save');
if (saveBtn) saveBtn.click(); if (saveBtn) saveBtn.click();
}); });
await new Promise(resolve => setTimeout(resolve, 2000)); await new Promise(resolve => setTimeout(resolve, 2000));
await page.screenshot({ path: 'test_step4_store_saved.png' }); await page.screenshot({ path: 'test_step4_store_saved.png' });
// Switch to SKUs tab // Switch to SKUs tab
console.log("Switching to SKUs tab..."); console.log("Switching to SKUs tab...");
await page.evaluate(() => { await page.evaluate(() => {
const btns = Array.from(document.querySelectorAll('button')); const btns = Array.from(document.querySelectorAll('button'));
const skuBtn = btns.find(b => b.textContent === 'SKUs'); const skuBtn = btns.find(b => b.textContent === 'SKUs');
if (skuBtn) skuBtn.click(); if (skuBtn) skuBtn.click();
}); });
await new Promise(resolve => setTimeout(resolve, 1000)); await new Promise(resolve => setTimeout(resolve, 1000));
await page.screenshot({ path: 'test_step5_skus_tab.png' }); await page.screenshot({ path: 'test_step5_skus_tab.png' });
console.log("Clicking Add SKU..."); console.log("Clicking Add SKU...");
await page.evaluate(() => { await page.evaluate(() => {
const btns = Array.from(document.querySelectorAll('button')); const btns = Array.from(document.querySelectorAll('button'));
const addBtn = btns.find(b => b.textContent.includes('Add SKU')); const addBtn = btns.find(b => b.textContent.includes('Add SKU'));
if (addBtn) addBtn.click(); if (addBtn) addBtn.click();
}); });
await new Promise(resolve => setTimeout(resolve, 500)); await new Promise(resolve => setTimeout(resolve, 500));
console.log("Filling new SKU form..."); console.log("Filling new SKU form...");
const skuInputs = await page.$$('input[placeholder]'); const skuInputs = await page.$$('input[placeholder]');
for (const input of skuInputs) { for (const input of skuInputs) {
const placeholder = await input.evaluate(el => el.getAttribute('placeholder')); const placeholder = await input.evaluate(el => el.getAttribute('placeholder'));
if (placeholder === 'Kode Item') await input.type('SKU99'); if (placeholder === 'Kode Item') await input.type('SKU99');
if (placeholder === 'Nama Item') await input.type('Item 99'); if (placeholder === 'Nama Item') await input.type('Item 99');
if (placeholder === 'Barcode') await input.type('12345'); if (placeholder === 'Barcode') await input.type('12345');
if (placeholder === 'Jenis Outer (e.g. DUS)') await input.type('DUS'); if (placeholder === 'Jenis Outer (e.g. DUS)') await input.type('DUS');
} }
await page.screenshot({ path: 'test_step6_sku_filled.png' }); await page.screenshot({ path: 'test_step6_sku_filled.png' });
console.log("Saving SKU..."); console.log("Saving SKU...");
await page.evaluate(() => { await page.evaluate(() => {
const btns = Array.from(document.querySelectorAll('button')); const btns = Array.from(document.querySelectorAll('button'));
const saveBtn = btns.find(b => b.textContent === 'Save'); const saveBtn = btns.find(b => b.textContent === 'Save');
if (saveBtn) saveBtn.click(); if (saveBtn) saveBtn.click();
}); });
await new Promise(resolve => setTimeout(resolve, 2000)); await new Promise(resolve => setTimeout(resolve, 2000));
await page.screenshot({ path: 'test_step7_sku_saved.png' }); await page.screenshot({ path: 'test_step7_sku_saved.png' });
console.log("Done! Screenshots saved."); console.log("Done! Screenshots saved.");
await browser.close(); await browser.close();
})(); })();
+18 -18
View File
@@ -1,18 +1,18 @@
import { defineConfig, globalIgnores } from "eslint/config"; import { defineConfig, globalIgnores } from "eslint/config";
import nextVitals from "eslint-config-next/core-web-vitals"; import nextVitals from "eslint-config-next/core-web-vitals";
import nextTs from "eslint-config-next/typescript"; import nextTs from "eslint-config-next/typescript";
const eslintConfig = defineConfig([ const eslintConfig = defineConfig([
...nextVitals, ...nextVitals,
...nextTs, ...nextTs,
// Override default ignores of eslint-config-next. // Override default ignores of eslint-config-next.
globalIgnores([ globalIgnores([
// Default ignores of eslint-config-next: // Default ignores of eslint-config-next:
".next/**", ".next/**",
"out/**", "out/**",
"build/**", "build/**",
"next-env.d.ts", "next-env.d.ts",
]), ]),
]); ]);
export default eslintConfig; export default eslintConfig;
+114 -114
View File
@@ -1,114 +1,114 @@
const fs = require('fs'); const fs = require('fs');
const path = require('path'); const path = require('path');
const { Client } = require('pg'); const { Client } = require('pg');
async function main() { async function main() {
console.log('=== STARTING SKU MASTER TSV IMPORT ==='); console.log('=== STARTING SKU MASTER TSV IMPORT ===');
const client = new Client({ const client = new Client({
host: 'paddleocr-db', host: 'paddleocr-db',
port: 5432, port: 5432,
user: 'postgres', user: 'postgres',
password: 'postgres', password: 'postgres',
database: 'dopfm' database: 'dopfm'
}); });
try { try {
await client.connect(); await client.connect();
console.log('Connected to database.'); console.log('Connected to database.');
// 1. Alter table to add new packaging columns if they don't exist // 1. Alter table to add new packaging columns if they don't exist
console.log('Ensuring table schema has new packaging columns...'); console.log('Ensuring table schema has new packaging columns...');
await client.query(` await client.query(`
ALTER TABLE sku_master ADD COLUMN IF NOT EXISTS standar_jumlah VARCHAR(50); ALTER TABLE sku_master ADD COLUMN IF NOT EXISTS standar_jumlah VARCHAR(50);
ALTER TABLE sku_master ADD COLUMN IF NOT EXISTS berat_kemasan NUMERIC(10, 3); ALTER TABLE sku_master ADD COLUMN IF NOT EXISTS berat_kemasan NUMERIC(10, 3);
ALTER TABLE sku_master ADD COLUMN IF NOT EXISTS isi_outer_kg NUMERIC(10, 3); ALTER TABLE sku_master ADD COLUMN IF NOT EXISTS isi_outer_kg NUMERIC(10, 3);
ALTER TABLE sku_master ADD COLUMN IF NOT EXISTS isi_outer_pac INTEGER; ALTER TABLE sku_master ADD COLUMN IF NOT EXISTS isi_outer_pac INTEGER;
ALTER TABLE sku_master ADD COLUMN IF NOT EXISTS jenis_outer VARCHAR(50); ALTER TABLE sku_master ADD COLUMN IF NOT EXISTS jenis_outer VARCHAR(50);
`); `);
console.log('Table schema verified/updated.'); console.log('Table schema verified/updated.');
// 2. Truncate old data // 2. Truncate old data
console.log('Clearing old SKU master data...'); console.log('Clearing old SKU master data...');
await client.query('TRUNCATE TABLE sku_master RESTART IDENTITY CASCADE'); await client.query('TRUNCATE TABLE sku_master RESTART IDENTITY CASCADE');
console.log('Old SKU master data cleared.'); console.log('Old SKU master data cleared.');
// 3. Read and parse TSV file // 3. Read and parse TSV file
const tsvPath = path.join(__dirname, 'sku_master.tsv'); const tsvPath = path.join(__dirname, 'sku_master.tsv');
if (!fs.existsSync(tsvPath)) { if (!fs.existsSync(tsvPath)) {
throw new Error(`File not found at ${tsvPath}`); throw new Error(`File not found at ${tsvPath}`);
} }
const tsvContent = fs.readFileSync(tsvPath, 'utf8'); const tsvContent = fs.readFileSync(tsvPath, 'utf8');
const lines = tsvContent.split(/\r?\n/); const lines = tsvContent.split(/\r?\n/);
let insertCount = 0; let insertCount = 0;
let skipCount = 0; let skipCount = 0;
console.log(`Parsing ${lines.length} lines from TSV...`); console.log(`Parsing ${lines.length} lines from TSV...`);
// We start from line 5 (0-indexed 4 is the header row, lines before are title headers) // We start from line 5 (0-indexed 4 is the header row, lines before are title headers)
for (let i = 5; i < lines.length; i++) { for (let i = 5; i < lines.length; i++) {
const line = lines[i].trim(); const line = lines[i].trim();
if (!line) continue; if (!line) continue;
const cols = line.split('\t').map(c => c.trim()); const cols = line.split('\t').map(c => c.trim());
if (cols.length < 3) { if (cols.length < 3) {
skipCount++; skipCount++;
continue; continue;
} }
const noSku = cols[1]; const noSku = cols[1];
const namaItem = cols[2]; const namaItem = cols[2];
// Verify SKU code format (must be standard 8-digit) // Verify SKU code format (must be standard 8-digit)
if (!noSku || !/^\d{8}$/.test(noSku)) { if (!noSku || !/^\d{8}$/.test(noSku)) {
skipCount++; skipCount++;
continue; continue;
} }
const standarJumlah = cols[3] || null; const standarJumlah = cols[3] || null;
// Parse numeric columns // Parse numeric columns
const beratKemasan = cols[4] ? parseFloat(cols[4].replace(',', '.')) : null; const beratKemasan = cols[4] ? parseFloat(cols[4].replace(',', '.')) : null;
const isiOuterKg = cols[5] ? parseFloat(cols[5].replace(',', '.')) : null; const isiOuterKg = cols[5] ? parseFloat(cols[5].replace(',', '.')) : null;
const isiOuterPac = cols[6] ? parseInt(cols[6], 10) : null; const isiOuterPac = cols[6] ? parseInt(cols[6], 10) : null;
const jenisOuter = cols[7] || null; const jenisOuter = cols[7] || null;
await client.query(` await client.query(`
INSERT INTO sku_master ( INSERT INTO sku_master (
no_sku, nama_item, standar_jumlah, berat_kemasan, isi_outer_kg, isi_outer_pac, jenis_outer no_sku, nama_item, standar_jumlah, berat_kemasan, isi_outer_kg, isi_outer_pac, jenis_outer
) VALUES ($1, $2, $3, $4, $5, $6, $7) ) VALUES ($1, $2, $3, $4, $5, $6, $7)
ON CONFLICT (no_sku) DO UPDATE SET ON CONFLICT (no_sku) DO UPDATE SET
nama_item = EXCLUDED.nama_item, nama_item = EXCLUDED.nama_item,
standar_jumlah = EXCLUDED.standar_jumlah, standar_jumlah = EXCLUDED.standar_jumlah,
berat_kemasan = EXCLUDED.berat_kemasan, berat_kemasan = EXCLUDED.berat_kemasan,
isi_outer_kg = EXCLUDED.isi_outer_kg, isi_outer_kg = EXCLUDED.isi_outer_kg,
isi_outer_pac = EXCLUDED.isi_outer_pac, isi_outer_pac = EXCLUDED.isi_outer_pac,
jenis_outer = EXCLUDED.jenis_outer jenis_outer = EXCLUDED.jenis_outer
`, [ `, [
noSku, noSku,
namaItem, namaItem,
standarJumlah, standarJumlah,
isNaN(beratKemasan) ? null : beratKemasan, isNaN(beratKemasan) ? null : beratKemasan,
isNaN(isiOuterKg) ? null : isiOuterKg, isNaN(isiOuterKg) ? null : isiOuterKg,
isNaN(isiOuterPac) ? null : isiOuterPac, isNaN(isiOuterPac) ? null : isiOuterPac,
jenisOuter jenisOuter
]); ]);
insertCount++; insertCount++;
} }
console.log(`\nImport Completed Successfully:`); console.log(`\nImport Completed Successfully:`);
console.log(`- Inserted/Updated: ${insertCount} SKU records`); console.log(`- Inserted/Updated: ${insertCount} SKU records`);
console.log(`- Skipped (headers/invalid): ${skipCount} lines`); console.log(`- Skipped (headers/invalid): ${skipCount} lines`);
} catch (err) { } catch (err) {
console.error('Import process failed:', err); console.error('Import process failed:', err);
} finally { } finally {
await client.end(); await client.end();
console.log('Database connection closed.'); console.log('Database connection closed.');
} }
} }
main(); main();
+299 -299
View File
@@ -1,299 +1,299 @@
const { Client } = require("pg"); const { Client } = require("pg");
function cleanFinalValue(val, preserveNewlines = false) { function cleanFinalValue(val, preserveNewlines = false) {
if (!val) return "Not Found"; if (!val) return "Not Found";
const cleaned = val.replace(/<[^>]*>/g, ""); const cleaned = val.replace(/<[^>]*>/g, "");
if (preserveNewlines) { if (preserveNewlines) {
return cleaned.split("\n").map(line => line.trim()).filter(Boolean).join("\n") || "Not Found"; return cleaned.split("\n").map(line => line.trim()).filter(Boolean).join("\n") || "Not Found";
} else { } else {
return cleaned.replace(/\s+/g, " ").trim() || "Not Found"; return cleaned.replace(/\s+/g, " ").trim() || "Not Found";
} }
} }
function parseDOMetadata(markdown) { function parseDOMetadata(markdown) {
const metadata = { const metadata = {
vendorInfo: "Not Found", vendorInfo: "Not Found",
customerInfo: "Not Found", customerInfo: "Not Found",
tanggal: "Not Found", tanggal: "Not Found",
noSO: "Not Found", noSO: "Not Found",
noDO: "Not Found", noDO: "Not Found",
noPO: "Not Found", noPO: "Not Found",
items: [] items: []
}; };
if (!markdown) return metadata; if (!markdown) return metadata;
const cleanMarkdown = markdown const cleanMarkdown = markdown
.replace(/<\/tr>/gi, "\n") .replace(/<\/tr>/gi, "\n")
.replace(/<br\s*\/?>/gi, "\n") .replace(/<br\s*\/?>/gi, "\n")
.replace(/<\/p>/gi, "\n") .replace(/<\/p>/gi, "\n")
.replace(/<[^>]*>/g, " "); .replace(/<[^>]*>/g, " ");
const lines = cleanMarkdown.split("\n").map(l => l.trim()).filter(Boolean); const lines = cleanMarkdown.split("\n").map(l => l.trim()).filter(Boolean);
// Vendor Info // Vendor Info
const vendorStop = /(?:no\.?\s*(?:so|do|po)|tanggal|date|Kepada|Yth|Customer|Deliver|Order\s+Untuk|#|\d{2}:\d{2}:\d{2})/i; const vendorStop = /(?:no\.?\s*(?:so|do|po)|tanggal|date|Kepada|Yth|Customer|Deliver|Order\s+Untuk|#|\d{2}:\d{2}:\d{2})/i;
const vendorStartIndex = lines.findIndex(line => const vendorStartIndex = lines.findIndex(line =>
/PT\./i.test(line) && !/(?:Kepada|Yth|Customer|Deliver|Order\s+Untuk|Alamat|no\.?\s*(?:so|do|po)|tanggal|date)/i.test(line) /PT\./i.test(line) && !/(?:Kepada|Yth|Customer|Deliver|Order\s+Untuk|Alamat|no\.?\s*(?:so|do|po)|tanggal|date)/i.test(line)
); );
if (vendorStartIndex !== -1) { if (vendorStartIndex !== -1) {
const vendorLines = [lines[vendorStartIndex]]; const vendorLines = [lines[vendorStartIndex]];
for (let i = vendorStartIndex + 1; i < Math.min(lines.length, vendorStartIndex + 4); i++) { for (let i = vendorStartIndex + 1; i < Math.min(lines.length, vendorStartIndex + 4); i++) {
if (vendorStop.test(lines[i])) break; if (vendorStop.test(lines[i])) break;
vendorLines.push(lines[i]); vendorLines.push(lines[i]);
} }
metadata.vendorInfo = vendorLines.join("\n"); metadata.vendorInfo = vendorLines.join("\n");
} else { } else {
const vendorMatch = cleanMarkdown.match(/(PT\.\s*CHAROEN[^\n]*)/i) || cleanMarkdown.match(/(PT\.[^\n]+)/i); const vendorMatch = cleanMarkdown.match(/(PT\.\s*CHAROEN[^\n]*)/i) || cleanMarkdown.match(/(PT\.[^\n]+)/i);
if (vendorMatch) metadata.vendorInfo = vendorMatch[1].trim(); if (vendorMatch) metadata.vendorInfo = vendorMatch[1].trim();
} }
// Customer Info // Customer Info
const customerStop = /(?:no\.?\s*(?:so|do|po)|tanggal|date|#|\d{2}:\d{2}:\d{2})/i; const customerStop = /(?:no\.?\s*(?:so|do|po)|tanggal|date|#|\d{2}:\d{2}:\d{2})/i;
let customerStartIndex = lines.findIndex(line => let customerStartIndex = lines.findIndex(line =>
/(?:Kepada Yth|Yth|Customer|Deliver To)\s*[:\-]/i.test(line) || /PT\.\s*PRIMAFOOD/i.test(line) /(?:Kepada Yth|Yth|Customer|Deliver To)\s*[:\-]/i.test(line) || /PT\.\s*PRIMAFOOD/i.test(line)
); );
if (customerStartIndex === -1) { if (customerStartIndex === -1) {
const ptIndices = lines.map((l, idx) => l.toUpperCase().includes("PT.") ? idx : -1).filter(idx => idx !== -1); const ptIndices = lines.map((l, idx) => l.toUpperCase().includes("PT.") ? idx : -1).filter(idx => idx !== -1);
const secondaryIndices = ptIndices.filter(idx => idx !== vendorStartIndex); const secondaryIndices = ptIndices.filter(idx => idx !== vendorStartIndex);
if (secondaryIndices.length > 0) { if (secondaryIndices.length > 0) {
customerStartIndex = secondaryIndices[0]; customerStartIndex = secondaryIndices[0];
} }
} }
if (customerStartIndex !== -1) { if (customerStartIndex !== -1) {
const customerLines = [lines[customerStartIndex]]; const customerLines = [lines[customerStartIndex]];
for (let i = customerStartIndex + 1; i < Math.min(lines.length, customerStartIndex + 4); i++) { for (let i = customerStartIndex + 1; i < Math.min(lines.length, customerStartIndex + 4); i++) {
if (customerStop.test(lines[i])) break; if (customerStop.test(lines[i])) break;
customerLines.push(lines[i]); customerLines.push(lines[i]);
} }
metadata.customerInfo = customerLines.join("\n"); metadata.customerInfo = customerLines.join("\n");
} else { } else {
const customerMatch = cleanMarkdown.match(/(?:Kepada Yth|Yth|Customer|Deliver To)[ \t]*[:\-][ \t]*([^\n]+)/i) || cleanMarkdown.match(/(PT\.[ \t]*PRIMAFOOD[^\n]*)/i); const customerMatch = cleanMarkdown.match(/(?:Kepada Yth|Yth|Customer|Deliver To)[ \t]*[:\-][ \t]*([^\n]+)/i) || cleanMarkdown.match(/(PT\.[ \t]*PRIMAFOOD[^\n]*)/i);
if (customerMatch) metadata.customerInfo = customerMatch[1].trim(); if (customerMatch) metadata.customerInfo = customerMatch[1].trim();
} }
// Direct matches // Direct matches
const tanggalMatch = cleanMarkdown.match(/Tanggal[ \t]*[:\-][ \t]*([^\n]+)/i) || cleanMarkdown.match(/(?:Date|D\.O\.[ \t]*Date)[ \t]*[:\- \t]*([\d\-\/A-Za-z \t]+)/i); const tanggalMatch = cleanMarkdown.match(/Tanggal[ \t]*[:\-][ \t]*([^\n]+)/i) || cleanMarkdown.match(/(?:Date|D\.O\.[ \t]*Date)[ \t]*[:\- \t]*([\d\-\/A-Za-z \t]+)/i);
if (tanggalMatch) metadata.tanggal = tanggalMatch[1].trim(); if (tanggalMatch) metadata.tanggal = tanggalMatch[1].trim();
const soMatch = cleanMarkdown.match(/(?:No\.?[ \t]*SO|SO[ \t]*No\.?)[ \t]*[:\-][ \t]*([A-Z0-9\-]+)/i); const soMatch = cleanMarkdown.match(/(?:No\.?[ \t]*SO|SO[ \t]*No\.?)[ \t]*[:\-][ \t]*([A-Z0-9\-]+)/i);
if (soMatch) metadata.noSO = soMatch[1].trim(); if (soMatch) metadata.noSO = soMatch[1].trim();
const doMatch = cleanMarkdown.match(/(?:No\.?[ \t]*DO|Delivery Order[ \t]*No|D\.O\.[ \t]*No|Order[ \t]*No)[ \t]*[:\- \t]*([A-Z0-9\-]+)/i); const doMatch = cleanMarkdown.match(/(?:No\.?[ \t]*DO|Delivery Order[ \t]*No|D\.O\.[ \t]*No|Order[ \t]*No)[ \t]*[:\- \t]*([A-Z0-9\-]+)/i);
if (doMatch) metadata.noDO = doMatch[1].trim(); if (doMatch) metadata.noDO = doMatch[1].trim();
const poMatch = cleanMarkdown.match(/(?:No\.?[ \t]*PO|PO[ \t]*No\.?)[ \t]*[:\-][ \t]*([A-Z0-9\-\/]+)/i); const poMatch = cleanMarkdown.match(/(?:No\.?[ \t]*PO|PO[ \t]*No\.?)[ \t]*[:\-][ \t]*([A-Z0-9\-\/]+)/i);
if (poMatch) metadata.noPO = poMatch[1].trim(); if (poMatch) metadata.noPO = poMatch[1].trim();
// Fallback block/sequential alignment if any of the metadata values are not found // Fallback block/sequential alignment if any of the metadata values are not found
if ( if (
metadata.tanggal === "Not Found" || !metadata.tanggal || metadata.tanggal === "Not Found" || !metadata.tanggal ||
metadata.noSO === "Not Found" || !metadata.noSO || metadata.noSO === "Not Found" || !metadata.noSO ||
metadata.noDO === "Not Found" || !metadata.noDO || metadata.noDO === "Not Found" || !metadata.noDO ||
metadata.noPO === "Not Found" || !metadata.noPO metadata.noPO === "Not Found" || !metadata.noPO
) { ) {
const idxTanggal = lines.findIndex(l => /^Tanggal\s*[:\-]?\s*$/i.test(l)); const idxTanggal = lines.findIndex(l => /^Tanggal\s*[:\-]?\s*$/i.test(l));
const idxSO = lines.findIndex(l => /^No\.?\s*SO\s*[:\-]?\s*$/i.test(l)); const idxSO = lines.findIndex(l => /^No\.?\s*SO\s*[:\-]?\s*$/i.test(l));
const idxDO = lines.findIndex(l => /^No\.?\s*DO\s*[:\-]?\s*$/i.test(l)); const idxDO = lines.findIndex(l => /^No\.?\s*DO\s*[:\-]?\s*$/i.test(l));
const idxPO = lines.findIndex(l => /^No\.?\s*PO\s*[:\-]?\s*$/i.test(l)); const idxPO = lines.findIndex(l => /^No\.?\s*PO\s*[:\-]?\s*$/i.test(l));
if (idxTanggal !== -1 || idxSO !== -1 || idxDO !== -1 || idxPO !== -1) { if (idxTanggal !== -1 || idxSO !== -1 || idxDO !== -1 || idxPO !== -1) {
const indices = [idxTanggal, idxSO, idxDO, idxPO].filter(idx => idx !== -1); const indices = [idxTanggal, idxSO, idxDO, idxPO].filter(idx => idx !== -1);
const minIndex = Math.min(...indices); const minIndex = Math.min(...indices);
const maxIndex = Math.max(...indices); const maxIndex = Math.max(...indices);
if (maxIndex - minIndex < 8) { if (maxIndex - minIndex < 8) {
const candidateLines = lines.slice(maxIndex + 1, maxIndex + 12); const candidateLines = lines.slice(maxIndex + 1, maxIndex + 12);
if (metadata.tanggal === "Not Found" || !metadata.tanggal) { if (metadata.tanggal === "Not Found" || !metadata.tanggal) {
const dateRegex = /\b\d{1,2}\s+(?:Jan|Feb|Mar|Apr|May|Jun|Jul|Aug|Sep|Oct|Nov|Dec)[a-z]*\s+\d{4}\b/i; const dateRegex = /\b\d{1,2}\s+(?:Jan|Feb|Mar|Apr|May|Jun|Jul|Aug|Sep|Oct|Nov|Dec)[a-z]*\s+\d{4}\b/i;
for (const line of candidateLines) { for (const line of candidateLines) {
const m = line.match(dateRegex); const m = line.match(dateRegex);
if (m) { if (m) {
metadata.tanggal = m[0]; metadata.tanggal = m[0];
break; break;
} }
} }
} }
const tenDigitNumbers = []; const tenDigitNumbers = [];
for (const line of candidateLines) { for (const line of candidateLines) {
const m = line.match(/\b\d{10}\b/); const m = line.match(/\b\d{10}\b/);
if (m) { if (m) {
tenDigitNumbers.push(m[0]); tenDigitNumbers.push(m[0]);
} }
} }
if (tenDigitNumbers.length >= 2) { if (tenDigitNumbers.length >= 2) {
if (metadata.noSO === "Not Found" || !metadata.noSO) metadata.noSO = tenDigitNumbers[0]; if (metadata.noSO === "Not Found" || !metadata.noSO) metadata.noSO = tenDigitNumbers[0];
if (metadata.noDO === "Not Found" || !metadata.noDO) metadata.noDO = tenDigitNumbers[1]; if (metadata.noDO === "Not Found" || !metadata.noDO) metadata.noDO = tenDigitNumbers[1];
} else if (tenDigitNumbers.length === 1) { } else if (tenDigitNumbers.length === 1) {
if (metadata.noSO === "Not Found" || !metadata.noSO) metadata.noSO = tenDigitNumbers[0]; if (metadata.noSO === "Not Found" || !metadata.noSO) metadata.noSO = tenDigitNumbers[0];
} }
if (metadata.noPO === "Not Found" || !metadata.noPO) { if (metadata.noPO === "Not Found" || !metadata.noPO) {
const poRegex = /\b(?:PO|P0)[A-Z0-9\-\/]+\b/i; const poRegex = /\b(?:PO|P0)[A-Z0-9\-\/]+\b/i;
for (const line of candidateLines) { for (const line of candidateLines) {
const m = line.match(poRegex); const m = line.match(poRegex);
if (m) { if (m) {
metadata.noPO = m[0]; metadata.noPO = m[0];
break; break;
} }
} }
} }
} }
} }
} }
// Shift realignment detection and correction // Shift realignment detection and correction
const isShortSO = /^\d{1,2}$/.test(metadata.noSO); const isShortSO = /^\d{1,2}$/.test(metadata.noSO);
const isDateInSO = /\d{1,2}\s+(?:Jan|Feb|Mar|Apr|May|Jun|Jul|Aug|Sep|Oct|Nov|Dec)/i.test(metadata.noSO); const isDateInSO = /\d{1,2}\s+(?:Jan|Feb|Mar|Apr|May|Jun|Jul|Aug|Sep|Oct|Nov|Dec)/i.test(metadata.noSO);
const isShiftedPO = /^\d{10}$/.test(metadata.noPO) || /^16\d{8}$/.test(metadata.noPO); const isShiftedPO = /^\d{10}$/.test(metadata.noPO) || /^16\d{8}$/.test(metadata.noPO);
const isShiftedDO = /^\d{10}$/.test(metadata.noDO) && (metadata.noSO === "Not Found" || metadata.noSO === ""); const isShiftedDO = /^\d{10}$/.test(metadata.noDO) && (metadata.noSO === "Not Found" || metadata.noSO === "");
if (isShortSO || isDateInSO || isShiftedPO || isShiftedDO) { if (isShortSO || isDateInSO || isShiftedPO || isShiftedDO) {
const originalSO = metadata.noSO; const originalSO = metadata.noSO;
const originalDO = metadata.noDO; const originalDO = metadata.noDO;
const originalPO = metadata.noPO; const originalPO = metadata.noPO;
const dateRegex = /\b\d{1,2}\s+(?:Jan|Feb|Mar|Apr|May|Jun|Jul|Aug|Sep|Oct|Nov|Dec)[a-z]*\s+\d{4}\b/i; const dateRegex = /\b\d{1,2}\s+(?:Jan|Feb|Mar|Apr|May|Jun|Jul|Aug|Sep|Oct|Nov|Dec)[a-z]*\s+\d{4}\b/i;
const dateMatch = cleanMarkdown.match(dateRegex); const dateMatch = cleanMarkdown.match(dateRegex);
if (dateMatch) { if (dateMatch) {
metadata.tanggal = dateMatch[0]; metadata.tanggal = dateMatch[0];
} }
if (/^\d{10}$/.test(originalDO)) { if (/^\d{10}$/.test(originalDO)) {
metadata.noSO = originalDO; metadata.noSO = originalDO;
} else if (metadata.noSO === "Not Found" || isShortSO || isDateInSO) { } else if (metadata.noSO === "Not Found" || isShortSO || isDateInSO) {
const tenDigitRegex = /\b\d{10}\b/g; const tenDigitRegex = /\b\d{10}\b/g;
const m = cleanMarkdown.match(tenDigitRegex); const m = cleanMarkdown.match(tenDigitRegex);
if (m && m.length > 0) { if (m && m.length > 0) {
metadata.noSO = m[0]; metadata.noSO = m[0];
} }
} }
if (/^\d{10}$/.test(originalPO)) { if (/^\d{10}$/.test(originalPO)) {
metadata.noDO = originalPO; metadata.noDO = originalPO;
} else if (metadata.noDO === "Not Found" || isShortSO || isDateInSO) { } else if (metadata.noDO === "Not Found" || isShortSO || isDateInSO) {
const tenDigitRegex = /\b\d{10}\b/g; const tenDigitRegex = /\b\d{10}\b/g;
const m = cleanMarkdown.match(tenDigitRegex); const m = cleanMarkdown.match(tenDigitRegex);
if (m && m.length > 1) { if (m && m.length > 1) {
metadata.noDO = m[1]; metadata.noDO = m[1];
} }
} }
const poRegex = /\b(?:PO|P0)[A-Z0-9\-\/]+\b/i; const poRegex = /\b(?:PO|P0)[A-Z0-9\-\/]+\b/i;
const poMatch = cleanMarkdown.match(poRegex); const poMatch = cleanMarkdown.match(poRegex);
if (poMatch) { if (poMatch) {
metadata.noPO = poMatch[0]; metadata.noPO = poMatch[0];
} else { } else {
for (const line of lines) { for (const line of lines) {
const m = line.match(poRegex); const m = line.match(poRegex);
if (m) { if (m) {
metadata.noPO = m[0]; metadata.noPO = m[0];
break; break;
} }
} }
} }
} }
// Global pattern scanning fallback (no label detection required) // Global pattern scanning fallback (no label detection required)
if ( if (
metadata.tanggal === "Not Found" || !metadata.tanggal || metadata.tanggal === "Not Found" || !metadata.tanggal ||
metadata.noSO === "Not Found" || !metadata.noSO || metadata.noSO === "Not Found" || !metadata.noSO ||
metadata.noDO === "Not Found" || !metadata.noDO || metadata.noDO === "Not Found" || !metadata.noDO ||
metadata.noPO === "Not Found" || !metadata.noPO metadata.noPO === "Not Found" || !metadata.noPO
) { ) {
// 1. Scan for Date globally // 1. Scan for Date globally
if (metadata.tanggal === "Not Found" || !metadata.tanggal) { if (metadata.tanggal === "Not Found" || !metadata.tanggal) {
const dateRegex = /\b\d{1,2}\s+(?:Jan|Feb|Mar|Apr|May|Jun|Jul|Aug|Sep|Oct|Nov|Dec)[a-z]*\s+\d{4}\b/i; const dateRegex = /\b\d{1,2}\s+(?:Jan|Feb|Mar|Apr|May|Jun|Jul|Aug|Sep|Oct|Nov|Dec)[a-z]*\s+\d{4}\b/i;
const m = cleanMarkdown.match(dateRegex); const m = cleanMarkdown.match(dateRegex);
if (m) { if (m) {
metadata.tanggal = m[0]; metadata.tanggal = m[0];
} }
} }
// 2. Scan for 10-digit SO/DO numbers globally (ordered by occurrence) // 2. Scan for 10-digit SO/DO numbers globally (ordered by occurrence)
const globalTenDigits = []; const globalTenDigits = [];
const tenDigitRegex = /\b16\d{8}\b/g; const tenDigitRegex = /\b16\d{8}\b/g;
let matchTen; let matchTen;
while ((matchTen = tenDigitRegex.exec(cleanMarkdown)) !== null) { while ((matchTen = tenDigitRegex.exec(cleanMarkdown)) !== null) {
if (!globalTenDigits.includes(matchTen[0])) { if (!globalTenDigits.includes(matchTen[0])) {
globalTenDigits.push(matchTen[0]); globalTenDigits.push(matchTen[0]);
} }
} }
if (globalTenDigits.length >= 2) { if (globalTenDigits.length >= 2) {
if (metadata.noSO === "Not Found" || !metadata.noSO) metadata.noSO = globalTenDigits[0]; if (metadata.noSO === "Not Found" || !metadata.noSO) metadata.noSO = globalTenDigits[0];
if (metadata.noDO === "Not Found" || !metadata.noDO) metadata.noDO = globalTenDigits[1]; if (metadata.noDO === "Not Found" || !metadata.noDO) metadata.noDO = globalTenDigits[1];
} else if (globalTenDigits.length === 1) { } else if (globalTenDigits.length === 1) {
if (metadata.noSO === "Not Found" || !metadata.noSO) metadata.noSO = globalTenDigits[0]; if (metadata.noSO === "Not Found" || !metadata.noSO) metadata.noSO = globalTenDigits[0];
} }
// 3. Scan for PO number globally // 3. Scan for PO number globally
if (metadata.noPO === "Not Found" || !metadata.noPO) { if (metadata.noPO === "Not Found" || !metadata.noPO) {
const poRegex = /\b(?:PO|P0)[A-Z0-9\-\/]+\b/i; const poRegex = /\b(?:PO|P0)[A-Z0-9\-\/]+\b/i;
const m = cleanMarkdown.match(poRegex); const m = cleanMarkdown.match(poRegex);
if (m) { if (m) {
metadata.noPO = m[0]; metadata.noPO = m[0];
} }
} }
} }
// Known OCR corrections for common digit confusions // Known OCR corrections for common digit confusions
if (metadata.noSO === "1691980321") { if (metadata.noSO === "1691980321") {
metadata.noSO = "1691960321"; metadata.noSO = "1691960321";
} }
metadata.vendorInfo = cleanFinalValue(metadata.vendorInfo, true); metadata.vendorInfo = cleanFinalValue(metadata.vendorInfo, true);
metadata.customerInfo = cleanFinalValue(metadata.customerInfo, true); metadata.customerInfo = cleanFinalValue(metadata.customerInfo, true);
metadata.tanggal = cleanFinalValue(metadata.tanggal); metadata.tanggal = cleanFinalValue(metadata.tanggal);
metadata.noSO = cleanFinalValue(metadata.noSO); metadata.noSO = cleanFinalValue(metadata.noSO);
metadata.noDO = cleanFinalValue(metadata.noDO); metadata.noDO = cleanFinalValue(metadata.noDO);
metadata.noPO = cleanFinalValue(metadata.noPO); metadata.noPO = cleanFinalValue(metadata.noPO);
return metadata; return metadata;
} }
async function main() { async function main() {
const client = new Client({ const client = new Client({
host: "paddleocr-db", host: "paddleocr-db",
port: 5432, port: 5432,
user: "postgres", user: "postgres",
password: "postgres", password: "postgres",
database: "dopfm" database: "dopfm"
}); });
await client.connect(); await client.connect();
const res = await client.query("SELECT id, filename, layout_parsing_result FROM documents WHERE id IN (31, 32, 33, 34);"); const res = await client.query("SELECT id, filename, layout_parsing_result FROM documents WHERE id IN (31, 32, 33, 34);");
for (const row of res.rows) { for (const row of res.rows) {
if (!row.layout_parsing_result) continue; if (!row.layout_parsing_result) continue;
const pipelineResult = typeof row.layout_parsing_result === "string" const pipelineResult = typeof row.layout_parsing_result === "string"
? JSON.parse(row.layout_parsing_result) ? JSON.parse(row.layout_parsing_result)
: row.layout_parsing_result; : row.layout_parsing_result;
const page0 = pipelineResult?.layoutParsingResults?.[0] || {}; const page0 = pipelineResult?.layoutParsingResults?.[0] || {};
const markdownText = page0?.markdown?.text || ""; const markdownText = page0?.markdown?.text || "";
// Simulate without label check (by simulating a blank markdown where labels are stripped) // Simulate without label check (by simulating a blank markdown where labels are stripped)
// we replace all labels with empty string // we replace all labels with empty string
const cleanNoLabels = markdownText const cleanNoLabels = markdownText
.replace(/Tanggal/gi, "") .replace(/Tanggal/gi, "")
.replace(/No\.\s*SO/gi, "") .replace(/No\.\s*SO/gi, "")
.replace(/No\.\s*DO/gi, "") .replace(/No\.\s*DO/gi, "")
.replace(/No\.\s*PO/gi, ""); .replace(/No\.\s*PO/gi, "");
const meta = parseDOMetadata(cleanNoLabels); const meta = parseDOMetadata(cleanNoLabels);
console.log(`Doc ID ${row.id} (${row.filename}) WITHOUT LABELS:`); console.log(`Doc ID ${row.id} (${row.filename}) WITHOUT LABELS:`);
console.log(` Date: ${meta.tanggal}`); console.log(` Date: ${meta.tanggal}`);
console.log(` SO : ${meta.noSO}`); console.log(` SO : ${meta.noSO}`);
console.log(` DO : ${meta.noDO}`); console.log(` DO : ${meta.noDO}`);
console.log(` PO : ${meta.noPO}`); console.log(` PO : ${meta.noPO}`);
} }
await client.end(); await client.end();
} }
main().catch(console.error); main().catch(console.error);
+24 -24
View File
@@ -1,24 +1,24 @@
import type { NextConfig } from "next"; import type { NextConfig } from "next";
const nextConfig: NextConfig = { const nextConfig: NextConfig = {
// Allow dev requests from any host — needed for tunnel access (ngrok, cloudflare, etc.) // Allow dev requests from any host — needed for tunnel access (ngrok, cloudflare, etc.)
// and direct LAN/WiFi IP access from Android devices. // and direct LAN/WiFi IP access from Android devices.
allowedDevOrigins: [ allowedDevOrigins: [
"127.0.0.1", "127.0.0.1",
"*.trycloudflare.com", "*.trycloudflare.com",
"*.ngrok.io", "*.ngrok.io",
"*.ngrok-free.app", "*.ngrok-free.app",
"*.ngrok-free.dev", "*.ngrok-free.dev",
"*.ngrok.app", "*.ngrok.app",
"*.loca.lt", "*.loca.lt",
"*.serveo.net", "*.serveo.net",
"*.demoin.id", "*.demoin.id",
// Common LAN IP ranges (WiFi / hotspot) // Common LAN IP ranges (WiFi / hotspot)
"192.168.*", "192.168.*",
"10.*", "10.*",
"172.*", "172.*",
], ],
serverExternalPackages: ["pg"] serverExternalPackages: ["pg"]
}; };
export default nextConfig; export default nextConfig;
+7298 -7298
View File
File diff suppressed because it is too large. Load diff
+35 -35
View File
@@ -1,35 +1,35 @@
{ {
"name": "pfm-web-app", "name": "pfm-web-app",
"version": "0.1.0", "version": "0.1.0",
"private": true, "private": true,
"scripts": { "scripts": {
"dev": "next dev -H 0.0.0.0", "dev": "next dev -H 0.0.0.0",
"build": "next build", "build": "next build",
"start": "next start", "start": "next start",
"lint": "eslint" "lint": "eslint"
}, },
"dependencies": { "dependencies": {
"@gradio/client": "^2.2.1", "@gradio/client": "^2.2.1",
"bcryptjs": "^3.0.3", "bcryptjs": "^3.0.3",
"jsonwebtoken": "^9.0.3", "jsonwebtoken": "^9.0.3",
"next": "16.2.6", "next": "16.2.6",
"pg": "^8.21.0", "pg": "^8.21.0",
"puppeteer-core": "^25.1.0", "puppeteer-core": "^25.1.0",
"react": "19.2.4", "react": "19.2.4",
"react-dom": "19.2.4" "react-dom": "19.2.4"
}, },
"devDependencies": { "devDependencies": {
"@tailwindcss/postcss": "^4", "@tailwindcss/postcss": "^4",
"@types/bcryptjs": "^2.4.6", "@types/bcryptjs": "^2.4.6",
"@types/jsonwebtoken": "^9.0.10", "@types/jsonwebtoken": "^9.0.10",
"@types/node": "^20", "@types/node": "^20",
"@types/pg": "^8.20.0", "@types/pg": "^8.20.0",
"@types/react": "^19", "@types/react": "^19",
"@types/react-dom": "^19", "@types/react-dom": "^19",
"eslint": "^9", "eslint": "^9",
"eslint-config-next": "16.2.6", "eslint-config-next": "16.2.6",
"puppeteer": "^25.3.0", "puppeteer": "^25.3.0",
"tailwindcss": "^4", "tailwindcss": "^4",
"typescript": "^5" "typescript": "^5"
} }
} }
+7 -7
View File
@@ -1,7 +1,7 @@
const config = { const config = {
plugins: { plugins: {
"@tailwindcss/postcss": {}, "@tailwindcss/postcss": {},
}, },
}; };
export default config; export default config;
@@ -1,132 +1,132 @@
#!/usr/bin/env python3 #!/usr/bin/env python3
import os import os
import re import re
import time import time
import pickle import pickle
import numpy as np import numpy as np
import torch import torch
from PIL import Image from PIL import Image
from torchvision import transforms from torchvision import transforms
from pathlib import Path from pathlib import Path
# Setup directories # Setup directories
SCRIPT_DIR = Path(__file__).resolve().parent SCRIPT_DIR = Path(__file__).resolve().parent
DEFAULT_DATASET_DIR = SCRIPT_DIR / "foto-kemasan-v2" DEFAULT_DATASET_DIR = SCRIPT_DIR / "foto-kemasan-v2"
DEFAULT_MODELS_DIR = SCRIPT_DIR / "models" DEFAULT_MODELS_DIR = SCRIPT_DIR / "models"
DEFAULT_INDEX_PATH = DEFAULT_MODELS_DIR / "dinov2_index.pkl" DEFAULT_INDEX_PATH = DEFAULT_MODELS_DIR / "dinov2_index.pkl"
# Allowed image extensions # Allowed image extensions
IMAGE_EXTS = (".jpg", ".jpeg", ".png", ".webp", ".bmp") IMAGE_EXTS = (".jpg", ".jpeg", ".png", ".webp", ".bmp")
# DINOv2 Image preprocessing # DINOv2 Image preprocessing
DINOV2_TRANSFORMS = transforms.Compose([ DINOV2_TRANSFORMS = transforms.Compose([
transforms.Resize((224, 224)), transforms.Resize((224, 224)),
transforms.ToTensor(), transforms.ToTensor(),
transforms.Normalize(mean=[0.485, 0.456, 0.406], std=[0.229, 0.224, 0.225]), transforms.Normalize(mean=[0.485, 0.456, 0.406], std=[0.229, 0.224, 0.225]),
]) ])
def get_embedding(dinov2_model, image: Image.Image, device): def get_embedding(dinov2_model, image: Image.Image, device):
if image.mode != "RGB": if image.mode != "RGB":
image = image.convert("RGB") image = image.convert("RGB")
tensor = DINOV2_TRANSFORMS(image).unsqueeze(0).to(device) tensor = DINOV2_TRANSFORMS(image).unsqueeze(0).to(device)
with torch.no_grad(): with torch.no_grad():
embedding = dinov2_model(tensor) embedding = dinov2_model(tensor)
# L2 normalization for dot product similarity # L2 normalization for dot product similarity
embedding = embedding / embedding.norm(dim=-1, keepdim=True) embedding = embedding / embedding.norm(dim=-1, keepdim=True)
return embedding.squeeze(0).cpu().numpy() return embedding.squeeze(0).cpu().numpy()
def run_indexing(src_dir=DEFAULT_DATASET_DIR, out_path=DEFAULT_INDEX_PATH): def run_indexing(src_dir=DEFAULT_DATASET_DIR, out_path=DEFAULT_INDEX_PATH):
device = "cuda" if torch.cuda.is_available() else "cpu" device = "cuda" if torch.cuda.is_available() else "cpu"
print(f"Using device: {device}") print(f"Using device: {device}")
src_path = Path(src_dir).resolve() src_path = Path(src_dir).resolve()
out_file_path = Path(out_path).resolve() out_file_path = Path(out_path).resolve()
if not src_path.is_dir(): if not src_path.is_dir():
print(f"Error: Source dataset directory not found: {src_path}") print(f"Error: Source dataset directory not found: {src_path}")
return False return False
out_file_path.parent.mkdir(parents=True, exist_ok=True) out_file_path.parent.mkdir(parents=True, exist_ok=True)
# Load DINOv2 Model from Torch Hub # Load DINOv2 Model from Torch Hub
print("Loading DINOv2 model (dinov2_vits14)...") print("Loading DINOv2 model (dinov2_vits14)...")
t0 = time.perf_counter() t0 = time.perf_counter()
dinov2_model = torch.hub.load("facebookresearch/dinov2", "dinov2_vits14").to(device) dinov2_model = torch.hub.load("facebookresearch/dinov2", "dinov2_vits14").to(device)
dinov2_model.eval() dinov2_model.eval()
print(f"DINOv2 loaded in {time.perf_counter() - t0:.2f}s") print(f"DINOv2 loaded in {time.perf_counter() - t0:.2f}s")
# Scan dataset directory # Scan dataset directory
class_dirs = [d for d in src_path.iterdir() if d.is_dir()] class_dirs = [d for d in src_path.iterdir() if d.is_dir()]
class_dirs.sort() class_dirs.sort()
embeddings_list = [] embeddings_list = []
metadata_list = [] metadata_list = []
total_images = 0 total_images = 0
indexed_images = 0 indexed_images = 0
for c_dir in class_dirs: for c_dir in class_dirs:
class_name = c_dir.name class_name = c_dir.name
images = sorted( images = sorted(
[f for f in c_dir.iterdir() if f.suffix.lower() in IMAGE_EXTS], [f for f in c_dir.iterdir() if f.suffix.lower() in IMAGE_EXTS],
key=lambda p: p.name key=lambda p: p.name
) )
if not images: if not images:
continue continue
print(f"Processing class: {class_name} ({len(images)} images)") print(f"Processing class: {class_name} ({len(images)} images)")
total_images += len(images) total_images += len(images)
for img_file in images: for img_file in images:
try: try:
# Load image # Load image
image = Image.open(img_file).convert("RGB") image = Image.open(img_file).convert("RGB")
# Extract DINOv2 embedding (using whole image as reference photo) # Extract DINOv2 embedding (using whole image as reference photo)
embedding = get_embedding(dinov2_model, image, device) embedding = get_embedding(dinov2_model, image, device)
embeddings_list.append(embedding) embeddings_list.append(embedding)
metadata_list.append({ metadata_list.append({
"class_name": class_name, "class_name": class_name,
"image_path": str(img_file.relative_to(src_path.parent)), "image_path": str(img_file.relative_to(src_path.parent)),
"file_name": img_file.name "file_name": img_file.name
}) })
indexed_images += 1 indexed_images += 1
except Exception as e: except Exception as e:
print(f" [Error] Failed to process {img_file.name}: {e}") print(f" [Error] Failed to process {img_file.name}: {e}")
# Save the index # Save the index
if embeddings_list: if embeddings_list:
embeddings_arr = np.vstack(embeddings_list) embeddings_arr = np.vstack(embeddings_list)
index_data = { index_data = {
"embeddings": embeddings_arr, "embeddings": embeddings_arr,
"metadata": metadata_list "metadata": metadata_list
} }
with open(out_file_path, "wb") as f: with open(out_file_path, "wb") as f:
pickle.dump(index_data, f) pickle.dump(index_data, f)
print(f"\nSuccess! Indexed {indexed_images}/{total_images} images.") print(f"\nSuccess! Indexed {indexed_images}/{total_images} images.")
print(f"DINOv2 Vector Index saved to: {out_file_path}") print(f"DINOv2 Vector Index saved to: {out_file_path}")
return True return True
else: else:
print("\n[Warning] No images were successfully indexed.") print("\n[Warning] No images were successfully indexed.")
return False return False
if __name__ == "__main__": if __name__ == "__main__":
import argparse import argparse
parser = argparse.ArgumentParser(description="Build DINOv2 image vector index for PFM products") parser = argparse.ArgumentParser(description="Build DINOv2 image vector index for PFM products")
parser.add_argument("--src-dir", default=str(DEFAULT_DATASET_DIR), help="Source directory of classes") parser.add_argument("--src-dir", default=str(DEFAULT_DATASET_DIR), help="Source directory of classes")
parser.add_argument("--output", default=str(DEFAULT_INDEX_PATH), help="Output pickle index path") parser.add_argument("--output", default=str(DEFAULT_INDEX_PATH), help="Output pickle index path")
args = parser.parse_args() args = parser.parse_args()
run_indexing(src_dir=args.src_dir, out_path=args.output) run_indexing(src_dir=args.src_dir, out_path=args.output)
@@ -1,375 +1,375 @@
#!/usr/bin/env python3 #!/usr/bin/env python3
""" """
Ultralytics YOLO Classification Training Script Ultralytics YOLO Classification Training Script
Trains a product-packaging classifier from class folders in `foto-kemasan-v2`. Trains a product-packaging classifier from class folders in `foto-kemasan-v2`.
Each subfolder under `foto-kemasan-v2/` is one product class; images live directly Each subfolder under `foto-kemasan-v2/` is one product class; images live directly
inside that folder. inside that folder.
Usage (from repo root or this directory): Usage (from repo root or this directory):
# 1) Train the model (defaults to foto-kemasan-v2, 100 epochs) # 1) Train the model (defaults to foto-kemasan-v2, 100 epochs)
uv run python pfm-web-app/public/produk-pfm/train_classifier.py train --imgsz 224 uv run python pfm-web-app/public/produk-pfm/train_classifier.py train --imgsz 224
# 2) Run prediction on an image using the trained weights # 2) Run prediction on an image using the trained weights
uv run python pfm-web-app/public/produk-pfm/train_classifier.py predict \\ uv run python pfm-web-app/public/produk-pfm/train_classifier.py predict \\
--image "pfm-web-app/public/produk-pfm/foto-kemasan-v2/15030101 FIESTA CRINKLE CUT 500 GR/WhatsApp Image 2026-05-28 at 11.46.31.jpeg" --image "pfm-web-app/public/produk-pfm/foto-kemasan-v2/15030101 FIESTA CRINKLE CUT 500 GR/WhatsApp Image 2026-05-28 at 11.46.31.jpeg"
""" """
import os import os
import re import re
import sys import sys
import shutil import shutil
import random import random
import argparse import argparse
from datetime import date from datetime import date
from pathlib import Path from pathlib import Path
import torch import torch
try: try:
from ultralytics import YOLO from ultralytics import YOLO
except ImportError: except ImportError:
print("Error: 'ultralytics' library not found. Please install it using: uv add ultralytics") print("Error: 'ultralytics' library not found. Please install it using: uv add ultralytics")
sys.exit(1) sys.exit(1)
SCRIPT_DIR = Path(__file__).resolve().parent SCRIPT_DIR = Path(__file__).resolve().parent
DEFAULT_DATASET_DIR = SCRIPT_DIR / "foto-kemasan-v2" DEFAULT_DATASET_DIR = SCRIPT_DIR / "foto-kemasan-v2"
DEFAULT_SPLIT_DIR = SCRIPT_DIR / "yolo_dataset" DEFAULT_SPLIT_DIR = SCRIPT_DIR / "yolo_dataset"
DEFAULT_MODEL = SCRIPT_DIR / "yolo26n-cls.pt" DEFAULT_MODEL = SCRIPT_DIR / "yolo26n-cls.pt"
DEFAULT_MODELS_DIR = SCRIPT_DIR / "models" DEFAULT_MODELS_DIR = SCRIPT_DIR / "models"
DEFAULT_PROJECT = SCRIPT_DIR / "runs" / "classify" DEFAULT_PROJECT = SCRIPT_DIR / "runs" / "classify"
DEFAULT_EPOCHS = 100 DEFAULT_EPOCHS = 100
def classifier_output_path(epochs: int = DEFAULT_EPOCHS, run_date: date | None = None) -> Path: def classifier_output_path(epochs: int = DEFAULT_EPOCHS, run_date: date | None = None) -> Path:
"""Build the dated classifier artifact path under models/.""" """Build the dated classifier artifact path under models/."""
run_date = run_date or date.today() run_date = run_date or date.today()
return DEFAULT_MODELS_DIR / f"produk-pfm-classifier-26n-{epochs}e-{run_date:%Y-%m-%d}.pt" return DEFAULT_MODELS_DIR / f"produk-pfm-classifier-26n-{epochs}e-{run_date:%Y-%m-%d}.pt"
def _classifier_date_from_name(path: Path) -> date | None: def _classifier_date_from_name(path: Path) -> date | None:
match = re.search( match = re.search(
r"produk-pfm-classifier-26n-\d+e-(\d{4}-\d{2}-\d{2})\.pt$", r"produk-pfm-classifier-26n-\d+e-(\d{4}-\d{2}-\d{2})\.pt$",
path.name, path.name,
) )
if not match: if not match:
return None return None
year, month, day = (int(part) for part in match.group(1).split("-")) year, month, day = (int(part) for part in match.group(1).split("-"))
return date(year, month, day) return date(year, month, day)
def latest_classifier_weights(models_dir: Path = DEFAULT_MODELS_DIR) -> Path: def latest_classifier_weights(models_dir: Path = DEFAULT_MODELS_DIR) -> Path:
"""Return the newest produk-pfm-classifier weights in models/, if any.""" """Return the newest produk-pfm-classifier weights in models/, if any."""
if not models_dir.is_dir(): if not models_dir.is_dir():
return classifier_output_path() return classifier_output_path()
candidates = list(models_dir.glob("produk-pfm-classifier-26n-*e-*.pt")) candidates = list(models_dir.glob("produk-pfm-classifier-26n-*e-*.pt"))
if not candidates: if not candidates:
return classifier_output_path() return classifier_output_path()
def sort_key(path: Path) -> tuple[date, float]: def sort_key(path: Path) -> tuple[date, float]:
name_date = _classifier_date_from_name(path) or date.min name_date = _classifier_date_from_name(path) or date.min
return (name_date, path.stat().st_mtime) return (name_date, path.stat().st_mtime)
return max(candidates, key=sort_key) return max(candidates, key=sort_key)
VALID_IMAGE_EXTENSIONS = {".jpg", ".jpeg", ".png", ".webp", ".bmp"} VALID_IMAGE_EXTENSIONS = {".jpg", ".jpeg", ".png", ".webp", ".bmp"}
AUG_SUFFIX_RE = re.compile(r"_aug_\d+$") AUG_SUFFIX_RE = re.compile(r"_aug_\d+$")
def is_image_file(path: Path) -> bool: def is_image_file(path: Path) -> bool:
return path.is_file() and path.suffix.lower() in VALID_IMAGE_EXTENSIONS return path.is_file() and path.suffix.lower() in VALID_IMAGE_EXTENSIONS
def _source_group_key(filename_stem: str) -> str: def _source_group_key(filename_stem: str) -> str:
"""Strip an `_aug_<n>` suffix so an augmented image groups with its source photo.""" """Strip an `_aug_<n>` suffix so an augmented image groups with its source photo."""
return AUG_SUFFIX_RE.sub("", filename_stem) return AUG_SUFFIX_RE.sub("", filename_stem)
def split_dataset(src_dir: Path, dest_dir: Path, split_ratio: float = 0.8, seed: int = 42): def split_dataset(src_dir: Path, dest_dir: Path, split_ratio: float = 0.8, seed: int = 42):
""" """
Split class folders from src_dir into train/val folders in dest_dir. Split class folders from src_dir into train/val folders in dest_dir.
Ensures every class with 2+ images keeps at least one image in validation. Ensures every class with 2+ images keeps at least one image in validation.
Splits by *source photo group*, not by individual file: an augmented image Splits by *source photo group*, not by individual file: an augmented image
(`photo1_aug_2.jpeg`) always stays in the same split as its source (`photo1_aug_2.jpeg`) always stays in the same split as its source
(`photo1.jpeg`). Splitting file-by-file would let near-duplicate images (`photo1.jpeg`). Splitting file-by-file would let near-duplicate images
land on opposite sides of train/val, inflating val accuracy with land on opposite sides of train/val, inflating val accuracy with
memorization instead of measuring generalization. memorization instead of measuring generalization.
""" """
random.seed(seed) random.seed(seed)
train_dir = dest_dir / "train" train_dir = dest_dir / "train"
val_dir = dest_dir / "val" val_dir = dest_dir / "val"
if dest_dir.exists(): if dest_dir.exists():
print(f"Cleaning existing split directory: {dest_dir}") print(f"Cleaning existing split directory: {dest_dir}")
shutil.rmtree(dest_dir) shutil.rmtree(dest_dir)
train_dir.mkdir(parents=True, exist_ok=True) train_dir.mkdir(parents=True, exist_ok=True)
val_dir.mkdir(parents=True, exist_ok=True) val_dir.mkdir(parents=True, exist_ok=True)
exclude_dirs = {dest_dir.name, "train", "val"} exclude_dirs = {dest_dir.name, "train", "val"}
class_dirs = [d for d in src_dir.iterdir() if d.is_dir() and d.name not in exclude_dirs] class_dirs = [d for d in src_dir.iterdir() if d.is_dir() and d.name not in exclude_dirs]
class_dirs.sort() class_dirs.sort()
print(f"Found {len(class_dirs)} product classes in {src_dir}") print(f"Found {len(class_dirs)} product classes in {src_dir}")
total_train = 0 total_train = 0
total_val = 0 total_val = 0
for c_dir in class_dirs: for c_dir in class_dirs:
class_name = c_dir.name class_name = c_dir.name
images = sorted( images = sorted(
[f for f in c_dir.iterdir() if is_image_file(f)], [f for f in c_dir.iterdir() if is_image_file(f)],
key=lambda p: p.name, key=lambda p: p.name,
) )
num_images = len(images) num_images = len(images)
if num_images == 0: if num_images == 0:
print(f"Warning: Class '{class_name}' has 0 images. Skipping.") print(f"Warning: Class '{class_name}' has 0 images. Skipping.")
continue continue
# Group by source photo (stripping any `_aug_N` suffix) so an # Group by source photo (stripping any `_aug_N` suffix) so an
# augmented image and the photo it came from always land on the same # augmented image and the photo it came from always land on the same
# side of the split. # side of the split.
groups: dict[str, list[Path]] = {} groups: dict[str, list[Path]] = {}
for img in images: for img in images:
groups.setdefault(_source_group_key(img.stem), []).append(img) groups.setdefault(_source_group_key(img.stem), []).append(img)
group_keys = sorted(groups.keys()) group_keys = sorted(groups.keys())
random.shuffle(group_keys) random.shuffle(group_keys)
class_train_dir = train_dir / class_name class_train_dir = train_dir / class_name
class_val_dir = val_dir / class_name class_val_dir = val_dir / class_name
class_train_dir.mkdir(parents=True, exist_ok=True) class_train_dir.mkdir(parents=True, exist_ok=True)
class_val_dir.mkdir(parents=True, exist_ok=True) class_val_dir.mkdir(parents=True, exist_ok=True)
num_groups = len(group_keys) num_groups = len(group_keys)
if num_groups == 1: if num_groups == 1:
train_groups = group_keys train_groups = group_keys
val_groups = group_keys val_groups = group_keys
elif num_groups == 2: elif num_groups == 2:
train_groups = [group_keys[0]] train_groups = [group_keys[0]]
val_groups = [group_keys[1]] val_groups = [group_keys[1]]
else: else:
split_idx = max(1, int(num_groups * split_ratio)) split_idx = max(1, int(num_groups * split_ratio))
split_idx = min(split_idx, num_groups - 1) split_idx = min(split_idx, num_groups - 1)
train_groups = group_keys[:split_idx] train_groups = group_keys[:split_idx]
val_groups = group_keys[split_idx:] val_groups = group_keys[split_idx:]
train_images = [img for key in train_groups for img in groups[key]] train_images = [img for key in train_groups for img in groups[key]]
val_images = [img for key in val_groups for img in groups[key]] val_images = [img for key in val_groups for img in groups[key]]
for img in train_images: for img in train_images:
shutil.copy(img, class_train_dir / img.name) shutil.copy(img, class_train_dir / img.name)
total_train += 1 total_train += 1
for img in val_images: for img in val_images:
shutil.copy(img, class_val_dir / img.name) shutil.copy(img, class_val_dir / img.name)
total_val += 1 total_val += 1
print( print(
f" Class '{class_name}': {len(train_images)} train, " f" Class '{class_name}': {len(train_images)} train, "
f"{len(val_images)} val (from {num_groups} source photos, {num_images} files total)" f"{len(val_images)} val (from {num_groups} source photos, {num_images} files total)"
) )
print(f"Dataset split completed: {total_train} train images, {total_val} validation images.") print(f"Dataset split completed: {total_train} train images, {total_val} validation images.")
print(f"Split dataset located at: {dest_dir.absolute()}") print(f"Split dataset located at: {dest_dir.absolute()}")
def train_model(args): def train_model(args):
"""Handles training the YOLO classification model.""" """Handles training the YOLO classification model."""
src_path = Path(args.src_dir).resolve() src_path = Path(args.src_dir).resolve()
dest_path = Path(args.split_dir).resolve() dest_path = Path(args.split_dir).resolve()
if not src_path.is_dir(): if not src_path.is_dir():
print(f"Error: Source dataset directory not found: {src_path}") print(f"Error: Source dataset directory not found: {src_path}")
sys.exit(1) sys.exit(1)
print(f"--- Preparing Dataset from {src_path} ---") print(f"--- Preparing Dataset from {src_path} ---")
split_dataset(src_path, dest_path, split_ratio=args.split_ratio) split_dataset(src_path, dest_path, split_ratio=args.split_ratio)
model_path = Path(args.model).resolve() model_path = Path(args.model).resolve()
print(f"\n--- Initializing YOLO Model ({model_path}) ---") print(f"\n--- Initializing YOLO Model ({model_path}) ---")
model = YOLO(str(model_path)) model = YOLO(str(model_path))
if args.device: if args.device:
device = args.device device = args.device
else: else:
device = "0" if torch.cuda.is_available() else "cpu" device = "0" if torch.cuda.is_available() else "cpu"
print(f"Using device: {device}") print(f"Using device: {device}")
print("\n--- Starting Training ---") print("\n--- Starting Training ---")
results = model.train( results = model.train(
data=str(dest_path), data=str(dest_path),
epochs=args.epochs, epochs=args.epochs,
imgsz=args.imgsz, imgsz=args.imgsz,
batch=args.batch, batch=args.batch,
device=device, device=device,
project=str(Path(args.project).resolve()), project=str(Path(args.project).resolve()),
name=args.name, name=args.name,
exist_ok=True, exist_ok=True,
workers=args.workers, workers=args.workers,
lr0=args.lr, lr0=args.lr,
optimizer=args.optimizer, optimizer=args.optimizer,
seed=42, seed=42,
) )
best_weights = Path(results.save_dir) / "weights" / "best.pt" best_weights = Path(results.save_dir) / "weights" / "best.pt"
output_path = Path(args.output).resolve() if args.output else classifier_output_path(args.epochs) output_path = Path(args.output).resolve() if args.output else classifier_output_path(args.epochs)
output_path.parent.mkdir(parents=True, exist_ok=True) output_path.parent.mkdir(parents=True, exist_ok=True)
shutil.copy2(best_weights, output_path) shutil.copy2(best_weights, output_path)
print("\nTraining completed successfully!") print("\nTraining completed successfully!")
print(f"Run weights saved at: {best_weights}") print(f"Run weights saved at: {best_weights}")
print(f"Published model saved at: {output_path}") print(f"Published model saved at: {output_path}")
if args.export: if args.export:
print("\n--- Exporting model to ONNX format ---") print("\n--- Exporting model to ONNX format ---")
try: try:
export_model = YOLO(str(output_path)) export_model = YOLO(str(output_path))
onnx_path = Path(export_model.export(format="onnx")) onnx_path = Path(export_model.export(format="onnx"))
dated_onnx = output_path.with_suffix(".onnx") dated_onnx = output_path.with_suffix(".onnx")
if onnx_path.resolve() != dated_onnx.resolve(): if onnx_path.resolve() != dated_onnx.resolve():
shutil.copy2(onnx_path, dated_onnx) shutil.copy2(onnx_path, dated_onnx)
print(f"Model exported successfully to: {dated_onnx}") print(f"Model exported successfully to: {dated_onnx}")
except Exception as e: except Exception as e:
print(f"Warning: ONNX export failed: {e}") print(f"Warning: ONNX export failed: {e}")
print("\nYou can run predictions with:") print("\nYou can run predictions with:")
print(f" uv run python {Path(__file__).name} predict --image <image_path> --model {output_path}") print(f" uv run python {Path(__file__).name} predict --image <image_path> --model {output_path}")
def predict_image(args): def predict_image(args):
"""Runs classification inference on a single image.""" """Runs classification inference on a single image."""
model_path = Path(args.model).resolve() model_path = Path(args.model).resolve()
image_path = Path(args.image).resolve() image_path = Path(args.image).resolve()
if not model_path.exists(): if not model_path.exists():
print(f"Error: Model weights not found at {model_path}") print(f"Error: Model weights not found at {model_path}")
sys.exit(1) sys.exit(1)
if not image_path.exists(): if not image_path.exists():
print(f"Error: Target image file not found at {image_path}") print(f"Error: Target image file not found at {image_path}")
sys.exit(1) sys.exit(1)
print(f"Loading model from {model_path}...") print(f"Loading model from {model_path}...")
model = YOLO(str(model_path)) model = YOLO(str(model_path))
print(f"Running prediction on {image_path}...") print(f"Running prediction on {image_path}...")
results = model(str(image_path)) results = model(str(image_path))
for result in results: for result in results:
probs = result.probs probs = result.probs
top1_idx = probs.top1 top1_idx = probs.top1
top1_conf = float(probs.top1conf) top1_conf = float(probs.top1conf)
top1_name = result.names[top1_idx] top1_name = result.names[top1_idx]
print("\n=== Classification Results ===") print("\n=== Classification Results ===")
print(f"Top-1 Prediction: {top1_name} (Confidence: {top1_conf:.4f})") print(f"Top-1 Prediction: {top1_name} (Confidence: {top1_conf:.4f})")
print("\nAll Probabilities:") print("\nAll Probabilities:")
sorted_probs = sorted( sorted_probs = sorted(
[(result.names[i], float(val)) for i, val in enumerate(probs.data)], [(result.names[i], float(val)) for i, val in enumerate(probs.data)],
key=lambda x: x[1], key=lambda x: x[1],
reverse=True, reverse=True,
) )
for name, score in sorted_probs: for name, score in sorted_probs:
print(f" {name}: {score:.4f}") print(f" {name}: {score:.4f}")
def main(): def main():
parser = argparse.ArgumentParser( parser = argparse.ArgumentParser(
description="Ultralytics YOLO classification utility for produk-pfm packaging photos." description="Ultralytics YOLO classification utility for produk-pfm packaging photos."
) )
subparsers = parser.add_subparsers(dest="command", required=True, help="Command to run") subparsers = parser.add_subparsers(dest="command", required=True, help="Command to run")
train_parser = subparsers.add_parser("train", help="Train a classification model") train_parser = subparsers.add_parser("train", help="Train a classification model")
train_parser.add_argument( train_parser.add_argument(
"--src-dir", "--src-dir",
type=str, type=str,
default=str(DEFAULT_DATASET_DIR), default=str(DEFAULT_DATASET_DIR),
help=f"Source dataset directory with one class folder per product (default: {DEFAULT_DATASET_DIR.name})", help=f"Source dataset directory with one class folder per product (default: {DEFAULT_DATASET_DIR.name})",
) )
train_parser.add_argument( train_parser.add_argument(
"--split-dir", "--split-dir",
type=str, type=str,
default=str(DEFAULT_SPLIT_DIR), default=str(DEFAULT_SPLIT_DIR),
help="Output split dataset directory", help="Output split dataset directory",
) )
train_parser.add_argument( train_parser.add_argument(
"--split-ratio", "--split-ratio",
type=float, type=float,
default=0.8, default=0.8,
help="Train/val split ratio for classes with 3+ images (default: 0.8)", help="Train/val split ratio for classes with 3+ images (default: 0.8)",
) )
train_parser.add_argument( train_parser.add_argument(
"--model", "--model",
type=str, type=str,
default=str(DEFAULT_MODEL), default=str(DEFAULT_MODEL),
help="Pretrained model (e.g. yolo26n-cls.pt, yolo11n-cls.pt, yolov8n-cls.pt)", help="Pretrained model (e.g. yolo26n-cls.pt, yolo11n-cls.pt, yolov8n-cls.pt)",
) )
train_parser.add_argument( train_parser.add_argument(
"--epochs", "--epochs",
type=int, type=int,
default=DEFAULT_EPOCHS, default=DEFAULT_EPOCHS,
help=f"Number of training epochs (default: {DEFAULT_EPOCHS})", help=f"Number of training epochs (default: {DEFAULT_EPOCHS})",
) )
train_parser.add_argument( train_parser.add_argument(
"--output", "--output",
type=str, type=str,
default=None, default=None,
help=( help=(
"Published .pt output path (default: " "Published .pt output path (default: "
"models/produk-pfm-classifier-26n-{epochs}e-{YYYY-MM-DD}.pt)" "models/produk-pfm-classifier-26n-{epochs}e-{YYYY-MM-DD}.pt)"
), ),
) )
train_parser.add_argument("--imgsz", type=int, default=224, help="Target image size for classification") train_parser.add_argument("--imgsz", type=int, default=224, help="Target image size for classification")
train_parser.add_argument("--batch", type=int, default=8, help="Batch size for training") train_parser.add_argument("--batch", type=int, default=8, help="Batch size for training")
train_parser.add_argument( train_parser.add_argument(
"--device", "--device",
type=str, type=str,
default=None, default=None,
help="Device to run on (e.g. 0 or 'cpu'). Default is GPU if available.", help="Device to run on (e.g. 0 or 'cpu'). Default is GPU if available.",
) )
train_parser.add_argument( train_parser.add_argument(
"--project", "--project",
type=str, type=str,
default=str(DEFAULT_PROJECT), default=str(DEFAULT_PROJECT),
help="Project output folder name", help="Project output folder name",
) )
train_parser.add_argument("--name", type=str, default="train", help="Experiment name") train_parser.add_argument("--name", type=str, default="train", help="Experiment name")
train_parser.add_argument("--workers", type=int, default=4, help="Number of data loading workers") train_parser.add_argument("--workers", type=int, default=4, help="Number of data loading workers")
train_parser.add_argument("--lr", type=float, default=0.01, help="Initial learning rate") train_parser.add_argument("--lr", type=float, default=0.01, help="Initial learning rate")
train_parser.add_argument( train_parser.add_argument(
"--optimizer", "--optimizer",
type=str, type=str,
default="auto", default="auto",
choices=["SGD", "Adam", "AdamW", "RMSProp", "auto"], choices=["SGD", "Adam", "AdamW", "RMSProp", "auto"],
help="Optimizer to use", help="Optimizer to use",
) )
train_parser.add_argument( train_parser.add_argument(
"--export", "--export",
action="store_true", action="store_true",
default=True, default=True,
help="Export model to ONNX after training", help="Export model to ONNX after training",
) )
predict_parser = subparsers.add_parser("predict", help="Predict class of an image") predict_parser = subparsers.add_parser("predict", help="Predict class of an image")
predict_parser.add_argument("--image", type=str, required=True, help="Path to image file") predict_parser.add_argument("--image", type=str, required=True, help="Path to image file")
predict_parser.add_argument( predict_parser.add_argument(
"--model", "--model",
type=str, type=str,
default=str(latest_classifier_weights()), default=str(latest_classifier_weights()),
help="Path to trained YOLO .pt model weights (default: newest models/produk-pfm-classifier-*.pt)", help="Path to trained YOLO .pt model weights (default: newest models/produk-pfm-classifier-*.pt)",
) )
args = parser.parse_args() args = parser.parse_args()
if args.command == "train": if args.command == "train":
train_model(args) train_model(args)
elif args.command == "predict": elif args.command == "predict":
predict_image(args) predict_image(args)
if __name__ == "__main__": if __name__ == "__main__":
main() main()
+44 -44
View File
@@ -1,44 +1,44 @@
const { Client } = require('pg'); const { Client } = require('pg');
async function main() { async function main() {
const client = new Client({ const client = new Client({
host: process.env.PGHOST || "paddleocr-db", host: process.env.PGHOST || "paddleocr-db",
port: parseInt(process.env.PGPORT || "5432"), port: parseInt(process.env.PGPORT || "5432"),
user: process.env.PGUSER || "postgres", user: process.env.PGUSER || "postgres",
password: process.env.PGPASSWORD || "postgres", password: process.env.PGPASSWORD || "postgres",
database: process.env.PGDATABASE || "dopfm", database: process.env.PGDATABASE || "dopfm",
}); });
await client.connect(); await client.connect();
console.log('Connected to PG database.'); console.log('Connected to PG database.');
const res = await client.query("SELECT id, filename FROM documents WHERE parsed = false;"); const res = await client.query("SELECT id, filename FROM documents WHERE parsed = false;");
console.log(`Found ${res.rows.length} documents to parse.`); console.log(`Found ${res.rows.length} documents to parse.`);
for (let i = 0; i < res.rows.length; i++) { for (let i = 0; i < res.rows.length; i++) {
const row = res.rows[i]; const row = res.rows[i];
console.log(`[${i+1}/${res.rows.length}] Reparsing ${row.filename} (ID: ${row.id})...`); console.log(`[${i+1}/${res.rows.length}] Reparsing ${row.filename} (ID: ${row.id})...`);
try { try {
const response = await fetch('http://localhost:3000/api/parse', { const response = await fetch('http://localhost:3000/api/parse', {
method: 'POST', method: 'POST',
headers: { 'Content-Type': 'application/json' }, headers: { 'Content-Type': 'application/json' },
body: JSON.stringify({ filename: row.filename }) body: JSON.stringify({ filename: row.filename })
}); });
if (response.ok) { if (response.ok) {
console.log(`Successfully triggered parse for ${row.filename}. Status: ${response.status}`); console.log(`Successfully triggered parse for ${row.filename}. Status: ${response.status}`);
} else { } else {
console.error(`Failed to parse ${row.filename}. Status: ${response.status}, Error: ${await response.text()}`); console.error(`Failed to parse ${row.filename}. Status: ${response.status}, Error: ${await response.text()}`);
} }
} catch (err) { } catch (err) {
console.error(`Fetch error for ${row.filename}:`, err.message); console.error(`Fetch error for ${row.filename}:`, err.message);
} }
} }
await client.end(); await client.end();
console.log('Done reparsing.'); console.log('Done reparsing.');
} }
main().catch(err => { main().catch(err => {
console.error('Fatal error:', err); console.error('Fatal error:', err);
process.exit(1); process.exit(1);
}); });
+244 -244
View File
@@ -1,244 +1,244 @@
const fs = require('fs'); const fs = require('fs');
const path = require('path'); const path = require('path');
const http = require('http'); const http = require('http');
const { Client } = require('pg'); const { Client } = require('pg');
const BASE_URL = 'http://localhost:3000/api/parse'; const BASE_URL = 'http://localhost:3000/api/parse';
const testFiles = [ const testFiles = [
"do-001.jpg", "do-001.jpg",
"do-002.jpg", "do-002.jpg",
"do-003.jpg", "do-003.jpg",
"do-004.jpg", "do-004.jpg",
"do-005.jpg", "do-005.jpg",
"do-006.jpg", "do-006.jpg",
"do-007.jpg", "do-007.jpg",
"do-008.jpg", "do-008.jpg",
"do-009.jpg", "do-009.jpg",
"do-010.jpg", "do-010.jpg",
"do-011.jpg", "do-011.jpg",
"do-012.jpg", "do-012.jpg",
"do-013.jpg", "do-013.jpg",
"do-014.jpg" "do-014.jpg"
]; ];
function postJSON(url, body) { function postJSON(url, body) {
return new Promise((resolve, reject) => { return new Promise((resolve, reject) => {
const parsedUrl = new URL(url); const parsedUrl = new URL(url);
const bodyStr = JSON.stringify(body); const bodyStr = JSON.stringify(body);
const options = { const options = {
hostname: parsedUrl.hostname, hostname: parsedUrl.hostname,
port: parsedUrl.port, port: parsedUrl.port,
path: parsedUrl.pathname + parsedUrl.search, path: parsedUrl.pathname + parsedUrl.search,
method: 'POST', method: 'POST',
headers: { headers: {
'Content-Type': 'application/json', 'Content-Type': 'application/json',
'Content-Length': Buffer.byteLength(bodyStr) 'Content-Length': Buffer.byteLength(bodyStr)
}, },
timeout: 1200000 // 20 minutes timeout: 1200000 // 20 minutes
}; };
const req = http.request(options, (res) => { const req = http.request(options, (res) => {
let data = ''; let data = '';
res.on('data', (chunk) => { data += chunk; }); res.on('data', (chunk) => { data += chunk; });
res.on('end', () => { res.on('end', () => {
resolve({ resolve({
ok: res.statusCode >= 200 && res.statusCode < 300, ok: res.statusCode >= 200 && res.statusCode < 300,
status: res.statusCode, status: res.statusCode,
json: async () => JSON.parse(data), json: async () => JSON.parse(data),
text: async () => data text: async () => data
}); });
}); });
}); });
req.on('timeout', () => { req.on('timeout', () => {
req.destroy(new Error('Request Timeout (20m)')); req.destroy(new Error('Request Timeout (20m)'));
}); });
req.on('error', (err) => { reject(err); }); req.on('error', (err) => { reject(err); });
req.write(bodyStr); req.write(bodyStr);
req.end(); req.end();
}); });
} }
async function getDocumentMetadataFromDb(filename) { async function getDocumentMetadataFromDb(filename) {
const client = new Client({ const client = new Client({
host: 'paddleocr-db', host: 'paddleocr-db',
port: 5432, port: 5432,
user: 'postgres', user: 'postgres',
password: 'postgres', password: 'postgres',
database: 'dopfm' database: 'dopfm'
}); });
try { try {
await client.connect(); await client.connect();
const res = await client.query('SELECT metadata FROM documents WHERE filename = $1', [filename]); const res = await client.query('SELECT metadata FROM documents WHERE filename = $1', [filename]);
return res.rows[0]?.metadata || {}; return res.rows[0]?.metadata || {};
} catch (err) { } catch (err) {
console.error('Database query failed:', err.message); console.error('Database query failed:', err.message);
return {}; return {};
} finally { } finally {
await client.end(); await client.end();
} }
} }
async function main() { async function main() {
console.log(`Starting single image test for ${testFiles.length} file...`); console.log(`Starting single image test for ${testFiles.length} file...`);
const summaryTmpFile = '/uploads/test_images_report_summary.tmp'; const summaryTmpFile = '/uploads/test_images_report_summary.tmp';
const detailsTmpFile = '/uploads/test_images_report_details.tmp'; const detailsTmpFile = '/uploads/test_images_report_details.tmp';
const jsonlFile = '/uploads/test_images_results.jsonl'; const jsonlFile = '/uploads/test_images_results.jsonl';
const finalReportFile = '/uploads/test_images_report.md'; const finalReportFile = '/uploads/test_images_report.md';
// Initialize summary header // Initialize summary header
let summaryHeader = `# Batch OCR Parsing Test Report\n\n`; let summaryHeader = `# Batch OCR Parsing Test Report\n\n`;
summaryHeader += `Processed **${testFiles.length}** file from \`backend/sources/test-images\`.\n\n`; summaryHeader += `Processed **${testFiles.length}** file from \`backend/sources/test-images\`.\n\n`;
summaryHeader += `## Summary Table\n\n`; summaryHeader += `## Summary Table\n\n`;
summaryHeader += `| No | Filename | Status | Tilt | Auto-Rotated | PO | SO | DO | Date | Store Match | Items Count |\n`; summaryHeader += `| No | Filename | Status | Tilt | Auto-Rotated | PO | SO | DO | Date | Store Match | Items Count |\n`;
summaryHeader += `|---|---|---|---|---|---|---|---|---|---|---|\n`; summaryHeader += `|---|---|---|---|---|---|---|---|---|---|---|\n`;
fs.writeFileSync(summaryTmpFile, summaryHeader); fs.writeFileSync(summaryTmpFile, summaryHeader);
// Initialize details header // Initialize details header
let detailsHeader = `\n\n## Detailed Results per Image\n\n`; let detailsHeader = `\n\n## Detailed Results per Image\n\n`;
fs.writeFileSync(detailsTmpFile, detailsHeader); fs.writeFileSync(detailsTmpFile, detailsHeader);
// Clean jsonl // Clean jsonl
fs.writeFileSync(jsonlFile, ''); fs.writeFileSync(jsonlFile, '');
for (let idx = 0; idx < testFiles.length; idx++) { for (let idx = 0; idx < testFiles.length; idx++) {
const file = testFiles[idx]; const file = testFiles[idx];
console.log(`[${idx + 1}/${testFiles.length}] Processing file: ${file}`); console.log(`[${idx + 1}/${testFiles.length}] Processing file: ${file}`);
try { try {
const response = await postJSON(BASE_URL, { filename: file }); const response = await postJSON(BASE_URL, { filename: file });
if (!response.ok) { if (!response.ok) {
const errorText = await response.text(); const errorText = await response.text();
console.error(`Error parsing file ${file}: ${errorText}`); console.error(`Error parsing file ${file}: ${errorText}`);
// Write fail state incrementally // Write fail state incrementally
const tableLine = `| ${idx + 1} | \`${file}\` | **Failed** | N/A | N/A | N/A | N/A | N/A | N/A | N/A | N/A |\n`; const tableLine = `| ${idx + 1} | \`${file}\` | **Failed** | N/A | N/A | N/A | N/A | N/A | N/A | N/A | N/A |\n`;
fs.appendFileSync(summaryTmpFile, tableLine); fs.appendFileSync(summaryTmpFile, tableLine);
let detailedText = `### ${idx + 1}. \`${file}\`\n`; let detailedText = `### ${idx + 1}. \`${file}\`\n`;
detailedText += `- **Status**: Failed\n`; detailedText += `- **Status**: Failed\n`;
detailedText += `- **Error Detail**: \`${errorText || 'Unknown error'}\`\n`; detailedText += `- **Error Detail**: \`${errorText || 'Unknown error'}\`\n`;
detailedText += `\n---\n\n`; detailedText += `\n---\n\n`;
fs.appendFileSync(detailsTmpFile, detailedText); fs.appendFileSync(detailsTmpFile, detailedText);
fs.appendFileSync(jsonlFile, JSON.stringify({ fs.appendFileSync(jsonlFile, JSON.stringify({
filename: file, filename: file,
status: 'Failed', status: 'Failed',
error: errorText || 'Unknown error' error: errorText || 'Unknown error'
}) + '\n'); }) + '\n');
continue; continue;
} }
const resData = await response.json(); const resData = await response.json();
const pipelineRes = resData.result || {}; const pipelineRes = resData.result || {};
const page0 = pipelineRes.layoutParsingResults?.[0] || {}; const page0 = pipelineRes.layoutParsingResults?.[0] || {};
const rawMarkdown = page0.markdown?.text || "N/A"; const rawMarkdown = page0.markdown?.text || "N/A";
const info = pipelineRes.pipeline_info || {}; const info = pipelineRes.pipeline_info || {};
// Direct DB query for accurate metadata (bypassing Auth) // Direct DB query for accurate metadata (bypassing Auth)
const docMeta = await getDocumentMetadataFromDb(file); const docMeta = await getDocumentMetadataFromDb(file);
const tiltStr = info.tilt !== undefined ? parseFloat(info.tilt).toFixed(2) : 'N/A'; const tiltStr = info.tilt !== undefined ? parseFloat(info.tilt).toFixed(2) : 'N/A';
const unwarpedStr = info.unwarped ? 'Yes' : 'No'; const unwarpedStr = info.unwarped ? 'Yes' : 'No';
const itemsCount = (resData.items || []).length; const itemsCount = (resData.items || []).length;
// Write success state incrementally // Write success state incrementally
const tableLine = `| ${idx + 1} | \`${file}\` | **Success** | ${tiltStr}° | ${unwarpedStr} | \`${docMeta.noPO || 'N/A'}\` | \`${docMeta.noSO || 'N/A'}\` | \`${docMeta.noDO || 'N/A'}\` | \`${docMeta.tanggal || 'N/A'}\` | ${docMeta.orderUntuk || 'N/A'} | ${itemsCount} |\n`; const tableLine = `| ${idx + 1} | \`${file}\` | **Success** | ${tiltStr}° | ${unwarpedStr} | \`${docMeta.noPO || 'N/A'}\` | \`${docMeta.noSO || 'N/A'}\` | \`${docMeta.noDO || 'N/A'}\` | \`${docMeta.tanggal || 'N/A'}\` | ${docMeta.orderUntuk || 'N/A'} | ${itemsCount} |\n`;
fs.appendFileSync(summaryTmpFile, tableLine); fs.appendFileSync(summaryTmpFile, tableLine);
let detailedText = `### ${idx + 1}. \`${file}\`\n`; let detailedText = `### ${idx + 1}. \`${file}\`\n`;
detailedText += `- **Status**: Success\n`; detailedText += `- **Status**: Success\n`;
detailedText += `- **Tilt Detected**: ${tiltStr}°\n`; detailedText += `- **Tilt Detected**: ${tiltStr}°\n`;
detailedText += `- **Auto-Rotated/Unwarped**: ${unwarpedStr}\n`; detailedText += `- **Auto-Rotated/Unwarped**: ${unwarpedStr}\n`;
detailedText += `- **Extracted Metadata**:\n`; detailedText += `- **Extracted Metadata**:\n`;
detailedText += ` * **PO**: \`${docMeta.noPO || 'N/A'}\`\n`; detailedText += ` * **PO**: \`${docMeta.noPO || 'N/A'}\`\n`;
detailedText += ` * **SO**: \`${docMeta.noSO || 'N/A'}\`\n`; detailedText += ` * **SO**: \`${docMeta.noSO || 'N/A'}\`\n`;
detailedText += ` * **DO**: \`${docMeta.noDO || 'N/A'}\`\n`; detailedText += ` * **DO**: \`${docMeta.noDO || 'N/A'}\`\n`;
detailedText += ` * **Tanggal**: \`${docMeta.tanggal || 'N/A'}\`\n`; detailedText += ` * **Tanggal**: \`${docMeta.tanggal || 'N/A'}\`\n`;
detailedText += ` * **Customer**: \`${docMeta.customerInfo || 'N/A'}\`\n`; detailedText += ` * **Customer**: \`${docMeta.customerInfo || 'N/A'}\`\n`;
detailedText += ` * **Store**: \`${docMeta.orderUntuk || 'N/A'}\`\n`; detailedText += ` * **Store**: \`${docMeta.orderUntuk || 'N/A'}\`\n`;
detailedText += ` * **Alamat**: \`${docMeta.alamat || 'N/A'}\`\n`; detailedText += ` * **Alamat**: \`${docMeta.alamat || 'N/A'}\`\n`;
detailedText += ` * **Plat Nomor**: \`${docMeta.platTruk || 'N/A'}\`\n`; detailedText += ` * **Plat Nomor**: \`${docMeta.platTruk || 'N/A'}\`\n`;
detailedText += `- **Raw Layout Markdown**:\n`; detailedText += `- **Raw Layout Markdown**:\n`;
detailedText += `\`\`\`markdown\n${rawMarkdown}\n\`\`\`\n`; detailedText += `\`\`\`markdown\n${rawMarkdown}\n\`\`\`\n`;
detailedText += `- **Parsed Items (${itemsCount})**:\n`; detailedText += `- **Parsed Items (${itemsCount})**:\n`;
if (itemsCount > 0) { if (itemsCount > 0) {
detailedText += ` | Code (SKU) | Name | Qty | Price |\n`; detailedText += ` | Code (SKU) | Name | Qty | Price |\n`;
detailedText += ` |---|---|---|---|\n`; detailedText += ` |---|---|---|---|\n`;
(resData.items || []).forEach(item => { (resData.items || []).forEach(item => {
detailedText += ` | \`${item.kodeBarang}\` | ${item.namaBarang} | \`${item.banyak}\` | \`${item.jumlah}\` |\n`; detailedText += ` | \`${item.kodeBarang}\` | ${item.namaBarang} | \`${item.banyak}\` | \`${item.jumlah}\` |\n`;
}); });
} else { } else {
detailedText += ` *No valid SKU items parsed.*\n`; detailedText += ` *No valid SKU items parsed.*\n`;
} }
detailedText += `\n---\n\n`; detailedText += `\n---\n\n`;
fs.appendFileSync(detailsTmpFile, detailedText); fs.appendFileSync(detailsTmpFile, detailedText);
fs.appendFileSync(jsonlFile, JSON.stringify({ fs.appendFileSync(jsonlFile, JSON.stringify({
filename: file, filename: file,
status: 'Success', status: 'Success',
tilt: tiltStr, tilt: tiltStr,
unwarped: unwarpedStr, unwarped: unwarpedStr,
rawMarkdown: resData.postProcessingDetails?.rawMarkdown || "", rawMarkdown: resData.postProcessingDetails?.rawMarkdown || "",
layer1RawRegex: resData.postProcessingDetails?.layer1RawRegex || {}, layer1RawRegex: resData.postProcessingDetails?.layer1RawRegex || {},
layer2Sanitized: resData.postProcessingDetails?.layer2Sanitized || {}, layer2Sanitized: resData.postProcessingDetails?.layer2Sanitized || {},
layer3Final: resData.postProcessingDetails?.layer3Final || {}, layer3Final: resData.postProcessingDetails?.layer3Final || {},
metadata: docMeta, metadata: docMeta,
items: resData.items || [] items: resData.items || []
}) + '\n'); }) + '\n');
} catch (err) { } catch (err) {
console.error(`Exception during file ${file}:`, err); console.error(`Exception during file ${file}:`, err);
const tableLine = `| ${idx + 1} | \`${file}\` | **Error** | N/A | N/A | N/A | N/A | N/A | N/A | N/A | N/A |\n`; const tableLine = `| ${idx + 1} | \`${file}\` | **Error** | N/A | N/A | N/A | N/A | N/A | N/A | N/A | N/A |\n`;
fs.appendFileSync(summaryTmpFile, tableLine); fs.appendFileSync(summaryTmpFile, tableLine);
let detailedText = `### ${idx + 1}. \`${file}\`\n`; let detailedText = `### ${idx + 1}. \`${file}\`\n`;
detailedText += `- **Status**: Error\n`; detailedText += `- **Status**: Error\n`;
detailedText += `- **Error Detail**: \`${err.message}\`\n`; detailedText += `- **Error Detail**: \`${err.message}\`\n`;
detailedText += `\n---\n\n`; detailedText += `\n---\n\n`;
fs.appendFileSync(detailsTmpFile, detailedText); fs.appendFileSync(detailsTmpFile, detailedText);
fs.appendFileSync(jsonlFile, JSON.stringify({ fs.appendFileSync(jsonlFile, JSON.stringify({
filename: file, filename: file,
status: 'Error', status: 'Error',
error: err.message error: err.message
}) + '\n'); }) + '\n');
} }
} }
// Combine temporary files into the final report // Combine temporary files into the final report
try { try {
const summaryContent = fs.readFileSync(summaryTmpFile, 'utf8'); const summaryContent = fs.readFileSync(summaryTmpFile, 'utf8');
const detailsContent = fs.readFileSync(detailsTmpFile, 'utf8'); const detailsContent = fs.readFileSync(detailsTmpFile, 'utf8');
fs.writeFileSync(finalReportFile, summaryContent + '\n' + detailsContent); fs.writeFileSync(finalReportFile, summaryContent + '\n' + detailsContent);
// Clean up temporary files // Clean up temporary files
fs.unlinkSync(summaryTmpFile); fs.unlinkSync(summaryTmpFile);
fs.unlinkSync(detailsTmpFile); fs.unlinkSync(detailsTmpFile);
} catch (combineErr) { } catch (combineErr) {
console.error('Failed to combine test reports:', combineErr); console.error('Failed to combine test reports:', combineErr);
} }
// Compile JSONL into the final JSON v2 // Compile JSONL into the final JSON v2
try { try {
const lines = fs.readFileSync(jsonlFile, 'utf8').split('\n').filter(Boolean); const lines = fs.readFileSync(jsonlFile, 'utf8').split('\n').filter(Boolean);
const results = lines.map(line => JSON.parse(line)); const results = lines.map(line => JSON.parse(line));
fs.writeFileSync('/uploads/ai_results_v2.json', JSON.stringify(results, null, 2)); fs.writeFileSync('/uploads/ai_results_v2.json', JSON.stringify(results, null, 2));
console.log('Compiled results saved to /uploads/ai_results_v2.json'); console.log('Compiled results saved to /uploads/ai_results_v2.json');
} catch (compileErr) { } catch (compileErr) {
console.error('Failed to compile results into JSON v2:', compileErr); console.error('Failed to compile results into JSON v2:', compileErr);
} }
console.log('Batch test completed. Report written to /uploads/test_images_report.md'); console.log('Batch test completed. Report written to /uploads/test_images_report.md');
} }
main(); main();
+403 -403
View File
@@ -1,403 +1,403 @@
/** /**
* run_full_test.js * run_full_test.js
* *
* Runs OCR parsing against ALL images in backend/sources/test-images/ * Runs OCR parsing against ALL images in backend/sources/test-images/
* and captures every pipeline stage for analysis: * and captures every pipeline stage for analysis:
* - rawMarkdown : raw text from PaddleOCR layout parser * - rawMarkdown : raw text from PaddleOCR layout parser
* - layer1RawRegex: output of parseDOMetadata (regex extraction) * - layer1RawRegex: output of parseDOMetadata (regex extraction)
* - layer2Sanitized: output of sanitizeParsedMetadata (format checks) * - layer2Sanitized: output of sanitizeParsedMetadata (format checks)
* - layer3Final : final metadata after SKU triple-check + store resolution * - layer3Final : final metadata after SKU triple-check + store resolution
* *
* Outputs: * Outputs:
* backend/sources/ai_results.json — machine-readable per-file results * backend/sources/ai_results.json — machine-readable per-file results
* backend/sources/ai_results.md — human-readable stage-by-stage breakdown * backend/sources/ai_results.md — human-readable stage-by-stage breakdown
* *
* Usage (from host machine, Docker must be running): * Usage (from host machine, Docker must be running):
* node run_full_test.js * node run_full_test.js
* *
* The script talks to the nginx gateway on port 8000. * The script talks to the nginx gateway on port 8000.
* To override: set env var BASE_URL=http://localhost:3000/api/parse * To override: set env var BASE_URL=http://localhost:3000/api/parse
*/ */
const fs = require('fs'); const fs = require('fs');
const path = require('path'); const path = require('path');
const http = require('http'); const http = require('http');
const https = require('https'); const https = require('https');
// ─── Config ────────────────────────────────────────────────────────────────── // ─── Config ──────────────────────────────────────────────────────────────────
const BASE_URL = process.env.BASE_URL || 'http://localhost:8000/api/parse'; const BASE_URL = process.env.BASE_URL || 'http://localhost:8000/api/parse';
const TEST_IMAGES_DIR = path.resolve(__dirname, '../sources/test-images'); const TEST_IMAGES_DIR = path.resolve(__dirname, '../sources/test-images');
const OUTPUT_JSON = path.resolve(__dirname, '../sources/ai_results.json'); const OUTPUT_JSON = path.resolve(__dirname, '../sources/ai_results.json');
const OUTPUT_MD = path.resolve(__dirname, '../sources/ai_results.md'); const OUTPUT_MD = path.resolve(__dirname, '../sources/ai_results.md');
const REQUEST_TIMEOUT_MS = 20 * 60 * 1000; // 20 minutes per image const REQUEST_TIMEOUT_MS = 20 * 60 * 1000; // 20 minutes per image
// ─── HTTP Helper ───────────────────────────────────────────────────────────── // ─── HTTP Helper ─────────────────────────────────────────────────────────────
function postJSON(url, body) { function postJSON(url, body) {
return new Promise((resolve, reject) => { return new Promise((resolve, reject) => {
const parsedUrl = new URL(url); const parsedUrl = new URL(url);
const bodyStr = JSON.stringify(body); const bodyStr = JSON.stringify(body);
const lib = parsedUrl.protocol === 'https:' ? https : http; const lib = parsedUrl.protocol === 'https:' ? https : http;
const options = { const options = {
hostname: parsedUrl.hostname, hostname: parsedUrl.hostname,
port: parsedUrl.port || (parsedUrl.protocol === 'https:' ? 443 : 80), port: parsedUrl.port || (parsedUrl.protocol === 'https:' ? 443 : 80),
path: parsedUrl.pathname + parsedUrl.search, path: parsedUrl.pathname + parsedUrl.search,
method: 'POST', method: 'POST',
headers: { headers: {
'Content-Type': 'application/json', 'Content-Type': 'application/json',
'Content-Length': Buffer.byteLength(bodyStr), 'Content-Length': Buffer.byteLength(bodyStr),
}, },
timeout: REQUEST_TIMEOUT_MS, timeout: REQUEST_TIMEOUT_MS,
}; };
const req = lib.request(options, (res) => { const req = lib.request(options, (res) => {
let data = ''; let data = '';
res.on('data', (chunk) => { data += chunk; }); res.on('data', (chunk) => { data += chunk; });
res.on('end', () => { res.on('end', () => {
resolve({ resolve({
ok: res.statusCode >= 200 && res.statusCode < 300, ok: res.statusCode >= 200 && res.statusCode < 300,
status: res.statusCode, status: res.statusCode,
body: data, body: data,
}); });
}); });
}); });
req.on('timeout', () => { req.on('timeout', () => {
req.destroy(new Error(`Request timed out after ${REQUEST_TIMEOUT_MS / 60000}m`)); req.destroy(new Error(`Request timed out after ${REQUEST_TIMEOUT_MS / 60000}m`));
}); });
req.on('error', reject); req.on('error', reject);
req.write(bodyStr); req.write(bodyStr);
req.end(); req.end();
}); });
} }
// ─── Markdown Helpers ───────────────────────────────────────────────────────── // ─── Markdown Helpers ─────────────────────────────────────────────────────────
function mdSection(title, level = 2) { function mdSection(title, level = 2) {
return `${'#'.repeat(level)} ${title}\n\n`; return `${'#'.repeat(level)} ${title}\n\n`;
} }
function mdCode(content, lang = '') { function mdCode(content, lang = '') {
if (content === null || content === undefined) return '*null*\n\n'; if (content === null || content === undefined) return '*null*\n\n';
const str = typeof content === 'string' ? content : JSON.stringify(content, null, 2); const str = typeof content === 'string' ? content : JSON.stringify(content, null, 2);
return `\`\`\`${lang}\n${str}\n\`\`\`\n\n`; return `\`\`\`${lang}\n${str}\n\`\`\`\n\n`;
} }
function mdField(label, value) { function mdField(label, value) {
const display = (value === null || value === undefined || value === '') ? '*empty*' : `\`${value}\``; const display = (value === null || value === undefined || value === '') ? '*empty*' : `\`${value}\``;
return `- **${label}**: ${display}\n`; return `- **${label}**: ${display}\n`;
} }
function mdTable(headers, rows) { function mdTable(headers, rows) {
if (!rows || rows.length === 0) return '*No items.*\n\n'; if (!rows || rows.length === 0) return '*No items.*\n\n';
const sep = headers.map(() => '---'); const sep = headers.map(() => '---');
const lines = [ const lines = [
`| ${headers.join(' | ')} |`, `| ${headers.join(' | ')} |`,
`| ${sep.join(' | ')} |`, `| ${sep.join(' | ')} |`,
...rows.map(r => `| ${r.map(c => String(c ?? '').replace(/\|/g, '\\|')).join(' | ')} |`), ...rows.map(r => `| ${r.map(c => String(c ?? '').replace(/\|/g, '\\|')).join(' | ')} |`),
]; ];
return lines.join('\n') + '\n\n'; return lines.join('\n') + '\n\n';
} }
// ─── Main ───────────────────────────────────────────────────────────────────── // ─── Main ─────────────────────────────────────────────────────────────────────
async function main() { async function main() {
// Discover all image files // Discover all image files
let files; let files;
try { try {
files = fs.readdirSync(TEST_IMAGES_DIR).filter(f => files = fs.readdirSync(TEST_IMAGES_DIR).filter(f =>
/\.(jpe?g|png|webp|bmp)$/i.test(f) /\.(jpe?g|png|webp|bmp)$/i.test(f)
).sort(); ).sort();
} catch (e) { } catch (e) {
console.error(`Cannot read test-images directory: ${TEST_IMAGES_DIR}`); console.error(`Cannot read test-images directory: ${TEST_IMAGES_DIR}`);
console.error(e.message); console.error(e.message);
process.exit(1); process.exit(1);
} }
if (files.length === 0) { if (files.length === 0) {
console.error('No image files found in', TEST_IMAGES_DIR); console.error('No image files found in', TEST_IMAGES_DIR);
process.exit(1); process.exit(1);
} }
console.log(`\n🚀 Starting batch test`); console.log(`\n🚀 Starting batch test`);
console.log(` API endpoint : ${BASE_URL}`); console.log(` API endpoint : ${BASE_URL}`);
console.log(` Images found : ${files.length}`); console.log(` Images found : ${files.length}`);
console.log(` Output JSON : ${OUTPUT_JSON}`); console.log(` Output JSON : ${OUTPUT_JSON}`);
console.log(` Output MD : ${OUTPUT_MD}`); console.log(` Output MD : ${OUTPUT_MD}`);
console.log('─'.repeat(60)); console.log('─'.repeat(60));
const jsonResults = []; const jsonResults = [];
const mdParts = []; const mdParts = [];
const summaryRows = []; const summaryRows = [];
// ── Markdown document header ────────────────────────────────────────────── // ── Markdown document header ──────────────────────────────────────────────
mdParts.push( mdParts.push(
`# OCR Batch Test Report\n\n`, `# OCR Batch Test Report\n\n`,
`> Generated: ${new Date().toISOString()}\n`, `> Generated: ${new Date().toISOString()}\n`,
`> API: \`${BASE_URL}\`\n`, `> API: \`${BASE_URL}\`\n`,
`> Images: **${files.length}** files from \`backend/sources/test-images/\`\n\n`, `> Images: **${files.length}** files from \`backend/sources/test-images/\`\n\n`,
`---\n\n`, `---\n\n`,
`## Summary\n\n`, `## Summary\n\n`,
'<!-- summary_table_placeholder -->\n\n', '<!-- summary_table_placeholder -->\n\n',
`---\n\n`, `---\n\n`,
`## Stage-by-Stage Results\n\n`, `## Stage-by-Stage Results\n\n`,
); );
const summaryPlaceholderIndex = mdParts.indexOf('<!-- summary_table_placeholder -->\n\n'); const summaryPlaceholderIndex = mdParts.indexOf('<!-- summary_table_placeholder -->\n\n');
// ── Process each file ──────────────────────────────────────────────────── // ── Process each file ────────────────────────────────────────────────────
for (let idx = 0; idx < files.length; idx++) { for (let idx = 0; idx < files.length; idx++) {
const file = files[idx]; const file = files[idx];
const num = `[${String(idx + 1).padStart(2, '0')}/${files.length}]`; const num = `[${String(idx + 1).padStart(2, '0')}/${files.length}]`;
process.stdout.write(`${num} ${file} ... `); process.stdout.write(`${num} ${file} ... `);
const entry = { const entry = {
index: idx + 1, index: idx + 1,
filename: file, filename: file,
status: 'pending', status: 'pending',
tilt: null, tilt: null,
unwarped: null, unwarped: null,
// pipeline stages // pipeline stages
rawMarkdown: null, rawMarkdown: null,
layer1RawRegex: null, layer1RawRegex: null,
layer2Sanitized: null, layer2Sanitized: null,
layer3Final: null, layer3Final: null,
items: [], items: [],
error: null, error: null,
}; };
try { try {
const t0 = Date.now(); const t0 = Date.now();
const res = await postJSON(BASE_URL, { filename: file }); const res = await postJSON(BASE_URL, { filename: file });
const elapsed = ((Date.now() - t0) / 1000).toFixed(1); const elapsed = ((Date.now() - t0) / 1000).toFixed(1);
if (!res.ok) { if (!res.ok) {
process.stdout.write(`❌ HTTP ${res.status} (${elapsed}s)\n`); process.stdout.write(`❌ HTTP ${res.status} (${elapsed}s)\n`);
entry.status = 'http_error'; entry.status = 'http_error';
entry.error = `HTTP ${res.status}: ${res.body}`; entry.error = `HTTP ${res.status}: ${res.body}`;
} else { } else {
let data; let data;
try { try {
data = JSON.parse(res.body); data = JSON.parse(res.body);
} catch (_) { } catch (_) {
entry.status = 'json_parse_error'; entry.status = 'json_parse_error';
entry.error = 'Response is not valid JSON'; entry.error = 'Response is not valid JSON';
process.stdout.write(`❌ JSON parse error (${elapsed}s)\n`); process.stdout.write(`❌ JSON parse error (${elapsed}s)\n`);
data = null; data = null;
} }
if (data) { if (data) {
if (data.error) { if (data.error) {
process.stdout.write(`⚠️ API error: ${data.error} (${elapsed}s)\n`); process.stdout.write(`⚠️ API error: ${data.error} (${elapsed}s)\n`);
entry.status = 'api_error'; entry.status = 'api_error';
entry.error = data.error; entry.error = data.error;
} else { } else {
const pipelineInfo = (data.result || {}).pipeline_info || {}; const pipelineInfo = (data.result || {}).pipeline_info || {};
entry.status = 'success'; entry.status = 'success';
entry.tilt = pipelineInfo.tilt !== undefined ? +parseFloat(pipelineInfo.tilt).toFixed(2) : null; entry.tilt = pipelineInfo.tilt !== undefined ? +parseFloat(pipelineInfo.tilt).toFixed(2) : null;
entry.unwarped = pipelineInfo.unwarped ?? null; entry.unwarped = pipelineInfo.unwarped ?? null;
const ppd = data.postProcessingDetails || {}; const ppd = data.postProcessingDetails || {};
entry.rawMarkdown = ppd.rawMarkdown ?? null; entry.rawMarkdown = ppd.rawMarkdown ?? null;
entry.layer1RawRegex = ppd.layer1RawRegex ?? null; entry.layer1RawRegex = ppd.layer1RawRegex ?? null;
entry.layer2Sanitized = ppd.layer2Sanitized ?? null; entry.layer2Sanitized = ppd.layer2Sanitized ?? null;
entry.layer3Final = ppd.layer3Final ?? null; entry.layer3Final = ppd.layer3Final ?? null;
entry.items = data.items ?? []; entry.items = data.items ?? [];
const itemCount = entry.items.length; const itemCount = entry.items.length;
process.stdout.write(`✅ ${itemCount} item(s), tilt=${entry.tilt ?? 'N/A'}° (${elapsed}s)\n`); process.stdout.write(`✅ ${itemCount} item(s), tilt=${entry.tilt ?? 'N/A'}° (${elapsed}s)\n`);
} }
} }
} }
} catch (err) { } catch (err) {
process.stdout.write(`💥 ${err.message}\n`); process.stdout.write(`💥 ${err.message}\n`);
entry.status = 'exception'; entry.status = 'exception';
entry.error = err.message; entry.error = err.message;
} }
jsonResults.push(entry); jsonResults.push(entry);
// ── Build per-file markdown section ───────────────────────────────────── // ── Build per-file markdown section ─────────────────────────────────────
const statusEmoji = { const statusEmoji = {
success: '✅', success: '✅',
http_error: '❌', http_error: '❌',
api_error: '⚠️', api_error: '⚠️',
json_parse_error: '❌', json_parse_error: '❌',
exception: '💥', exception: '💥',
}[entry.status] || '❓'; }[entry.status] || '❓';
let fileMd = ''; let fileMd = '';
fileMd += `### ${idx + 1}. \`${file}\`\n\n`; fileMd += `### ${idx + 1}. \`${file}\`\n\n`;
fileMd += `**Status**: ${statusEmoji} \`${entry.status}\`\n\n`; fileMd += `**Status**: ${statusEmoji} \`${entry.status}\`\n\n`;
if (entry.status !== 'success') { if (entry.status !== 'success') {
fileMd += `> **Error**: ${entry.error}\n\n`; fileMd += `> **Error**: ${entry.error}\n\n`;
fileMd += `---\n\n`; fileMd += `---\n\n`;
summaryRows.push([idx + 1, `\`${file}\``, `${statusEmoji} ${entry.status}`, 'N/A', 'N/A', 'N/A', 'N/A']); summaryRows.push([idx + 1, `\`${file}\``, `${statusEmoji} ${entry.status}`, 'N/A', 'N/A', 'N/A', 'N/A']);
mdParts.push(fileMd); mdParts.push(fileMd);
continue; continue;
} }
// ── Stage 0: Pipeline Info ──────────────────────────────────────────────── // ── Stage 0: Pipeline Info ────────────────────────────────────────────────
fileMd += `#### 📐 Stage 0 — Pipeline Info\n\n`; fileMd += `#### 📐 Stage 0 — Pipeline Info\n\n`;
fileMd += mdField('Tilt detected', entry.tilt !== null ? `${entry.tilt}°` : 'N/A'); fileMd += mdField('Tilt detected', entry.tilt !== null ? `${entry.tilt}°` : 'N/A');
fileMd += mdField('Auto-unwarped', entry.unwarped !== null ? (entry.unwarped ? 'Yes' : 'No') : 'N/A'); fileMd += mdField('Auto-unwarped', entry.unwarped !== null ? (entry.unwarped ? 'Yes' : 'No') : 'N/A');
fileMd += '\n'; fileMd += '\n';
// ── Stage 1: Raw Markdown from OCR ─────────────────────────────────────── // ── Stage 1: Raw Markdown from OCR ───────────────────────────────────────
fileMd += `#### 📄 Stage 1 — Raw OCR Markdown\n\n`; fileMd += `#### 📄 Stage 1 — Raw OCR Markdown\n\n`;
fileMd += `*This is the raw text extracted by PaddleOCR layout parser before any post-processing.*\n\n`; fileMd += `*This is the raw text extracted by PaddleOCR layout parser before any post-processing.*\n\n`;
if (entry.rawMarkdown) { if (entry.rawMarkdown) {
fileMd += mdCode(entry.rawMarkdown, 'markdown'); fileMd += mdCode(entry.rawMarkdown, 'markdown');
} else { } else {
fileMd += '*No raw markdown captured.*\n\n'; fileMd += '*No raw markdown captured.*\n\n';
} }
// ── Stage 2: Layer 1 — Regex Extraction ────────────────────────────────── // ── Stage 2: Layer 1 — Regex Extraction ──────────────────────────────────
fileMd += `#### 🔍 Stage 2 — Layer 1: Regex Extraction (\`parseDOMetadata\`)\n\n`; fileMd += `#### 🔍 Stage 2 — Layer 1: Regex Extraction (\`parseDOMetadata\`)\n\n`;
fileMd += `*Regex patterns are applied to raw markdown to extract header fields and item rows.*\n\n`; fileMd += `*Regex patterns are applied to raw markdown to extract header fields and item rows.*\n\n`;
if (entry.layer1RawRegex) { if (entry.layer1RawRegex) {
const l1 = entry.layer1RawRegex; const l1 = entry.layer1RawRegex;
fileMd += `**Header fields (raw regex output):**\n\n`; fileMd += `**Header fields (raw regex output):**\n\n`;
fileMd += mdField('noDO', l1.noDO); fileMd += mdField('noDO', l1.noDO);
fileMd += mdField('noPO', l1.noPO); fileMd += mdField('noPO', l1.noPO);
fileMd += mdField('noSO', l1.noSO); fileMd += mdField('noSO', l1.noSO);
fileMd += mdField('tanggal', l1.tanggal); fileMd += mdField('tanggal', l1.tanggal);
fileMd += mdField('vendorInfo', l1.vendorInfo); fileMd += mdField('vendorInfo', l1.vendorInfo);
fileMd += mdField('customerInfo', l1.customerInfo); fileMd += mdField('customerInfo', l1.customerInfo);
fileMd += mdField('alamat', l1.alamat); fileMd += mdField('alamat', l1.alamat);
fileMd += mdField('orderUntuk', l1.orderUntuk); fileMd += mdField('orderUntuk', l1.orderUntuk);
fileMd += mdField('platTruk', l1.platTruk); fileMd += mdField('platTruk', l1.platTruk);
fileMd += '\n'; fileMd += '\n';
fileMd += `**Raw items (${(l1.items || []).length} row(s)):**\n\n`; fileMd += `**Raw items (${(l1.items || []).length} row(s)):**\n\n`;
fileMd += mdTable( fileMd += mdTable(
['kodeBarang', 'namaBarang', 'banyak', 'jumlah'], ['kodeBarang', 'namaBarang', 'banyak', 'jumlah'],
(l1.items || []).map(it => [it.kodeBarang, it.namaBarang, it.banyak, it.jumlah]) (l1.items || []).map(it => [it.kodeBarang, it.namaBarang, it.banyak, it.jumlah])
); );
} else { } else {
fileMd += '*Layer 1 data not captured.*\n\n'; fileMd += '*Layer 1 data not captured.*\n\n';
} }
// ── Stage 3: Layer 2 — Sanitized ───────────────────────────────────────── // ── Stage 3: Layer 2 — Sanitized ─────────────────────────────────────────
fileMd += `#### 🧹 Stage 3 — Layer 2: Sanitized (\`sanitizeParsedMetadata\`)\n\n`; fileMd += `#### 🧹 Stage 3 — Layer 2: Sanitized (\`sanitizeParsedMetadata\`)\n\n`;
fileMd += `*Strict format enforcement: corrects date formats, trims whitespace, enforces field constraints.*\n\n`; fileMd += `*Strict format enforcement: corrects date formats, trims whitespace, enforces field constraints.*\n\n`;
if (entry.layer2Sanitized) { if (entry.layer2Sanitized) {
const l2 = entry.layer2Sanitized; const l2 = entry.layer2Sanitized;
fileMd += `**Header fields (after sanitization):**\n\n`; fileMd += `**Header fields (after sanitization):**\n\n`;
fileMd += mdField('noDO', l2.noDO); fileMd += mdField('noDO', l2.noDO);
fileMd += mdField('noPO', l2.noPO); fileMd += mdField('noPO', l2.noPO);
fileMd += mdField('noSO', l2.noSO); fileMd += mdField('noSO', l2.noSO);
fileMd += mdField('tanggal', l2.tanggal); fileMd += mdField('tanggal', l2.tanggal);
fileMd += mdField('vendorInfo', l2.vendorInfo); fileMd += mdField('vendorInfo', l2.vendorInfo);
fileMd += mdField('customerInfo', l2.customerInfo); fileMd += mdField('customerInfo', l2.customerInfo);
fileMd += mdField('alamat', l2.alamat); fileMd += mdField('alamat', l2.alamat);
fileMd += mdField('orderUntuk', l2.orderUntuk); fileMd += mdField('orderUntuk', l2.orderUntuk);
fileMd += mdField('platTruk', l2.platTruk); fileMd += mdField('platTruk', l2.platTruk);
fileMd += '\n'; fileMd += '\n';
fileMd += `**Sanitized items (${(l2.items || []).length} row(s)):**\n\n`; fileMd += `**Sanitized items (${(l2.items || []).length} row(s)):**\n\n`;
fileMd += mdTable( fileMd += mdTable(
['kodeBarang', 'namaBarang', 'banyak', 'jumlah'], ['kodeBarang', 'namaBarang', 'banyak', 'jumlah'],
(l2.items || []).map(it => [it.kodeBarang, it.namaBarang, it.banyak, it.jumlah]) (l2.items || []).map(it => [it.kodeBarang, it.namaBarang, it.banyak, it.jumlah])
); );
} else { } else {
fileMd += '*Layer 2 data not captured.*\n\n'; fileMd += '*Layer 2 data not captured.*\n\n';
} }
// ── Stage 4: Layer 3 — Final (SKU triple-check + store resolution) ──────── // ── Stage 4: Layer 3 — Final (SKU triple-check + store resolution) ────────
fileMd += `#### ✅ Stage 4 — Layer 3: Final (\`SKU triple-check + store resolution\`)\n\n`; fileMd += `#### ✅ Stage 4 — Layer 3: Final (\`SKU triple-check + store resolution\`)\n\n`;
fileMd += `*SKU validated against master list (score ≥ 0.6 threshold). Items with noise SKU codes are filtered out. Store resolved from DB.*\n\n`; fileMd += `*SKU validated against master list (score ≥ 0.6 threshold). Items with noise SKU codes are filtered out. Store resolved from DB.*\n\n`;
if (entry.layer3Final) { if (entry.layer3Final) {
const l3 = entry.layer3Final; const l3 = entry.layer3Final;
fileMd += `**Final metadata:**\n\n`; fileMd += `**Final metadata:**\n\n`;
fileMd += mdField('noDO', l3.noDO); fileMd += mdField('noDO', l3.noDO);
fileMd += mdField('noPO', l3.noPO); fileMd += mdField('noPO', l3.noPO);
fileMd += mdField('noSO', l3.noSO); fileMd += mdField('noSO', l3.noSO);
fileMd += mdField('tanggal', l3.tanggal); fileMd += mdField('tanggal', l3.tanggal);
fileMd += mdField('vendorInfo', l3.vendorInfo); fileMd += mdField('vendorInfo', l3.vendorInfo);
fileMd += mdField('customerInfo', l3.customerInfo); fileMd += mdField('customerInfo', l3.customerInfo);
fileMd += mdField('alamat', l3.alamat); fileMd += mdField('alamat', l3.alamat);
fileMd += mdField('orderUntuk', l3.orderUntuk); fileMd += mdField('orderUntuk', l3.orderUntuk);
fileMd += mdField('platTruk', l3.platTruk); fileMd += mdField('platTruk', l3.platTruk);
fileMd += '\n'; fileMd += '\n';
fileMd += `**Final items after SKU validation (${(l3.items || []).length} row(s)):**\n\n`; fileMd += `**Final items after SKU validation (${(l3.items || []).length} row(s)):**\n\n`;
fileMd += mdTable( fileMd += mdTable(
['kodeBarangOriginal', 'kodeBarang (corrected)', 'namaBarang', 'banyak', 'jumlah'], ['kodeBarangOriginal', 'kodeBarang (corrected)', 'namaBarang', 'banyak', 'jumlah'],
(l3.items || []).map(it => [ (l3.items || []).map(it => [
it.kodeBarangOriginal ?? it.kodeBarang, it.kodeBarangOriginal ?? it.kodeBarang,
it.kodeBarang, it.kodeBarang,
it.namaBarang, it.namaBarang,
it.banyak, it.banyak,
it.jumlah it.jumlah
]) ])
); );
} else { } else {
fileMd += '*Layer 3 data not captured.*\n\n'; fileMd += '*Layer 3 data not captured.*\n\n';
} }
// ── Stage 5: Final submitted items (from root items[]) ─────────────────── // ── Stage 5: Final submitted items (from root items[]) ───────────────────
fileMd += `#### 🗃️ Stage 5 — Submitted Items (ready-to-use JSON)\n\n`; fileMd += `#### 🗃️ Stage 5 — Submitted Items (ready-to-use JSON)\n\n`;
fileMd += `*These are the items actually returned to the caller and saved to the database.*\n\n`; fileMd += `*These are the items actually returned to the caller and saved to the database.*\n\n`;
fileMd += mdTable( fileMd += mdTable(
['kodeBarangOriginal', 'kodeBarang', 'namaBarang', 'banyak', 'jumlah'], ['kodeBarangOriginal', 'kodeBarang', 'namaBarang', 'banyak', 'jumlah'],
(entry.items || []).map(it => [ (entry.items || []).map(it => [
it.kodeBarangOriginal ?? it.kodeBarang, it.kodeBarangOriginal ?? it.kodeBarang,
it.kodeBarang, it.kodeBarang,
it.namaBarang, it.namaBarang,
it.banyak, it.banyak,
it.jumlah it.jumlah
]) ])
); );
fileMd += `---\n\n`; fileMd += `---\n\n`;
// ── Summary row ────────────────────────────────────────────────────────── // ── Summary row ──────────────────────────────────────────────────────────
const l3meta = entry.layer3Final || {}; const l3meta = entry.layer3Final || {};
summaryRows.push([ summaryRows.push([
idx + 1, idx + 1,
`\`${file}\``, `\`${file}\``,
`${statusEmoji} success`, `${statusEmoji} success`,
entry.tilt !== null ? `${entry.tilt}°` : 'N/A', entry.tilt !== null ? `${entry.tilt}°` : 'N/A',
entry.unwarped !== null ? (entry.unwarped ? 'Yes' : 'No') : 'N/A', entry.unwarped !== null ? (entry.unwarped ? 'Yes' : 'No') : 'N/A',
`\`${l3meta.noDO ?? 'N/A'}\``, `\`${l3meta.noDO ?? 'N/A'}\``,
`\`${l3meta.noPO ?? 'N/A'}\``, `\`${l3meta.noPO ?? 'N/A'}\``,
`${entry.items.length}`, `${entry.items.length}`,
]); ]);
mdParts.push(fileMd); mdParts.push(fileMd);
} }
// ── Inject summary table ────────────────────────────────────────────────── // ── Inject summary table ──────────────────────────────────────────────────
const summaryTable = mdTable( const summaryTable = mdTable(
['#', 'Filename', 'Status', 'Tilt', 'Unwarped', 'DO', 'PO', 'Items'], ['#', 'Filename', 'Status', 'Tilt', 'Unwarped', 'DO', 'PO', 'Items'],
summaryRows summaryRows
); );
mdParts[summaryPlaceholderIndex] = summaryTable; mdParts[summaryPlaceholderIndex] = summaryTable;
// ── Write outputs ───────────────────────────────────────────────────────── // ── Write outputs ─────────────────────────────────────────────────────────
const jsonOut = JSON.stringify(jsonResults, null, 2); const jsonOut = JSON.stringify(jsonResults, null, 2);
fs.writeFileSync(OUTPUT_JSON, jsonOut, 'utf8'); fs.writeFileSync(OUTPUT_JSON, jsonOut, 'utf8');
console.log(`\n✅ JSON saved → ${OUTPUT_JSON}`); console.log(`\n✅ JSON saved → ${OUTPUT_JSON}`);
const mdOut = mdParts.join(''); const mdOut = mdParts.join('');
fs.writeFileSync(OUTPUT_MD, mdOut, 'utf8'); fs.writeFileSync(OUTPUT_MD, mdOut, 'utf8');
console.log(`✅ MD saved → ${OUTPUT_MD}`); console.log(`✅ MD saved → ${OUTPUT_MD}`);
// ── Final stats ─────────────────────────────────────────────────────────── // ── Final stats ───────────────────────────────────────────────────────────
const succeeded = jsonResults.filter(r => r.status === 'success').length; const succeeded = jsonResults.filter(r => r.status === 'success').length;
const failed = jsonResults.length - succeeded; const failed = jsonResults.length - succeeded;
console.log('\n─'.repeat(60)); console.log('\n─'.repeat(60));
console.log(` Total : ${jsonResults.length}`); console.log(` Total : ${jsonResults.length}`);
console.log(` Success: ${succeeded}`); console.log(` Success: ${succeeded}`);
console.log(` Failed : ${failed}`); console.log(` Failed : ${failed}`);
console.log('─'.repeat(60)); console.log('─'.repeat(60));
} }
main().catch(err => { main().catch(err => {
console.error('Fatal error:', err); console.error('Fatal error:', err);
process.exit(1); process.exit(1);
}); });
File diff suppressed because it is too large. Load diff
@@ -1,419 +1,419 @@
"use client"; "use client";
import React, { useState, useEffect } from "react"; import React, { useState, useEffect } from "react";
export default function MasterDataPage() { export default function MasterDataPage() {
const [token, setToken] = useState<string | null>(null); const [token, setToken] = useState<string | null>(null);
const [username, setUsername] = useState(""); const [username, setUsername] = useState("");
const [password, setPassword] = useState(""); const [password, setPassword] = useState("");
const [loginError, setLoginError] = useState(""); const [loginError, setLoginError] = useState("");
const [activeTab, setActiveTab] = useState<"stores" | "skus">("stores"); const [activeTab, setActiveTab] = useState<"stores" | "skus">("stores");
const [stores, setStores] = useState<any[]>([]); const [stores, setStores] = useState<any[]>([]);
const [skus, setSkus] = useState<any[]>([]); const [skus, setSkus] = useState<any[]>([]);
useEffect(() => { useEffect(() => {
const savedToken = localStorage.getItem("adminToken"); const savedToken = localStorage.getItem("adminToken");
if (savedToken) { if (savedToken) {
setToken(savedToken); setToken(savedToken);
fetchData(savedToken, activeTab); fetchData(savedToken, activeTab);
} }
}, [activeTab]); }, [activeTab]);
const handleLogin = async (e: React.FormEvent) => { const handleLogin = async (e: React.FormEvent) => {
e.preventDefault(); e.preventDefault();
setLoginError(""); setLoginError("");
try { try {
const res = await fetch("/api/v1/auth/login", { const res = await fetch("/api/v1/auth/login", {
method: "POST", method: "POST",
headers: { "Content-Type": "application/json" }, headers: { "Content-Type": "application/json" },
body: JSON.stringify({ username, password }) body: JSON.stringify({ username, password })
}); });
const data = await res.json(); const data = await res.json();
if (!res.ok) throw new Error(data.message || "Login failed"); if (!res.ok) throw new Error(data.message || "Login failed");
const tokenStr = data.data?.token || data.token; const tokenStr = data.data?.token || data.token;
localStorage.setItem("adminToken", tokenStr); localStorage.setItem("adminToken", tokenStr);
setToken(tokenStr); setToken(tokenStr);
fetchData(tokenStr, activeTab); fetchData(tokenStr, activeTab);
} catch (err: any) { } catch (err: any) {
setLoginError(err.message); setLoginError(err.message);
} }
}; };
const handleLogout = () => { const handleLogout = () => {
localStorage.removeItem("adminToken"); localStorage.removeItem("adminToken");
setToken(null); setToken(null);
}; };
const fetchData = async (authToken: string, tab: "stores" | "skus") => { const fetchData = async (authToken: string, tab: "stores" | "skus") => {
try { try {
const res = await fetch(`/api/v1/master/${tab}`, { const res = await fetch(`/api/v1/master/${tab}`, {
headers: { "Authorization": `Bearer ${authToken}` } headers: { "Authorization": `Bearer ${authToken}` }
}); });
if (res.status === 401 || res.status === 403) { if (res.status === 401 || res.status === 403) {
handleLogout(); handleLogout();
return; return;
} }
const data = await res.json(); const data = await res.json();
if (res.ok) { if (res.ok) {
if (tab === "stores") setStores(data.data || []); if (tab === "stores") setStores(data.data || []);
else setSkus(data.data || []); else setSkus(data.data || []);
} }
} catch (err) { } catch (err) {
console.error(err); console.error(err);
} }
}; };
if (!token) { if (!token) {
return ( return (
<div className="min-h-screen bg-[radial-gradient(ellipse_at_top,_var(--tw-gradient-stops))] from-slate-900 via-slate-950 to-slate-950 flex items-center justify-center p-4"> <div className="min-h-screen bg-[radial-gradient(ellipse_at_top,_var(--tw-gradient-stops))] from-slate-900 via-slate-950 to-slate-950 flex items-center justify-center p-4">
<div className="bg-slate-900/40 border border-slate-800/80 shadow-2xl backdrop-blur-md rounded-2xl p-8 w-full max-w-md"> <div className="bg-slate-900/40 border border-slate-800/80 shadow-2xl backdrop-blur-md rounded-2xl p-8 w-full max-w-md">
<h1 className="text-2xl font-extrabold bg-gradient-to-r from-teal-400 to-emerald-400 bg-clip-text text-transparent text-center mb-6"> <h1 className="text-2xl font-extrabold bg-gradient-to-r from-teal-400 to-emerald-400 bg-clip-text text-transparent text-center mb-6">
Admin Login Admin Login
</h1> </h1>
<form onSubmit={handleLogin} className="space-y-5"> <form onSubmit={handleLogin} className="space-y-5">
<div> <div>
<label className="block text-xs font-semibold text-slate-400 uppercase tracking-wider mb-2">Username</label> <label className="block text-xs font-semibold text-slate-400 uppercase tracking-wider mb-2">Username</label>
<input <input
type="text" type="text"
className="w-full bg-slate-950/60 border border-slate-800 text-slate-100 placeholder-slate-600 rounded-xl p-3 focus:outline-none focus:border-teal-500/60 focus:ring-1 focus:ring-teal-500/60 transition-all duration-200 text-sm" className="w-full bg-slate-950/60 border border-slate-800 text-slate-100 placeholder-slate-600 rounded-xl p-3 focus:outline-none focus:border-teal-500/60 focus:ring-1 focus:ring-teal-500/60 transition-all duration-200 text-sm"
value={username} value={username}
onChange={e => setUsername(e.target.value)} onChange={e => setUsername(e.target.value)}
placeholder="Enter admin username" placeholder="Enter admin username"
/> />
</div> </div>
<div> <div>
<label className="block text-xs font-semibold text-slate-400 uppercase tracking-wider mb-2">Password</label> <label className="block text-xs font-semibold text-slate-400 uppercase tracking-wider mb-2">Password</label>
<input <input
type="password" type="password"
className="w-full bg-slate-950/60 border border-slate-800 text-slate-100 placeholder-slate-650 rounded-xl p-3 focus:outline-none focus:border-teal-500/60 focus:ring-1 focus:ring-teal-500/60 transition-all duration-200 text-sm" className="w-full bg-slate-950/60 border border-slate-800 text-slate-100 placeholder-slate-650 rounded-xl p-3 focus:outline-none focus:border-teal-500/60 focus:ring-1 focus:ring-teal-500/60 transition-all duration-200 text-sm"
value={password} value={password}
onChange={e => setPassword(e.target.value)} onChange={e => setPassword(e.target.value)}
placeholder="••••••••" placeholder="••••••••"
/> />
</div> </div>
{loginError && ( {loginError && (
<div className="bg-rose-950/30 border border-rose-800/40 p-3 rounded-xl text-xs text-rose-450 flex items-center gap-2"> <div className="bg-rose-950/30 border border-rose-800/40 p-3 rounded-xl text-xs text-rose-450 flex items-center gap-2">
<span>⚠️</span> <span>⚠️</span>
<span>{loginError}</span> <span>{loginError}</span>
</div> </div>
)} )}
<button <button
type="submit" type="submit"
className="w-full bg-teal-600 hover:bg-teal-500 text-slate-950 font-bold p-3 rounded-xl transition-all duration-200 shadow-lg shadow-teal-900/20 text-sm cursor-pointer" className="w-full bg-teal-600 hover:bg-teal-500 text-slate-950 font-bold p-3 rounded-xl transition-all duration-200 shadow-lg shadow-teal-900/20 text-sm cursor-pointer"
> >
Log In Log In
</button> </button>
</form> </form>
</div> </div>
</div> </div>
); );
} }
return ( return (
<div className="min-h-screen bg-[radial-gradient(ellipse_at_top,_var(--tw-gradient-stops))] from-slate-900 via-slate-950 to-slate-950 p-8 text-slate-100"> <div className="min-h-screen bg-[radial-gradient(ellipse_at_top,_var(--tw-gradient-stops))] from-slate-900 via-slate-950 to-slate-950 p-8 text-slate-100">
<div className="max-w-6xl mx-auto"> <div className="max-w-6xl mx-auto">
<div className="flex justify-between items-center mb-8 border-b border-slate-800/60 pb-4"> <div className="flex justify-between items-center mb-8 border-b border-slate-800/60 pb-4">
<h1 className="text-2xl font-extrabold bg-gradient-to-r from-teal-400 to-emerald-400 bg-clip-text text-transparent flex items-center gap-2"> <h1 className="text-2xl font-extrabold bg-gradient-to-r from-teal-400 to-emerald-400 bg-clip-text text-transparent flex items-center gap-2">
<span>⚙️</span> Master Data Management <span>⚙️</span> Master Data Management
</h1> </h1>
<button <button
onClick={handleLogout} onClick={handleLogout}
className="text-slate-400 hover:text-slate-100 bg-slate-900/60 hover:bg-slate-900 border border-slate-850 px-4 py-2 rounded-xl text-xs font-semibold transition-all duration-200 cursor-pointer" className="text-slate-400 hover:text-slate-100 bg-slate-900/60 hover:bg-slate-900 border border-slate-850 px-4 py-2 rounded-xl text-xs font-semibold transition-all duration-200 cursor-pointer"
> >
Logout Logout
</button> </button>
</div> </div>
<div className="flex space-x-2 mb-6 border-b border-slate-800/60 pb-px"> <div className="flex space-x-2 mb-6 border-b border-slate-800/60 pb-px">
<button <button
className={`pb-2.5 px-4 text-sm font-semibold transition-all border-b-2 -mb-px cursor-pointer ${ className={`pb-2.5 px-4 text-sm font-semibold transition-all border-b-2 -mb-px cursor-pointer ${
activeTab === 'stores' activeTab === 'stores'
? 'border-teal-500 text-teal-400' ? 'border-teal-500 text-teal-400'
: 'border-transparent text-slate-400 hover:text-slate-200' : 'border-transparent text-slate-400 hover:text-slate-200'
}`} }`}
onClick={() => setActiveTab('stores')} onClick={() => setActiveTab('stores')}
> >
Stores Stores
</button> </button>
<button <button
className={`pb-2.5 px-4 text-sm font-semibold transition-all border-b-2 -mb-px cursor-pointer ${ className={`pb-2.5 px-4 text-sm font-semibold transition-all border-b-2 -mb-px cursor-pointer ${
activeTab === 'skus' activeTab === 'skus'
? 'border-teal-500 text-teal-400' ? 'border-teal-500 text-teal-400'
: 'border-transparent text-slate-400 hover:text-slate-200' : 'border-transparent text-slate-400 hover:text-slate-200'
}`} }`}
onClick={() => setActiveTab('skus')} onClick={() => setActiveTab('skus')}
> >
SKUs SKUs
</button> </button>
</div> </div>
<div className="bg-slate-900/40 border border-slate-850 rounded-2xl p-6 shadow-xl backdrop-blur-md"> <div className="bg-slate-900/40 border border-slate-850 rounded-2xl p-6 shadow-xl backdrop-blur-md">
{activeTab === 'stores' && <StoreManager stores={stores} token={token} onRefresh={() => fetchData(token, 'stores')} />} {activeTab === 'stores' && <StoreManager stores={stores} token={token} onRefresh={() => fetchData(token, 'stores')} />}
{activeTab === 'skus' && <SkuManager skus={skus} token={token} onRefresh={() => fetchData(token, 'skus')} />} {activeTab === 'skus' && <SkuManager skus={skus} token={token} onRefresh={() => fetchData(token, 'skus')} />}
</div> </div>
</div> </div>
</div> </div>
); );
} }
function StoreManager({ stores, token, onRefresh }: { stores: any[], token: string, onRefresh: () => void }) { function StoreManager({ stores, token, onRefresh }: { stores: any[], token: string, onRefresh: () => void }) {
const [isAdding, setIsAdding] = useState(false); const [isAdding, setIsAdding] = useState(false);
const [form, setForm] = useState({ kode_toko: "", nama_toko: "", alamat: "" }); const [form, setForm] = useState({ kode_toko: "", nama_toko: "", alamat: "" });
const [error, setError] = useState(""); const [error, setError] = useState("");
const handleSubmit = async (e: React.FormEvent) => { const handleSubmit = async (e: React.FormEvent) => {
e.preventDefault(); e.preventDefault();
setError(""); setError("");
try { try {
const res = await fetch("/api/v1/master/stores", { const res = await fetch("/api/v1/master/stores", {
method: "POST", method: "POST",
headers: { "Content-Type": "application/json", "Authorization": `Bearer ${token}` }, headers: { "Content-Type": "application/json", "Authorization": `Bearer ${token}` },
body: JSON.stringify(form) body: JSON.stringify(form)
}); });
const data = await res.json(); const data = await res.json();
if (!res.ok) throw new Error(data.message); if (!res.ok) throw new Error(data.message);
setIsAdding(false); setIsAdding(false);
setForm({ kode_toko: "", nama_toko: "", alamat: "" }); setForm({ kode_toko: "", nama_toko: "", alamat: "" });
onRefresh(); onRefresh();
} catch (err: any) { } catch (err: any) {
setError(err.message); setError(err.message);
} }
}; };
const handleDelete = async (kode: string) => { const handleDelete = async (kode: string) => {
if (!confirm(`Delete store ${kode}?`)) return; if (!confirm(`Delete store ${kode}?`)) return;
try { try {
const res = await fetch(`/api/v1/master/stores/${kode}`, { const res = await fetch(`/api/v1/master/stores/${kode}`, {
method: "DELETE", method: "DELETE",
headers: { "Authorization": `Bearer ${token}` } headers: { "Authorization": `Bearer ${token}` }
}); });
if (!res.ok) { if (!res.ok) {
const data = await res.json(); const data = await res.json();
throw new Error(data.message); throw new Error(data.message);
} }
onRefresh(); onRefresh();
} catch (err: any) { } catch (err: any) {
alert(err.message); alert(err.message);
} }
}; };
return ( return (
<div> <div>
<div className="flex justify-between items-center mb-6"> <div className="flex justify-between items-center mb-6">
<h2 className="text-lg font-bold text-slate-200 flex items-center gap-2"> <h2 className="text-lg font-bold text-slate-200 flex items-center gap-2">
<span>🏪</span> Store Master <span>🏪</span> Store Master
</h2> </h2>
<button <button
onClick={() => setIsAdding(true)} onClick={() => setIsAdding(true)}
className="bg-teal-600 hover:bg-teal-500 text-slate-950 px-4 py-2 rounded-xl text-xs font-bold transition-all duration-200 shadow-md shadow-teal-900/10 cursor-pointer" className="bg-teal-600 hover:bg-teal-500 text-slate-950 px-4 py-2 rounded-xl text-xs font-bold transition-all duration-200 shadow-md shadow-teal-900/10 cursor-pointer"
> >
+ Add Store + Add Store
</button> </button>
</div> </div>
{isAdding && ( {isAdding && (
<form onSubmit={handleSubmit} className="mb-6 bg-slate-950/40 p-5 rounded-xl border border-slate-800/60"> <form onSubmit={handleSubmit} className="mb-6 bg-slate-950/40 p-5 rounded-xl border border-slate-800/60">
<h3 className="text-xs font-bold text-slate-450 uppercase tracking-wider mb-4"> <h3 className="text-xs font-bold text-slate-450 uppercase tracking-wider mb-4">
Add New Store <span className="text-[10px] text-teal-500 font-normal lowercase">(Will auto-generate account with "123" password)</span> Add New Store <span className="text-[10px] text-teal-500 font-normal lowercase">(Will auto-generate account with "123" password)</span>
</h3> </h3>
<div className="grid grid-cols-1 md:grid-cols-3 gap-4 mb-4"> <div className="grid grid-cols-1 md:grid-cols-3 gap-4 mb-4">
<input <input
placeholder="Kode Toko" placeholder="Kode Toko"
className="bg-slate-900/60 border border-slate-800/80 text-slate-100 placeholder-slate-600 rounded-lg p-2.5 text-xs focus:outline-none focus:border-teal-500/60" className="bg-slate-900/60 border border-slate-800/80 text-slate-100 placeholder-slate-600 rounded-lg p-2.5 text-xs focus:outline-none focus:border-teal-500/60"
value={form.kode_toko} value={form.kode_toko}
onChange={e => setForm({...form, kode_toko: e.target.value})} onChange={e => setForm({...form, kode_toko: e.target.value})}
required required
/> />
<input <input
placeholder="Nama Toko" placeholder="Nama Toko"
className="bg-slate-900/60 border border-slate-800/80 text-slate-100 placeholder-slate-600 rounded-lg p-2.5 text-xs focus:outline-none focus:border-teal-500/60" className="bg-slate-900/60 border border-slate-800/80 text-slate-100 placeholder-slate-600 rounded-lg p-2.5 text-xs focus:outline-none focus:border-teal-500/60"
value={form.nama_toko} value={form.nama_toko}
onChange={e => setForm({...form, nama_toko: e.target.value})} onChange={e => setForm({...form, nama_toko: e.target.value})}
required required
/> />
<input <input
placeholder="Alamat" placeholder="Alamat"
className="bg-slate-900/60 border border-slate-800/80 text-slate-100 placeholder-slate-600 rounded-lg p-2.5 text-xs focus:outline-none focus:border-teal-500/60" className="bg-slate-900/60 border border-slate-800/80 text-slate-100 placeholder-slate-600 rounded-lg p-2.5 text-xs focus:outline-none focus:border-teal-500/60"
value={form.alamat} value={form.alamat}
onChange={e => setForm({...form, alamat: e.target.value})} onChange={e => setForm({...form, alamat: e.target.value})}
/> />
</div> </div>
{error && <div className="text-rose-450 text-xs mb-3 font-semibold">⚠️ {error}</div>} {error && <div className="text-rose-450 text-xs mb-3 font-semibold">⚠️ {error}</div>}
<div className="flex space-x-2"> <div className="flex space-x-2">
<button type="submit" className="bg-emerald-600 hover:bg-emerald-500 text-slate-950 font-bold px-4 py-2 rounded-lg text-xs transition-colors cursor-pointer">Save</button> <button type="submit" className="bg-emerald-600 hover:bg-emerald-500 text-slate-950 font-bold px-4 py-2 rounded-lg text-xs transition-colors cursor-pointer">Save</button>
<button type="button" onClick={() => setIsAdding(false)} className="bg-slate-800 hover:bg-slate-750 text-slate-300 px-4 py-2 rounded-lg text-xs transition-colors cursor-pointer">Cancel</button> <button type="button" onClick={() => setIsAdding(false)} className="bg-slate-800 hover:bg-slate-750 text-slate-300 px-4 py-2 rounded-lg text-xs transition-colors cursor-pointer">Cancel</button>
</div> </div>
</form> </form>
)} )}
<div className="overflow-x-auto rounded-xl border border-slate-800/60"> <div className="overflow-x-auto rounded-xl border border-slate-800/60">
<table className="w-full text-left text-xs border-collapse"> <table className="w-full text-left text-xs border-collapse">
<thead className="bg-slate-950/60 border-b border-slate-800/80 text-slate-400"> <thead className="bg-slate-950/60 border-b border-slate-800/80 text-slate-400">
<tr> <tr>
<th className="p-3.5 font-bold font-mono text-[10px] tracking-wider uppercase">Kode Toko</th> <th className="p-3.5 font-bold font-mono text-[10px] tracking-wider uppercase">Kode Toko</th>
<th className="p-3.5 font-bold font-mono text-[10px] tracking-wider uppercase">Nama Toko</th> <th className="p-3.5 font-bold font-mono text-[10px] tracking-wider uppercase">Nama Toko</th>
<th className="p-3.5 font-bold font-mono text-[10px] tracking-wider uppercase">Alamat</th> <th className="p-3.5 font-bold font-mono text-[10px] tracking-wider uppercase">Alamat</th>
<th className="p-3.5 font-bold font-mono text-[10px] tracking-wider uppercase w-24">Actions</th> <th className="p-3.5 font-bold font-mono text-[10px] tracking-wider uppercase w-24">Actions</th>
</tr> </tr>
</thead> </thead>
<tbody> <tbody>
{stores.map(s => ( {stores.map(s => (
<tr key={s.kode_toko} className="border-b border-slate-850/50 hover:bg-slate-900/30 transition-colors"> <tr key={s.kode_toko} className="border-b border-slate-850/50 hover:bg-slate-900/30 transition-colors">
<td className="p-3.5 font-semibold text-slate-200 font-mono">{s.kode_toko}</td> <td className="p-3.5 font-semibold text-slate-200 font-mono">{s.kode_toko}</td>
<td className="p-3.5 text-slate-300 font-medium">{s.nama_toko}</td> <td className="p-3.5 text-slate-300 font-medium">{s.nama_toko}</td>
<td className="p-3.5 text-slate-400 truncate max-w-xs">{s.alamat}</td> <td className="p-3.5 text-slate-400 truncate max-w-xs">{s.alamat}</td>
<td className="p-3.5"> <td className="p-3.5">
<button <button
onClick={() => handleDelete(s.kode_toko)} onClick={() => handleDelete(s.kode_toko)}
className="text-rose-400 hover:text-rose-355 transition-colors font-bold cursor-pointer font-mono" className="text-rose-400 hover:text-rose-355 transition-colors font-bold cursor-pointer font-mono"
> >
Delete Delete
</button> </button>
</td> </td>
</tr> </tr>
))} ))}
{stores.length === 0 && ( {stores.length === 0 && (
<tr><td colSpan={4} className="p-6 text-center text-slate-500">No stores found.</td></tr> <tr><td colSpan={4} className="p-6 text-center text-slate-500">No stores found.</td></tr>
)} )}
</tbody> </tbody>
</table> </table>
</div> </div>
</div> </div>
); );
} }
function SkuManager({ skus, token, onRefresh }: { skus: any[], token: string, onRefresh: () => void }) { function SkuManager({ skus, token, onRefresh }: { skus: any[], token: string, onRefresh: () => void }) {
const [isAdding, setIsAdding] = useState(false); const [isAdding, setIsAdding] = useState(false);
const [form, setForm] = useState({ no_sku: "", nama_item: "", jenis_outer: "", standar_jumlah: "1" }); const [form, setForm] = useState({ no_sku: "", nama_item: "", jenis_outer: "", standar_jumlah: "1" });
const [error, setError] = useState(""); const [error, setError] = useState("");
const handleSubmit = async (e: React.FormEvent) => { const handleSubmit = async (e: React.FormEvent) => {
e.preventDefault(); e.preventDefault();
setError(""); setError("");
try { try {
const payload = { ...form, standar_jumlah: parseInt(form.standar_jumlah) || 1 }; const payload = { ...form, standar_jumlah: parseInt(form.standar_jumlah) || 1 };
const res = await fetch("/api/v1/master/skus", { const res = await fetch("/api/v1/master/skus", {
method: "POST", method: "POST",
headers: { "Content-Type": "application/json", "Authorization": `Bearer ${token}` }, headers: { "Content-Type": "application/json", "Authorization": `Bearer ${token}` },
body: JSON.stringify(payload) body: JSON.stringify(payload)
}); });
const data = await res.json(); const data = await res.json();
if (!res.ok) throw new Error(data.message); if (!res.ok) throw new Error(data.message);
setIsAdding(false); setIsAdding(false);
setForm({ no_sku: "", nama_item: "", jenis_outer: "", standar_jumlah: "1" }); setForm({ no_sku: "", nama_item: "", jenis_outer: "", standar_jumlah: "1" });
onRefresh(); onRefresh();
} catch (err: any) { } catch (err: any) {
setError(err.message); setError(err.message);
} }
}; };
const handleDelete = async (kode: string) => { const handleDelete = async (kode: string) => {
if (!confirm(`Delete SKU ${kode}?`)) return; if (!confirm(`Delete SKU ${kode}?`)) return;
try { try {
const res = await fetch(`/api/v1/master/skus/${kode}`, { const res = await fetch(`/api/v1/master/skus/${kode}`, {
method: "DELETE", method: "DELETE",
headers: { "Authorization": `Bearer ${token}` } headers: { "Authorization": `Bearer ${token}` }
}); });
if (!res.ok) { if (!res.ok) {
const data = await res.json(); const data = await res.json();
throw new Error(data.message); throw new Error(data.message);
} }
onRefresh(); onRefresh();
} catch (err: any) { } catch (err: any) {
alert(err.message); alert(err.message);
} }
}; };
return ( return (
<div> <div>
<div className="flex justify-between items-center mb-6"> <div className="flex justify-between items-center mb-6">
<h2 className="text-lg font-bold text-slate-200 flex items-center gap-2"> <h2 className="text-lg font-bold text-slate-200 flex items-center gap-2">
<span>📦</span> SKU Master <span>📦</span> SKU Master
</h2> </h2>
<button <button
onClick={() => setIsAdding(true)} onClick={() => setIsAdding(true)}
className="bg-teal-600 hover:bg-teal-500 text-slate-950 px-4 py-2 rounded-xl text-xs font-bold transition-all duration-200 shadow-md shadow-teal-900/10 cursor-pointer" className="bg-teal-600 hover:bg-teal-500 text-slate-950 px-4 py-2 rounded-xl text-xs font-bold transition-all duration-200 shadow-md shadow-teal-900/10 cursor-pointer"
> >
+ Add SKU + Add SKU
</button> </button>
</div> </div>
{isAdding && ( {isAdding && (
<form onSubmit={handleSubmit} className="mb-6 bg-slate-950/40 p-5 rounded-xl border border-slate-800/60"> <form onSubmit={handleSubmit} className="mb-6 bg-slate-950/40 p-5 rounded-xl border border-slate-800/60">
<h3 className="text-xs font-bold text-slate-450 uppercase tracking-wider mb-4 font-mono">Add New SKU</h3> <h3 className="text-xs font-bold text-slate-450 uppercase tracking-wider mb-4 font-mono">Add New SKU</h3>
<div className="grid grid-cols-2 md:grid-cols-4 gap-4 mb-4"> <div className="grid grid-cols-2 md:grid-cols-4 gap-4 mb-4">
<input <input
placeholder="No SKU" placeholder="No SKU"
className="bg-slate-900/60 border border-slate-800/80 text-slate-100 placeholder-slate-600 rounded-lg p-2.5 text-xs focus:outline-none focus:border-teal-500/60" className="bg-slate-900/60 border border-slate-800/80 text-slate-100 placeholder-slate-600 rounded-lg p-2.5 text-xs focus:outline-none focus:border-teal-500/60"
value={form.no_sku} value={form.no_sku}
onChange={e => setForm({...form, no_sku: e.target.value})} onChange={e => setForm({...form, no_sku: e.target.value})}
required required
/> />
<input <input
placeholder="Nama Item" placeholder="Nama Item"
className="bg-slate-900/60 border border-slate-800/80 text-slate-100 placeholder-slate-600 rounded-lg p-2.5 text-xs focus:outline-none focus:border-teal-500/60" className="bg-slate-900/60 border border-slate-800/80 text-slate-100 placeholder-slate-600 rounded-lg p-2.5 text-xs focus:outline-none focus:border-teal-500/60"
value={form.nama_item} value={form.nama_item}
onChange={e => setForm({...form, nama_item: e.target.value})} onChange={e => setForm({...form, nama_item: e.target.value})}
required required
/> />
<input <input
placeholder="Jenis Outer (e.g. DUS)" placeholder="Jenis Outer (e.g. DUS)"
className="bg-slate-900/60 border border-slate-800/80 text-slate-100 placeholder-slate-600 rounded-lg p-2.5 text-xs focus:outline-none focus:border-teal-500/60" className="bg-slate-900/60 border border-slate-800/80 text-slate-100 placeholder-slate-600 rounded-lg p-2.5 text-xs focus:outline-none focus:border-teal-500/60"
value={form.jenis_outer} value={form.jenis_outer}
onChange={e => setForm({...form, jenis_outer: e.target.value})} onChange={e => setForm({...form, jenis_outer: e.target.value})}
/> />
<input <input
type="number" type="number"
placeholder="Std Qty" placeholder="Std Qty"
className="bg-slate-900/60 border border-slate-800/80 text-slate-100 placeholder-slate-600 rounded-lg p-2.5 text-xs focus:outline-none focus:border-teal-500/60" className="bg-slate-900/60 border border-slate-800/80 text-slate-100 placeholder-slate-600 rounded-lg p-2.5 text-xs focus:outline-none focus:border-teal-500/60"
value={form.standar_jumlah} value={form.standar_jumlah}
onChange={e => setForm({...form, standar_jumlah: e.target.value})} onChange={e => setForm({...form, standar_jumlah: e.target.value})}
/> />
</div> </div>
{error && <div className="text-rose-450 text-xs mb-3 font-semibold">⚠️ {error}</div>} {error && <div className="text-rose-450 text-xs mb-3 font-semibold">⚠️ {error}</div>}
<div className="flex space-x-2"> <div className="flex space-x-2">
<button type="submit" className="bg-emerald-600 hover:bg-emerald-500 text-slate-950 font-bold px-4 py-2 rounded-lg text-xs transition-colors cursor-pointer">Save</button> <button type="submit" className="bg-emerald-600 hover:bg-emerald-500 text-slate-950 font-bold px-4 py-2 rounded-lg text-xs transition-colors cursor-pointer">Save</button>
<button type="button" onClick={() => setIsAdding(false)} className="bg-slate-800 hover:bg-slate-750 text-slate-300 px-4 py-2 rounded-lg text-xs transition-colors cursor-pointer">Cancel</button> <button type="button" onClick={() => setIsAdding(false)} className="bg-slate-800 hover:bg-slate-750 text-slate-300 px-4 py-2 rounded-lg text-xs transition-colors cursor-pointer">Cancel</button>
</div> </div>
</form> </form>
)} )}
<div className="overflow-x-auto rounded-xl border border-slate-800/60"> <div className="overflow-x-auto rounded-xl border border-slate-800/60">
<table className="w-full text-left text-xs border-collapse"> <table className="w-full text-left text-xs border-collapse">
<thead className="bg-slate-950/60 border-b border-slate-800/80 text-slate-400"> <thead className="bg-slate-950/60 border-b border-slate-800/80 text-slate-400">
<tr> <tr>
<th className="p-3.5 font-bold font-mono text-[10px] tracking-wider uppercase">No SKU</th> <th className="p-3.5 font-bold font-mono text-[10px] tracking-wider uppercase">No SKU</th>
<th className="p-3.5 font-bold font-mono text-[10px] tracking-wider uppercase">Nama Item</th> <th className="p-3.5 font-bold font-mono text-[10px] tracking-wider uppercase">Nama Item</th>
<th className="p-3.5 font-bold font-mono text-[10px] tracking-wider uppercase">Outer</th> <th className="p-3.5 font-bold font-mono text-[10px] tracking-wider uppercase">Outer</th>
<th className="p-3.5 font-bold font-mono text-[10px] tracking-wider uppercase">Std Qty</th> <th className="p-3.5 font-bold font-mono text-[10px] tracking-wider uppercase">Std Qty</th>
<th className="p-3.5 font-bold font-mono text-[10px] tracking-wider uppercase w-24">Actions</th> <th className="p-3.5 font-bold font-mono text-[10px] tracking-wider uppercase w-24">Actions</th>
</tr> </tr>
</thead> </thead>
<tbody> <tbody>
{skus.map(s => ( {skus.map(s => (
<tr key={s.no_sku} className="border-b border-slate-850/50 hover:bg-slate-900/30 transition-colors"> <tr key={s.no_sku} className="border-b border-slate-850/50 hover:bg-slate-900/30 transition-colors">
<td className="p-3.5 font-semibold text-slate-200 font-mono">{s.no_sku}</td> <td className="p-3.5 font-semibold text-slate-200 font-mono">{s.no_sku}</td>
<td className="p-3.5 text-slate-300 font-medium">{s.nama_item}</td> <td className="p-3.5 text-slate-300 font-medium">{s.nama_item}</td>
<td className="p-3.5 text-slate-400 font-mono">{s.jenis_outer}</td> <td className="p-3.5 text-slate-400 font-mono">{s.jenis_outer}</td>
<td className="p-3.5 text-slate-400 font-mono">{s.standar_jumlah}</td> <td className="p-3.5 text-slate-400 font-mono">{s.standar_jumlah}</td>
<td className="p-3.5"> <td className="p-3.5">
<button <button
onClick={() => handleDelete(s.no_sku)} onClick={() => handleDelete(s.no_sku)}
className="text-rose-400 hover:text-rose-350 transition-colors font-bold cursor-pointer font-mono" className="text-rose-400 hover:text-rose-350 transition-colors font-bold cursor-pointer font-mono"
> >
Delete Delete
</button> </button>
</td> </td>
</tr> </tr>
))} ))}
{skus.length === 0 && ( {skus.length === 0 && (
<tr><td colSpan={5} className="p-6 text-center text-slate-500">No SKUs found.</td></tr> <tr><td colSpan={5} className="p-6 text-center text-slate-500">No SKUs found.</td></tr>
)} )}
</tbody> </tbody>
</table> </table>
</div> </div>
</div> </div>
); );
} }
+265 -265
View File
@@ -1,265 +1,265 @@
import { NextRequest, NextResponse } from "next/server"; import { NextRequest, NextResponse } from "next/server";
import fs from "fs"; import fs from "fs";
import path from "path"; import path from "path";
import { Client } from "@gradio/client"; import { Client } from "@gradio/client";
import { query } from "../../../db"; import { query } from "../../../db";
import { errorResponse } from "@/utils/api-error"; import { errorResponse } from "@/utils/api-error";
export const maxDuration = 120; // Allow up to 120 seconds for slow model inference export const maxDuration = 120; // Allow up to 120 seconds for slow model inference
export async function GET(req: NextRequest) { export async function GET(req: NextRequest) {
try { try {
const { searchParams } = new URL(req.url); const { searchParams } = new URL(req.url);
const action = searchParams.get("action") || "list"; const action = searchParams.get("action") || "list";
const runId = searchParams.get("runId"); const runId = searchParams.get("runId");
const imageType = searchParams.get("imageType"); // 'do', 'product' or null for all const imageType = searchParams.get("imageType"); // 'do', 'product' or null for all
if (runId) { if (runId) {
const runRes = await query(` const runRes = await query(`
SELECT id, image_path, engine, status, ocr_result, time_elapsed_ms, image_type, created_at SELECT id, image_path, engine, status, ocr_result, time_elapsed_ms, image_type, created_at
FROM arena_runs FROM arena_runs
WHERE id = $1 WHERE id = $1
`, [parseInt(runId)]); `, [parseInt(runId)]);
if (runRes.rowCount === 0) { if (runRes.rowCount === 0) {
return errorResponse(404, "Run not found"); return errorResponse(404, "Run not found");
} }
return NextResponse.json({ success: true, run: runRes.rows[0] }); return NextResponse.json({ success: true, run: runRes.rows[0] });
} }
if (action === "stats") { if (action === "stats") {
let queryText = ` let queryText = `
SELECT SELECT
engine, engine,
COUNT(*)::integer as total_runs, COUNT(*)::integer as total_runs,
COUNT(CASE WHEN status = 'done' THEN 1 END)::integer as success_runs, COUNT(CASE WHEN status = 'done' THEN 1 END)::integer as success_runs,
COUNT(CASE WHEN status = 'failed' THEN 1 END)::integer as failed_runs, COUNT(CASE WHEN status = 'failed' THEN 1 END)::integer as failed_runs,
ROUND(AVG(CASE WHEN status = 'done' THEN time_elapsed_ms END))::integer as avg_time_ms, ROUND(AVG(CASE WHEN status = 'done' THEN time_elapsed_ms END))::integer as avg_time_ms,
MIN(CASE WHEN status = 'done' THEN time_elapsed_ms END)::integer as min_time_ms, MIN(CASE WHEN status = 'done' THEN time_elapsed_ms END)::integer as min_time_ms,
MAX(CASE WHEN status = 'done' THEN time_elapsed_ms END)::integer as max_time_ms MAX(CASE WHEN status = 'done' THEN time_elapsed_ms END)::integer as max_time_ms
FROM arena_runs FROM arena_runs
`; `;
const params: any[] = []; const params: any[] = [];
if (imageType === "do" || imageType === "product") { if (imageType === "do" || imageType === "product") {
queryText += ` WHERE image_type = $1`; queryText += ` WHERE image_type = $1`;
params.push(imageType); params.push(imageType);
} }
queryText += ` GROUP BY engine`; queryText += ` GROUP BY engine`;
const statsRes = await query(queryText, params); const statsRes = await query(queryText, params);
return NextResponse.json({ success: true, stats: statsRes.rows }); return NextResponse.json({ success: true, stats: statsRes.rows });
} }
const limit = parseInt(searchParams.get("limit") || "50"); const limit = parseInt(searchParams.get("limit") || "50");
let queryText = ` let queryText = `
SELECT id, image_path, engine, status, time_elapsed_ms, image_type, created_at SELECT id, image_path, engine, status, time_elapsed_ms, image_type, created_at
FROM arena_runs FROM arena_runs
`; `;
const params: any[] = []; const params: any[] = [];
if (imageType === "do" || imageType === "product") { if (imageType === "do" || imageType === "product") {
queryText += ` WHERE image_type = $1`; queryText += ` WHERE image_type = $1`;
params.push(imageType); params.push(imageType);
} }
queryText += ` ORDER BY created_at DESC LIMIT $${params.length + 1}`; queryText += ` ORDER BY created_at DESC LIMIT $${params.length + 1}`;
params.push(limit); params.push(limit);
const runsRes = await query(queryText, params); const runsRes = await query(queryText, params);
return NextResponse.json({ success: true, runs: runsRes.rows }); return NextResponse.json({ success: true, runs: runsRes.rows });
} catch (error: any) { } catch (error: any) {
console.error("Failed to fetch arena runs/stats:", error); console.error("Failed to fetch arena runs/stats:", error);
return errorResponse(500, error.message); return errorResponse(500, error.message);
} }
} }
export async function POST(req: NextRequest) { export async function POST(req: NextRequest) {
const startTime = Date.now(); const startTime = Date.now();
let engine: string | undefined; let engine: string | undefined;
let image: string | undefined; let image: string | undefined;
let imageType = "do"; let imageType = "do";
try { try {
const body = await req.json().catch(() => ({})); const body = await req.json().catch(() => ({}));
engine = body.engine; engine = body.engine;
image = body.image; image = body.image;
if (!engine || !image) { if (!engine || !image) {
return errorResponse(400, "Missing engine or image"); return errorResponse(400, "Missing engine or image");
} }
imageType = body.imageType || "do"; imageType = body.imageType || "do";
if (typeof image === "string") { if (typeof image === "string") {
if (image.startsWith("/produk-pfm/") || image.includes("produk-pfm") || image.includes("Product")) { if (image.startsWith("/produk-pfm/") || image.includes("produk-pfm") || image.includes("Product")) {
imageType = "product"; imageType = "product";
} else if (image.startsWith("/do-pfm/") || image.includes("do-pfm")) { } else if (image.startsWith("/do-pfm/") || image.includes("do-pfm")) {
imageType = "do"; imageType = "do";
} }
} }
let imageBuffer: Buffer; let imageBuffer: Buffer;
let base64Image = ""; let base64Image = "";
// 1. Resolve image (local file or base64) // 1. Resolve image (local file or base64)
if (typeof image === "string" && (image.startsWith("/do-pfm/") || image.startsWith("/produk-pfm/"))) { if (typeof image === "string" && (image.startsWith("/do-pfm/") || image.startsWith("/produk-pfm/"))) {
// Resolve path in public folder // Resolve path in public folder
const cleanPath = image.startsWith("/") ? image.slice(1) : image; const cleanPath = image.startsWith("/") ? image.slice(1) : image;
const filePath = path.join(process.cwd(), "public", cleanPath); const filePath = path.join(process.cwd(), "public", cleanPath);
if (!fs.existsSync(filePath)) { if (!fs.existsSync(filePath)) {
return errorResponse(404, `File not found on server: ${image}`); return errorResponse(404, `File not found on server: ${image}`);
} }
imageBuffer = fs.readFileSync(filePath); imageBuffer = fs.readFileSync(filePath);
base64Image = `data:image/jpeg;base64,${imageBuffer.toString("base64")}`; base64Image = `data:image/jpeg;base64,${imageBuffer.toString("base64")}`;
} else if (typeof image === "string" && image.startsWith("data:")) { } else if (typeof image === "string" && image.startsWith("data:")) {
// Base64 data URI // Base64 data URI
base64Image = image; base64Image = image;
const base64Data = image.split(",")[1]; const base64Data = image.split(",")[1];
imageBuffer = Buffer.from(base64Data, "base64"); imageBuffer = Buffer.from(base64Data, "base64");
} else if (typeof image === "string") { } else if (typeof image === "string") {
// Raw base64 string // Raw base64 string
base64Image = `data:image/jpeg;base64,${image}`; base64Image = `data:image/jpeg;base64,${image}`;
imageBuffer = Buffer.from(image, "base64"); imageBuffer = Buffer.from(image, "base64");
} else { } else {
return errorResponse(400, "Invalid image format"); return errorResponse(400, "Invalid image format");
} }
let outputText = ""; let outputText = "";
// 2. Route to the requested OCR engine // 2. Route to the requested OCR engine
if (engine === "deepseek") { if (engine === "deepseek") {
const blob = new Blob([new Uint8Array(imageBuffer)], { type: "image/jpeg" }); const blob = new Blob([new Uint8Array(imageBuffer)], { type: "image/jpeg" });
const gradioUrl = process.env.DEEPSEEK_GRADIO_URL || "http://host.docker.internal:7873/v2/"; const gradioUrl = process.env.DEEPSEEK_GRADIO_URL || "http://host.docker.internal:7873/v2/";
const client = await Client.connect(gradioUrl); const client = await Client.connect(gradioUrl);
const result = await client.predict(2, [blob, "Default", "Markdown", ""]); const result = await client.predict(2, [blob, "Default", "Markdown", ""]);
const data = result.data as any[]; const data = result.data as any[];
outputText = data[1] || data[0] || ""; outputText = data[1] || data[0] || "";
} else if (engine === "lightonocr") { } else if (engine === "lightonocr") {
const url = process.env.LIGHTONOCR_API_URL || "http://host.docker.internal:7678/layout-parsing"; const url = process.env.LIGHTONOCR_API_URL || "http://host.docker.internal:7678/layout-parsing";
const res = await fetch(url, { const res = await fetch(url, {
method: "POST", method: "POST",
headers: { "Content-Type": "application/json" }, headers: { "Content-Type": "application/json" },
body: JSON.stringify({ body: JSON.stringify({
file: base64Image, file: base64Image,
useLayoutDetection: false useLayoutDetection: false
}) })
}); });
if (!res.ok) { if (!res.ok) {
throw new Error(`LightOnOCR backend error: ${res.status} ${await res.text()}`); throw new Error(`LightOnOCR backend error: ${res.status} ${await res.text()}`);
} }
const data = await res.json(); const data = await res.json();
outputText = data.result?.layoutParsingResults?.[0]?.markdown?.text || ""; outputText = data.result?.layoutParsingResults?.[0]?.markdown?.text || "";
} else if (engine === "nemotron") { } else if (engine === "nemotron") {
const url = process.env.NEMOTRON_API_URL || "http://host.docker.internal:8009/layout-parsing"; const url = process.env.NEMOTRON_API_URL || "http://host.docker.internal:8009/layout-parsing";
const res = await fetch(url, { const res = await fetch(url, {
method: "POST", method: "POST",
headers: { "Content-Type": "application/json" }, headers: { "Content-Type": "application/json" },
body: JSON.stringify({ body: JSON.stringify({
file: base64Image, file: base64Image,
model: "Multilingual (en, zh, ja, ko, ru, …)", model: "Multilingual (en, zh, ja, ko, ru, …)",
merge_level: "layout" merge_level: "layout"
}) })
}); });
if (!res.ok) { if (!res.ok) {
throw new Error(`Nemotron backend error: ${res.status} ${await res.text()}`); throw new Error(`Nemotron backend error: ${res.status} ${await res.text()}`);
} }
const data = await res.json(); const data = await res.json();
outputText = data.result?.layoutParsingResults?.[0]?.markdown?.text || ""; outputText = data.result?.layoutParsingResults?.[0]?.markdown?.text || "";
} else if (engine === "paddle") { } else if (engine === "paddle") {
const url = process.env.PIPELINE_URL || "http://paddleocr-pipeline-api:8090/layout-parsing"; const url = process.env.PIPELINE_URL || "http://paddleocr-pipeline-api:8090/layout-parsing";
const rawB64 = base64Image.includes(",") ? base64Image.split(",")[1] : base64Image; const rawB64 = base64Image.includes(",") ? base64Image.split(",")[1] : base64Image;
const res = await fetch(url, { const res = await fetch(url, {
method: "POST", method: "POST",
headers: { "Content-Type": "application/json" }, headers: { "Content-Type": "application/json" },
body: JSON.stringify({ body: JSON.stringify({
file: rawB64, file: rawB64,
matchHistoryJob: false, matchHistoryJob: false,
useLayoutDetection: true, useLayoutDetection: true,
fileType: 1, fileType: 1,
useDocUnwarping: false, useDocUnwarping: false,
useDocOrientationClassify: false useDocOrientationClassify: false
}) })
}); });
if (!res.ok) { if (!res.ok) {
throw new Error(`PaddleOCR backend error: ${res.status} ${await res.text()}`); throw new Error(`PaddleOCR backend error: ${res.status} ${await res.text()}`);
} }
const data = await res.json(); const data = await res.json();
const pipelineResult = data.result || data; const pipelineResult = data.result || data;
outputText = pipelineResult?.layoutParsingResults?.[0]?.markdown?.text || ""; outputText = pipelineResult?.layoutParsingResults?.[0]?.markdown?.text || "";
} else if (engine === "dots") { } else if (engine === "dots") {
// Calling python API directly // Calling python API directly
const url = process.env.DOTS_API_URL || "http://host.docker.internal:7872/layout-parsing"; const url = process.env.DOTS_API_URL || "http://host.docker.internal:7872/layout-parsing";
const res = await fetch(url, { const res = await fetch(url, {
method: "POST", method: "POST",
headers: { "Content-Type": "application/json" }, headers: { "Content-Type": "application/json" },
body: JSON.stringify({ body: JSON.stringify({
file: base64Image, file: base64Image,
promptLabel: "ocr", promptLabel: "ocr",
useLayoutDetection: true useLayoutDetection: true
}) })
}); });
if (!res.ok) { if (!res.ok) {
throw new Error(`Dots OCR backend error: ${res.status} ${await res.text()}`); throw new Error(`Dots OCR backend error: ${res.status} ${await res.text()}`);
} }
const data = await res.json(); const data = await res.json();
outputText = data.result?.layoutParsingResults?.[0]?.markdown?.text || ""; outputText = data.result?.layoutParsingResults?.[0]?.markdown?.text || "";
} else if (engine === "glm") { } else if (engine === "glm") {
const gradioUrl = process.env.GLM_GRADIO_URL || "http://host.docker.internal:7875/"; const gradioUrl = process.env.GLM_GRADIO_URL || "http://host.docker.internal:7875/";
const client = await Client.connect(gradioUrl); const client = await Client.connect(gradioUrl);
const result = await client.predict(2, ["Text", base64Image, 1024, 60]); const result = await client.predict(2, ["Text", base64Image, 1024, 60]);
const data = result.data as any[]; const data = result.data as any[];
outputText = data[0] || ""; outputText = data[0] || "";
} else { } else {
return errorResponse(400, `Unknown engine: ${engine}`); return errorResponse(400, `Unknown engine: ${engine}`);
} }
const elapsedMs = Date.now() - startTime; const elapsedMs = Date.now() - startTime;
// Record successful run // Record successful run
try { try {
const loggedImagePath = (typeof image === "string" && image.startsWith("data:")) const loggedImagePath = (typeof image === "string" && image.startsWith("data:"))
? `[Base64 Upload: ${image.length} chars]` ? `[Base64 Upload: ${image.length} chars]`
: (typeof image === "string" && image.length > 500) : (typeof image === "string" && image.length > 500)
? `[Raw Base64: ${image.length} chars]` ? `[Raw Base64: ${image.length} chars]`
: image; : image;
await query( await query(
`INSERT INTO arena_runs (image_path, engine, status, ocr_result, time_elapsed_ms, image_type) `INSERT INTO arena_runs (image_path, engine, status, ocr_result, time_elapsed_ms, image_type)
VALUES ($1, $2, $3, $4, $5, $6)`, VALUES ($1, $2, $3, $4, $5, $6)`,
[loggedImagePath, engine, "done", outputText, elapsedMs, imageType] [loggedImagePath, engine, "done", outputText, elapsedMs, imageType]
); );
} catch (dbErr) { } catch (dbErr) {
console.error("Failed to log success to arena_runs:", dbErr); console.error("Failed to log success to arena_runs:", dbErr);
} }
return NextResponse.json({ return NextResponse.json({
success: true, success: true,
text: outputText, text: outputText,
elapsedMs elapsedMs
}); });
} catch (error: any) { } catch (error: any) {
console.error("OCR Arena proxy error:", error); console.error("OCR Arena proxy error:", error);
const elapsedMs = Date.now() - startTime; const elapsedMs = Date.now() - startTime;
// Record failed run // Record failed run
try { try {
const loggedImagePath = (typeof image === "string" && image.startsWith("data:")) const loggedImagePath = (typeof image === "string" && image.startsWith("data:"))
? `[Base64 Upload: ${image.length} chars]` ? `[Base64 Upload: ${image.length} chars]`
: (typeof image === "string" && image.length > 500) : (typeof image === "string" && image.length > 500)
? `[Raw Base64: ${image.length} chars]` ? `[Raw Base64: ${image.length} chars]`
: image; : image;
await query( await query(
`INSERT INTO arena_runs (image_path, engine, status, ocr_result, time_elapsed_ms, image_type) `INSERT INTO arena_runs (image_path, engine, status, ocr_result, time_elapsed_ms, image_type)
VALUES ($1, $2, $3, $4, $5, $6)`, VALUES ($1, $2, $3, $4, $5, $6)`,
[loggedImagePath || "unknown", engine || "unknown", "failed", error.message || "Unknown error", elapsedMs, imageType] [loggedImagePath || "unknown", engine || "unknown", "failed", error.message || "Unknown error", elapsedMs, imageType]
); );
} catch (dbErr) { } catch (dbErr) {
console.error("Failed to log failure to arena_runs:", dbErr); console.error("Failed to log failure to arena_runs:", dbErr);
} }
return errorResponse(500, error.message || "Failed to process OCR request"); return errorResponse(500, error.message || "Failed to process OCR request");
} }
} }
+56 -56
View File
@@ -1,56 +1,56 @@
import { NextRequest, NextResponse } from "next/server"; import { NextRequest, NextResponse } from "next/server";
import fs from "fs"; import fs from "fs";
import path from "path"; import path from "path";
import { query } from "../../../db"; import { query } from "../../../db";
import crypto from "crypto"; import crypto from "crypto";
import { errorResponse } from "@/utils/api-error"; import { errorResponse } from "@/utils/api-error";
const UPLOADS_DIR = "/uploads"; const UPLOADS_DIR = "/uploads";
const PUBLIC_DIR = path.join(process.cwd(), "public/do-pfm"); const PUBLIC_DIR = path.join(process.cwd(), "public/do-pfm");
export async function POST(req: NextRequest) { export async function POST(req: NextRequest) {
try { try {
const { filename, image } = await req.json(); const { filename, image } = await req.json();
if (!filename || !image) { if (!filename || !image) {
return errorResponse(400, "Filename and image base64 data are required"); return errorResponse(400, "Filename and image base64 data are required");
} }
const safeFile = path.basename(filename); const safeFile = path.basename(filename);
const isSample = fs.existsSync(path.join(PUBLIC_DIR, safeFile)); const isSample = fs.existsSync(path.join(PUBLIC_DIR, safeFile));
const filePath = isSample const filePath = isSample
? path.join(PUBLIC_DIR, safeFile) ? path.join(PUBLIC_DIR, safeFile)
: path.join(UPLOADS_DIR, safeFile); : path.join(UPLOADS_DIR, safeFile);
const base64Data = image.replace(/^data:image\/\w+;base64,/, ""); const base64Data = image.replace(/^data:image\/\w+;base64,/, "");
const buffer = Buffer.from(base64Data, "base64"); const buffer = Buffer.from(base64Data, "base64");
// Write file to disk // Write file to disk
fs.writeFileSync(filePath, buffer); fs.writeFileSync(filePath, buffer);
console.log(`Cropped file saved successfully at ${filePath}`); console.log(`Cropped file saved successfully at ${filePath}`);
// Update database fields // Update database fields
const fileHash = crypto.createHash("sha256").update(buffer).digest("hex"); const fileHash = crypto.createHash("sha256").update(buffer).digest("hex");
const stats = fs.statSync(filePath); const stats = fs.statSync(filePath);
// Update document to unparsed state since layout changes // Update document to unparsed state since layout changes
await query( await query(
"UPDATE documents SET size = $1, file_hash = $2, parsed = false, layout_parsing_result = NULL WHERE filename = $3", "UPDATE documents SET size = $1, file_hash = $2, parsed = false, layout_parsing_result = NULL WHERE filename = $3",
[stats.size, fileHash, filename] [stats.size, fileHash, filename]
); );
// Clear old items for this document // Clear old items for this document
const docRes = await query("SELECT id FROM documents WHERE filename = $1", [filename]); const docRes = await query("SELECT id FROM documents WHERE filename = $1", [filename]);
if (docRes.rowCount && docRes.rowCount > 0) { if (docRes.rowCount && docRes.rowCount > 0) {
const docId = docRes.rows[0].id; const docId = docRes.rows[0].id;
await query("DELETE FROM ocr_items WHERE document_id = $1", [docId]); await query("DELETE FROM ocr_items WHERE document_id = $1", [docId]);
} }
return NextResponse.json({ success: true }); return NextResponse.json({ success: true });
} catch (error: unknown) { } catch (error: unknown) {
console.error("Error cropping file:", error); console.error("Error cropping file:", error);
const message = error instanceof Error ? error.message : "Internal server error"; const message = error instanceof Error ? error.message : "Internal server error";
return errorResponse(500, message); return errorResponse(500, message);
} }
} }
@@ -1,38 +1,38 @@
import { NextRequest, NextResponse } from "next/server"; import { NextRequest, NextResponse } from "next/server";
import { query } from "../../../../../db"; import { query } from "../../../../../db";
import { errorResponse } from "@/utils/api-error"; import { errorResponse } from "@/utils/api-error";
export const dynamic = "force-dynamic"; export const dynamic = "force-dynamic";
export async function GET( export async function GET(
req: NextRequest, req: NextRequest,
{ params }: { params: Promise<{ id: string }> | { id: string } } { params }: { params: Promise<{ id: string }> | { id: string } }
) { ) {
try { try {
// Handle both Promise and synchronous params for Next.js version compatibility // Handle both Promise and synchronous params for Next.js version compatibility
const resolvedParams = await params; const resolvedParams = await params;
const { id } = resolvedParams; const { id } = resolvedParams;
const docId = parseInt(id, 10); const docId = parseInt(id, 10);
if (isNaN(docId)) { if (isNaN(docId)) {
return errorResponse(400, "Invalid document ID"); return errorResponse(400, "Invalid document ID");
} }
const res = await query( const res = await query(
"SELECT filename, processing_logs FROM documents WHERE id = $1", "SELECT filename, processing_logs FROM documents WHERE id = $1",
[docId] [docId]
); );
if (res.rowCount === 0 || !res.rows[0]) { if (res.rowCount === 0 || !res.rows[0]) {
return errorResponse(404, "Document not found"); return errorResponse(404, "Document not found");
} }
return NextResponse.json({ return NextResponse.json({
filename: res.rows[0].filename, filename: res.rows[0].filename,
processing_logs: res.rows[0].processing_logs || null processing_logs: res.rows[0].processing_logs || null
}); });
} catch (error: any) { } catch (error: any) {
console.error("Error fetching document logs:", error); console.error("Error fetching document logs:", error);
return errorResponse(500, error.message); return errorResponse(500, error.message);
} }
} }
+47 -47
View File
@@ -1,47 +1,47 @@
import { NextRequest, NextResponse } from "next/server"; import { NextRequest, NextResponse } from "next/server";
import fs from "fs"; import fs from "fs";
import path from "path"; import path from "path";
import { errorResponse } from "@/utils/api-error"; import { errorResponse } from "@/utils/api-error";
const UPLOADS_DIR = "/uploads"; const UPLOADS_DIR = "/uploads";
export async function GET(req: NextRequest) { export async function GET(req: NextRequest) {
try { try {
const filename = req.nextUrl.searchParams.get("file"); const filename = req.nextUrl.searchParams.get("file");
if (!filename) { if (!filename) {
return errorResponse(400, "File name is required"); return errorResponse(400, "File name is required");
} }
const safeFile = path.basename(filename); const safeFile = path.basename(filename);
const filePath = path.join(UPLOADS_DIR, safeFile); const filePath = path.join(UPLOADS_DIR, safeFile);
if (!fs.existsSync(filePath)) { if (!fs.existsSync(filePath)) {
return errorResponse(404, "File not found"); return errorResponse(404, "File not found");
} }
// Determine content type based on extension // Determine content type based on extension
const ext = path.extname(safeFile).toLowerCase(); const ext = path.extname(safeFile).toLowerCase();
let contentType = "application/octet-stream"; let contentType = "application/octet-stream";
if (ext === ".jpg" || ext === ".jpeg") { if (ext === ".jpg" || ext === ".jpeg") {
contentType = "image/jpeg"; contentType = "image/jpeg";
} else if (ext === ".png") { } else if (ext === ".png") {
contentType = "image/png"; contentType = "image/png";
} else if (ext === ".gif") { } else if (ext === ".gif") {
contentType = "image/gif"; contentType = "image/gif";
} else if (ext === ".pdf") { } else if (ext === ".pdf") {
contentType = "application/pdf"; contentType = "application/pdf";
} }
const fileBuffer = fs.readFileSync(filePath); const fileBuffer = fs.readFileSync(filePath);
return new Response(fileBuffer, { return new Response(fileBuffer, {
headers: { headers: {
"Content-Type": contentType, "Content-Type": contentType,
"Cache-Control": "public, max-age=31536000, immutable" "Cache-Control": "public, max-age=31536000, immutable"
} }
}); });
} catch (error: unknown) { } catch (error: unknown) {
console.error("Error serving file from uploads:", error); console.error("Error serving file from uploads:", error);
const message = error instanceof Error ? error.message : "Internal server error"; const message = error instanceof Error ? error.message : "Internal server error";
return errorResponse(500, message); return errorResponse(500, message);
} }
} }
+159 -159
View File
@@ -1,159 +1,159 @@
import { NextRequest, NextResponse } from "next/server"; import { NextRequest, NextResponse } from "next/server";
import { import {
getGpuInfo, getGpuInfo,
getContainerStatus, getContainerStatus,
manageContainer, manageContainer,
recreateContainer, recreateContainer,
getEnvSettings, getEnvSettings,
saveEnvSettings, saveEnvSettings,
getProcessName, getProcessName,
unloadOtherEngines unloadOtherEngines
} from "../../../utils/docker"; } from "../../../utils/docker";
import { errorResponse } from "@/utils/api-error"; import { errorResponse } from "@/utils/api-error";
export async function GET(req: NextRequest) { export async function GET(req: NextRequest) {
try { try {
const gpus = await getGpuInfo(); const gpus = await getGpuInfo();
const settings = await getEnvSettings(); const settings = await getEnvSettings();
const containers = { const containers = {
nginx: await getContainerStatus("paddleocr-nginx"), nginx: await getContainerStatus("paddleocr-nginx"),
vllmServer: await getContainerStatus("paddleocr-vllm-server"), vllmServer: await getContainerStatus("paddleocr-vllm-server"),
pipelineApi: await getContainerStatus("paddleocr-pipeline-api"), pipelineApi: await getContainerStatus("paddleocr-pipeline-api"),
gradioUi: await getContainerStatus("paddleocr-gradio-ui"), gradioUi: await getContainerStatus("paddleocr-gradio-ui"),
pfmWebApp: await getContainerStatus("paddleocr-pfm-web-app"), pfmWebApp: await getContainerStatus("paddleocr-pfm-web-app"),
db: await getContainerStatus("paddleocr-db") db: await getContainerStatus("paddleocr-db")
}; };
return NextResponse.json({ return NextResponse.json({
success: true, success: true,
gpus, gpus,
settings, settings,
containers containers
}); });
} catch (error: any) { } catch (error: any) {
console.error("Failed to fetch GPU/container status:", error); console.error("Failed to fetch GPU/container status:", error);
return errorResponse(500, error.message); return errorResponse(500, error.message);
} }
} }
export async function POST(req: NextRequest) { export async function POST(req: NextRequest) {
try { try {
const body = await req.json().catch(() => ({})); const body = await req.json().catch(() => ({}));
const { action } = body; const { action } = body;
if (action === "kill") { if (action === "kill") {
const pid = parseInt(body.pid); const pid = parseInt(body.pid);
if (!pid || isNaN(pid)) { if (!pid || isNaN(pid)) {
return errorResponse(400, "Invalid PID"); return errorResponse(400, "Invalid PID");
} }
// Check if process is protected (same rules as admin_panel.py) // Check if process is protected (same rules as admin_panel.py)
const procName = getProcessName(pid); const procName = getProcessName(pid);
const procNameLower = procName.toLowerCase(); const procNameLower = procName.toLowerCase();
const protectedKeywords = ["rustdesk", "xorg", "nginx", "systemd", "dockerd", "python3", "node"]; const protectedKeywords = ["rustdesk", "xorg", "nginx", "systemd", "dockerd", "python3", "node"];
if (anyKeywordMatch(procNameLower, protectedKeywords)) { if (anyKeywordMatch(procNameLower, protectedKeywords)) {
return errorResponse(403, `Operation Denied: Process ${pid} (${procName || "system"}) is protected and cannot be killed.`); return errorResponse(403, `Operation Denied: Process ${pid} (${procName || "system"}) is protected and cannot be killed.`);
} }
try { try {
process.kill(pid, 9); process.kill(pid, 9);
return NextResponse.json({ success: true, message: `Successfully killed process ${pid}` }); return NextResponse.json({ success: true, message: `Successfully killed process ${pid}` });
} catch (err: any) { } catch (err: any) {
return errorResponse(500, `Failed to kill process: ${err.message}`); return errorResponse(500, `Failed to kill process: ${err.message}`);
} }
} }
if (action === "container") { if (action === "container") {
const { containerName, containerAction } = body; const { containerName, containerAction } = body;
const validActions = ["start", "stop", "restart"]; const validActions = ["start", "stop", "restart"];
const validContainers = [ const validContainers = [
"paddleocr-nginx", "paddleocr-nginx",
"paddleocr-vllm-server", "paddleocr-vllm-server",
"paddleocr-pipeline-api", "paddleocr-pipeline-api",
"paddleocr-gradio-ui", "paddleocr-gradio-ui",
"paddleocr-pfm-web-app", "paddleocr-pfm-web-app",
"paddleocr-db" "paddleocr-db"
]; ];
if (!validActions.includes(containerAction) || !validContainers.includes(containerName)) { if (!validActions.includes(containerAction) || !validContainers.includes(containerName)) {
return errorResponse(400, "Invalid container name or action"); return errorResponse(400, "Invalid container name or action");
} }
// Prevent self-stopping nextjs app accidentally through UI // Prevent self-stopping nextjs app accidentally through UI
if (containerName === "paddleocr-pfm-web-app" && containerAction === "stop") { if (containerName === "paddleocr-pfm-web-app" && containerAction === "stop") {
return errorResponse(400, "Cannot stop the active web application container itself."); return errorResponse(400, "Cannot stop the active web application container itself.");
} }
await manageContainer(containerName, containerAction); await manageContainer(containerName, containerAction);
return NextResponse.json({ return NextResponse.json({
success: true, success: true,
message: `Command 'docker-compose ${containerAction} ${containerName.replace("paddleocr-", "")}' executed successfully.` message: `Command 'docker-compose ${containerAction} ${containerName.replace("paddleocr-", "")}' executed successfully.`
}); });
} }
if (action === "saveSettings") { if (action === "saveSettings") {
const { cudaDevices } = body; const { cudaDevices } = body;
if (typeof cudaDevices !== "string" || cudaDevices.trim() === "") { if (typeof cudaDevices !== "string" || cudaDevices.trim() === "") {
return errorResponse(400, "Invalid GPU allocation settings"); return errorResponse(400, "Invalid GPU allocation settings");
} }
const cleanCuda = cudaDevices.trim(); const cleanCuda = cudaDevices.trim();
await saveEnvSettings(cleanCuda); await saveEnvSettings(cleanCuda);
// Recreate GPU containers to apply env settings // Recreate GPU containers to apply env settings
try { try {
await recreateContainer("paddleocr-vllm-server", cleanCuda); await recreateContainer("paddleocr-vllm-server", cleanCuda);
} catch (err: any) { } catch (err: any) {
console.error("Failed to recreate vllm-server container:", err); console.error("Failed to recreate vllm-server container:", err);
} }
try { try {
await recreateContainer("paddleocr-pipeline-api", cleanCuda); await recreateContainer("paddleocr-pipeline-api", cleanCuda);
} catch (err: any) { } catch (err: any) {
console.error("Failed to recreate pipeline-api container:", err); console.error("Failed to recreate pipeline-api container:", err);
} }
return NextResponse.json({ return NextResponse.json({
success: true, success: true,
message: `GPU settings updated to device index ${cleanCuda}. Core services recreated successfully.` message: `GPU settings updated to device index ${cleanCuda}. Core services recreated successfully.`
}); });
} }
if (action === "unload") { if (action === "unload") {
const { stopped, failed } = await unloadOtherEngines(); const { stopped, failed } = await unloadOtherEngines();
if (stopped.length === 0 && failed.length === 0) { if (stopped.length === 0 && failed.length === 0) {
return NextResponse.json({ return NextResponse.json({
success: true, success: true,
message: "All other OCR engines are already stopped/unloaded." message: "All other OCR engines are already stopped/unloaded."
}); });
} }
let msg = ""; let msg = "";
if (stopped.length > 0) { if (stopped.length > 0) {
msg += `Successfully stopped/unloaded: ${stopped.join(", ")}. `; msg += `Successfully stopped/unloaded: ${stopped.join(", ")}. `;
} }
if (failed.length > 0) { if (failed.length > 0) {
msg += `Failed to stop: ${failed.join(", ")}.`; msg += `Failed to stop: ${failed.join(", ")}.`;
} }
return NextResponse.json({ return NextResponse.json({
success: failed.length === 0, success: failed.length === 0,
message: msg.trim(), message: msg.trim(),
error: failed.length > 0 ? `Failed to stop some containers: ${failed.join(", ")}` : undefined error: failed.length > 0 ? `Failed to stop some containers: ${failed.join(", ")}` : undefined
}); });
} }
return errorResponse(400, "Invalid API action"); return errorResponse(400, "Invalid API action");
} catch (error: any) { } catch (error: any) {
console.error("GPU API POST error:", error); console.error("GPU API POST error:", error);
return errorResponse(500, error.message); return errorResponse(500, error.message);
} }
} }
function anyKeywordMatch(str: string, keywords: string[]): boolean { function anyKeywordMatch(str: string, keywords: string[]): boolean {
for (const kw of keywords) { for (const kw of keywords) {
if (str.includes(kw)) return true; if (str.includes(kw)) return true;
} }
return false; return false;
} }
+202 -202
View File
@@ -1,202 +1,202 @@
import { NextRequest, NextResponse } from "next/server"; import { NextRequest, NextResponse } from "next/server";
import { query, cleanupAndReindexItems } from "../../../db"; import { query, cleanupAndReindexItems } from "../../../db";
import fs from "fs"; import fs from "fs";
import path from "path"; import path from "path";
import { errorResponse } from "@/utils/api-error"; import { errorResponse } from "@/utils/api-error";
const UPLOADS_DIR = "/uploads"; const UPLOADS_DIR = "/uploads";
export async function GET(req: NextRequest) { export async function GET(req: NextRequest) {
try { try {
const fileParam = req.nextUrl.searchParams.get("file"); const fileParam = req.nextUrl.searchParams.get("file");
if (fileParam) { if (fileParam) {
const safeFile = path.basename(fileParam); const safeFile = path.basename(fileParam);
// 1. Try to load from database first // 1. Try to load from database first
const docRes = await query( const docRes = await query(
"SELECT id, layout_parsing_result, metadata FROM documents WHERE filename = $1", "SELECT id, layout_parsing_result, metadata FROM documents WHERE filename = $1",
[safeFile] [safeFile]
); );
if (docRes.rowCount && docRes.rowCount > 0) { if (docRes.rowCount && docRes.rowCount > 0) {
const doc = docRes.rows[0]; const doc = docRes.rows[0];
const docId = doc.id; const docId = doc.id;
const pipelineResult = doc.layout_parsing_result; const pipelineResult = doc.layout_parsing_result;
// Clean up and re-index invalid items first // Clean up and re-index invalid items first
await cleanupAndReindexItems(docId); await cleanupAndReindexItems(docId);
// Fetch items // Fetch items
const itemsRes = await query( const itemsRes = await query(
`SELECT row_index, `SELECT row_index,
kode_barang, nama_barang, banyak, jumlah, kode_barang, nama_barang, banyak, jumlah,
is_flagged, remark is_flagged, remark
FROM ocr_items FROM ocr_items
WHERE document_id = $1 WHERE document_id = $1
ORDER BY row_index`, ORDER BY row_index`,
[docId] [docId]
); );
const items = itemsRes.rows.map(row => ({ const items = itemsRes.rows.map(row => ({
kodeBarang: row.kode_barang, kodeBarang: row.kode_barang,
namaBarang: row.nama_barang, namaBarang: row.nama_barang,
banyak: row.banyak, banyak: row.banyak,
jumlah: row.jumlah jumlah: row.jumlah
})); }));
const flagged: Record<number, boolean> = {}; const flagged: Record<number, boolean> = {};
const remarks: Record<number, string> = {}; const remarks: Record<number, string> = {};
itemsRes.rows.forEach(row => { itemsRes.rows.forEach(row => {
if (row.is_flagged) { if (row.is_flagged) {
flagged[row.row_index] = true; flagged[row.row_index] = true;
} }
if (row.remark && row.remark.trim()) { if (row.remark && row.remark.trim()) {
remarks[row.row_index] = row.remark; remarks[row.row_index] = row.remark;
} }
}); });
return NextResponse.json({ return NextResponse.json({
errorCode: 0, errorCode: 0,
errorMsg: "Success", errorMsg: "Success",
result: pipelineResult, result: pipelineResult,
items, items,
flagged, flagged,
remarks, remarks,
headerRemark: (doc.metadata as any)?.headerRemark || "" headerRemark: (doc.metadata as any)?.headerRemark || ""
}); });
} }
// 2. Fallback to filesystem // 2. Fallback to filesystem
const jsonPath = path.join(UPLOADS_DIR, `${safeFile}.json`); const jsonPath = path.join(UPLOADS_DIR, `${safeFile}.json`);
if (fs.existsSync(jsonPath)) { if (fs.existsSync(jsonPath)) {
const jsonData = fs.readFileSync(jsonPath, "utf8"); const jsonData = fs.readFileSync(jsonPath, "utf8");
const data = JSON.parse(jsonData); const data = JSON.parse(jsonData);
return NextResponse.json({ return NextResponse.json({
errorCode: 0, errorCode: 0,
errorMsg: "Success", errorMsg: "Success",
result: data.result || data result: data.result || data
}); });
} }
return errorResponse(404, "Document not found"); return errorResponse(404, "Document not found");
} }
// List view: return history list from DB // List view: return history list from DB
const showAll = req.nextUrl.searchParams.get("all") === "true"; const showAll = req.nextUrl.searchParams.get("all") === "true";
let listRes; let listRes;
if (showAll) { if (showAll) {
listRes = await query( listRes = await query(
`SELECT id, filename, upload_time, size, parsed, is_sample, metadata, `SELECT id, filename, upload_time, size, parsed, is_sample, metadata,
(SELECT COUNT(*) FROM ocr_items WHERE document_id = documents.id) as total_items, (SELECT COUNT(*) FROM ocr_items WHERE document_id = documents.id) as total_items,
(SELECT COUNT(*) FROM ocr_items WHERE document_id = documents.id AND is_flagged = TRUE) as flagged_items (SELECT COUNT(*) FROM ocr_items WHERE document_id = documents.id AND is_flagged = TRUE) as flagged_items
FROM documents FROM documents
ORDER BY upload_time DESC` ORDER BY upload_time DESC`
); );
} else { } else {
listRes = await query( listRes = await query(
`SELECT id, filename, upload_time, size, parsed, is_sample, metadata, `SELECT id, filename, upload_time, size, parsed, is_sample, metadata,
(SELECT COUNT(*) FROM ocr_items WHERE document_id = documents.id) as total_items, (SELECT COUNT(*) FROM ocr_items WHERE document_id = documents.id) as total_items,
(SELECT COUNT(*) FROM ocr_items WHERE document_id = documents.id AND is_flagged = TRUE) as flagged_items (SELECT COUNT(*) FROM ocr_items WHERE document_id = documents.id AND is_flagged = TRUE) as flagged_items
FROM documents FROM documents
WHERE is_sample = FALSE WHERE is_sample = FALSE
ORDER BY upload_time DESC` ORDER BY upload_time DESC`
); );
} }
const history = listRes.rows.map(row => ({ const history = listRes.rows.map(row => ({
id: row.id, id: row.id,
filename: row.filename, filename: row.filename,
uploadTime: row.upload_time.toISOString(), uploadTime: row.upload_time.toISOString(),
size: row.size, size: row.size,
parsed: row.parsed, parsed: row.parsed,
isSample: row.is_sample, isSample: row.is_sample,
metadata: row.metadata, metadata: row.metadata,
totalItems: parseInt(row.total_items || "0"), totalItems: parseInt(row.total_items || "0"),
flaggedItems: parseInt(row.flagged_items || "0") flaggedItems: parseInt(row.flagged_items || "0")
})); }));
return NextResponse.json({ history }); return NextResponse.json({ history });
} catch (error: unknown) { } catch (error: unknown) {
console.error("Error in history API route:", error); console.error("Error in history API route:", error);
const message = error instanceof Error ? error.message : "Internal server error"; const message = error instanceof Error ? error.message : "Internal server error";
return errorResponse(500, message); return errorResponse(500, message);
} }
} }
export async function DELETE(req: NextRequest) { export async function DELETE(req: NextRequest) {
try { try {
const { filename } = await req.json(); const { filename } = await req.json();
if (!filename) { if (!filename) {
return errorResponse(400, "Filename is required"); return errorResponse(400, "Filename is required");
} }
const safeFile = path.basename(filename); const safeFile = path.basename(filename);
// Check if it exists and get its status // Check if it exists and get its status
const checkRes = await query( const checkRes = await query(
"SELECT id, is_sample FROM documents WHERE filename = $1", "SELECT id, is_sample FROM documents WHERE filename = $1",
[safeFile] [safeFile]
); );
if (checkRes.rowCount && checkRes.rowCount > 0) { if (checkRes.rowCount && checkRes.rowCount > 0) {
const doc = checkRes.rows[0]; const doc = checkRes.rows[0];
const isSample = doc.is_sample; const isSample = doc.is_sample;
// Delete from DB (cascading delete will remove ocr_items) // Delete from DB (cascading delete will remove ocr_items)
await query("DELETE FROM documents WHERE filename = $1", [safeFile]); await query("DELETE FROM documents WHERE filename = $1", [safeFile]);
// If it is a custom upload, clean up files from /uploads directory // If it is a custom upload, clean up files from /uploads directory
if (!isSample) { if (!isSample) {
const imagePath = path.join(UPLOADS_DIR, safeFile); const imagePath = path.join(UPLOADS_DIR, safeFile);
const jsonPath = path.join(UPLOADS_DIR, `${safeFile}.json`); const jsonPath = path.join(UPLOADS_DIR, `${safeFile}.json`);
if (fs.existsSync(imagePath)) { if (fs.existsSync(imagePath)) {
fs.unlinkSync(imagePath); fs.unlinkSync(imagePath);
} }
if (fs.existsSync(jsonPath)) { if (fs.existsSync(jsonPath)) {
fs.unlinkSync(jsonPath); fs.unlinkSync(jsonPath);
} }
} }
return NextResponse.json({ success: true }); return NextResponse.json({ success: true });
} }
return errorResponse(404, "Document not found"); return errorResponse(404, "Document not found");
} catch (error: unknown) { } catch (error: unknown) {
console.error("Error in DELETE history API route:", error); console.error("Error in DELETE history API route:", error);
const message = error instanceof Error ? error.message : "Internal server error"; const message = error instanceof Error ? error.message : "Internal server error";
return errorResponse(500, message); return errorResponse(500, message);
} }
} }
export async function POST(req: NextRequest) { export async function POST(req: NextRequest) {
try { try {
const { filename, remark } = await req.json(); const { filename, remark } = await req.json();
if (!filename) { if (!filename) {
return errorResponse(400, "Filename is required"); return errorResponse(400, "Filename is required");
} }
const safeFile = path.basename(filename); const safeFile = path.basename(filename);
const valueJson = JSON.stringify(remark || ""); const valueJson = JSON.stringify(remark || "");
const updateRes = await query( const updateRes = await query(
`UPDATE documents `UPDATE documents
SET metadata = jsonb_set(coalesce(metadata, '{}'::jsonb), '{headerRemark}', $1::jsonb) SET metadata = jsonb_set(coalesce(metadata, '{}'::jsonb), '{headerRemark}', $1::jsonb)
WHERE filename = $2`, WHERE filename = $2`,
[valueJson, safeFile] [valueJson, safeFile]
); );
if (updateRes.rowCount && updateRes.rowCount > 0) { if (updateRes.rowCount && updateRes.rowCount > 0) {
return NextResponse.json({ success: true }); return NextResponse.json({ success: true });
} }
return errorResponse(404, "Document not found"); return errorResponse(404, "Document not found");
} catch (error: unknown) { } catch (error: unknown) {
console.error("Error in POST history API route:", error); console.error("Error in POST history API route:", error);
const message = error instanceof Error ? error.message : "Internal server error"; const message = error instanceof Error ? error.message : "Internal server error";
return errorResponse(500, message); return errorResponse(500, message);
} }
} }
@@ -1,23 +1,23 @@
import { NextResponse } from "next/server"; import { NextResponse } from "next/server";
import fs from "fs"; import fs from "fs";
import path from "path"; import path from "path";
import { errorResponse } from "@/utils/api-error"; import { errorResponse } from "@/utils/api-error";
export async function GET() { export async function GET() {
try { try {
const dirPath = path.join(process.cwd(), "..", "sources", "test-images"); const dirPath = path.join(process.cwd(), "..", "sources", "test-images");
if (!fs.existsSync(dirPath)) { if (!fs.existsSync(dirPath)) {
return NextResponse.json({ files: [] }); return NextResponse.json({ files: [] });
} }
const files = fs.readdirSync(dirPath).filter(file => { const files = fs.readdirSync(dirPath).filter(file => {
const ext = path.extname(file).toLowerCase(); const ext = path.extname(file).toLowerCase();
return ext === ".jpg" || ext === ".jpeg" || ext === ".png"; return ext === ".jpg" || ext === ".jpeg" || ext === ".png";
}); });
// Sort files to keep consistent ordering in UI // Sort files to keep consistent ordering in UI
files.sort(); files.sort();
return NextResponse.json({ files }); return NextResponse.json({ files });
} catch (error: any) { } catch (error: any) {
console.error("Error reading test-images directory:", error); console.error("Error reading test-images directory:", error);
return errorResponse(500, error.message); return errorResponse(500, error.message);
} }
} }
@@ -1,238 +1,238 @@
import { NextRequest, NextResponse } from "next/server"; import { NextRequest, NextResponse } from "next/server";
import fs from "fs"; import fs from "fs";
import path from "path"; import path from "path";
import crypto from "crypto"; import crypto from "crypto";
import { query } from "@/db"; import { query } from "@/db";
import { errorResponse } from "@/utils/api-error"; import { errorResponse } from "@/utils/api-error";
// Separate from DO manual_labels.json - product scan ground truth only // Separate from DO manual_labels.json - product scan ground truth only
const LABELS_PATH = path.join(process.cwd(), "..", "sources", "product_manual_labels.json"); const LABELS_PATH = path.join(process.cwd(), "..", "sources", "product_manual_labels.json");
interface ProductScanLabel { interface ProductScanLabel {
filename: string; filename: string;
no_sku: string; no_sku: string;
nama_item: string; nama_item: string;
expiry_date: string; expiry_date: string;
top1_confidence: number | null; top1_confidence: number | null;
notes: string; notes: string;
saved_at: string; saved_at: string;
} }
function sanitizeFilename(filename: string): string { function sanitizeFilename(filename: string): string {
let cleaned = filename.replace(/\\/g, "/"); let cleaned = filename.replace(/\\/g, "/");
while (cleaned.startsWith("/")) { while (cleaned.startsWith("/")) {
cleaned = cleaned.substring(1); cleaned = cleaned.substring(1);
} }
return cleaned.replace(/\.\.\//g, ""); return cleaned.replace(/\.\.\//g, "");
} }
function normalizeDateString(dateStr: string): string { function normalizeDateString(dateStr: string): string {
if (!dateStr) return ""; if (!dateStr) return "";
const trimmed = dateStr.trim(); const trimmed = dateStr.trim();
// Pattern 1: d Month YYYY (e.g. 7 June 2026) // Pattern 1: d Month YYYY (e.g. 7 June 2026)
const textPattern = /^(\d{1,2})\s+([a-zA-Z]+)\s+(\d{4})$/; const textPattern = /^(\d{1,2})\s+([a-zA-Z]+)\s+(\d{4})$/;
const tm = trimmed.match(textPattern); const tm = trimmed.match(textPattern);
if (tm) { if (tm) {
const day = tm[1].padStart(2, "0"); const day = tm[1].padStart(2, "0");
const month = tm[2].charAt(0).toUpperCase() + tm[2].slice(1).toLowerCase(); const month = tm[2].charAt(0).toUpperCase() + tm[2].slice(1).toLowerCase();
const year = tm[3]; const year = tm[3];
return `${day} ${month} ${year}`; return `${day} ${month} ${year}`;
} }
// Pattern 2: d/m/YYYY or d-m-YYYY or d.m.YYYY (e.g. 7/6/2026) // Pattern 2: d/m/YYYY or d-m-YYYY or d.m.YYYY (e.g. 7/6/2026)
const digitPattern = /^(\d{1,2})([-./])(\d{1,2})\2(\d{2,4})$/; const digitPattern = /^(\d{1,2})([-./])(\d{1,2})\2(\d{2,4})$/;
const dm = trimmed.match(digitPattern); const dm = trimmed.match(digitPattern);
if (dm) { if (dm) {
const day = dm[1].padStart(2, "0"); const day = dm[1].padStart(2, "0");
const month = dm[3].padStart(2, "0"); const month = dm[3].padStart(2, "0");
let year = dm[4]; let year = dm[4];
if (year.length === 2) { if (year.length === 2) {
year = "20" + year; year = "20" + year;
} }
return `${day}/${month}/${year}`; return `${day}/${month}/${year}`;
} }
return trimmed; return trimmed;
} }
let purged = false; let purged = false;
function readLabels(): ProductScanLabel[] { function readLabels(): ProductScanLabel[] {
if (!fs.existsSync(LABELS_PATH)) { if (!fs.existsSync(LABELS_PATH)) {
return []; return [];
} }
const raw = fs.readFileSync(LABELS_PATH, "utf8"); const raw = fs.readFileSync(LABELS_PATH, "utf8");
if (!raw.trim()) return []; if (!raw.trim()) return [];
let labels: ProductScanLabel[] = JSON.parse(raw); let labels: ProductScanLabel[] = JSON.parse(raw);
// Cleanup phantom uploaded-* entries once // Cleanup phantom uploaded-* entries once
if (!purged) { if (!purged) {
const valid = labels.filter((l) => !l.filename.startsWith("uploaded-")); const valid = labels.filter((l) => !l.filename.startsWith("uploaded-"));
if (valid.length !== labels.length) { if (valid.length !== labels.length) {
writeLabels(valid); writeLabels(valid);
labels = valid; labels = valid;
} }
purged = true; purged = true;
} }
return labels; return labels;
} }
function writeLabels(labels: ProductScanLabel[]) { function writeLabels(labels: ProductScanLabel[]) {
const dir = path.dirname(LABELS_PATH); const dir = path.dirname(LABELS_PATH);
if (!fs.existsSync(dir)) { if (!fs.existsSync(dir)) {
fs.mkdirSync(dir, { recursive: true }); fs.mkdirSync(dir, { recursive: true });
} }
fs.writeFileSync(LABELS_PATH, JSON.stringify(labels, null, 2), "utf8"); fs.writeFileSync(LABELS_PATH, JSON.stringify(labels, null, 2), "utf8");
} }
export async function GET(req: NextRequest) { export async function GET(req: NextRequest) {
try { try {
const { searchParams } = new URL(req.url); const { searchParams } = new URL(req.url);
const filename = searchParams.get("filename"); const filename = searchParams.get("filename");
if (!filename) { if (!filename) {
const labels = readLabels(); const labels = readLabels();
return NextResponse.json(labels); return NextResponse.json(labels);
} }
const safeFilename = sanitizeFilename(filename); const safeFilename = sanitizeFilename(filename);
const labels = readLabels(); const labels = readLabels();
const existing = labels.find((l) => l.filename === safeFilename); const existing = labels.find((l) => l.filename === safeFilename);
// Inferred values from filename/directory structure // Inferred values from filename/directory structure
let inferredSku = ""; let inferredSku = "";
let inferredNamaItem = ""; let inferredNamaItem = "";
const parts = safeFilename.split("/"); const parts = safeFilename.split("/");
if (parts.length > 1) { if (parts.length > 1) {
const folderName = parts[0]; const folderName = parts[0];
const match = folderName.match(/^(\d{8})/); const match = folderName.match(/^(\d{8})/);
if (match) { if (match) {
inferredSku = match[1]; inferredSku = match[1];
} else if (/^\d{8}$/.test(folderName)) { } else if (/^\d{8}$/.test(folderName)) {
inferredSku = folderName; inferredSku = folderName;
} }
} }
if (inferredSku) { if (inferredSku) {
try { try {
const dbRes = await query("SELECT nama_item FROM sku_master WHERE no_sku = $1", [inferredSku]); const dbRes = await query("SELECT nama_item FROM sku_master WHERE no_sku = $1", [inferredSku]);
if (dbRes.rowCount && dbRes.rowCount > 0) { if (dbRes.rowCount && dbRes.rowCount > 0) {
inferredNamaItem = dbRes.rows[0].nama_item; inferredNamaItem = dbRes.rows[0].nama_item;
} }
} catch (dbErr) { } catch (dbErr) {
console.error("Failed to query sku_master for manual label:", dbErr); console.error("Failed to query sku_master for manual label:", dbErr);
} }
} }
// Inferred expiry date from sibling files in the same parent directory // Inferred expiry date from sibling files in the same parent directory
let siblingExpiry = ""; let siblingExpiry = "";
let parentFolder = ""; let parentFolder = "";
if (parts.length > 1) { if (parts.length > 1) {
parentFolder = parts.slice(0, -1).join("/"); parentFolder = parts.slice(0, -1).join("/");
} }
if (parentFolder) { if (parentFolder) {
const sibling = labels.find( const sibling = labels.find(
(l) => l.filename.startsWith(parentFolder + "/") && l.expiry_date (l) => l.filename.startsWith(parentFolder + "/") && l.expiry_date
); );
if (sibling) { if (sibling) {
siblingExpiry = sibling.expiry_date; siblingExpiry = sibling.expiry_date;
} }
} }
if (existing) { if (existing) {
return NextResponse.json({ return NextResponse.json({
...existing, ...existing,
no_sku: existing.no_sku || inferredSku, no_sku: existing.no_sku || inferredSku,
nama_item: existing.nama_item || inferredNamaItem, nama_item: existing.nama_item || inferredNamaItem,
expiry_date: existing.expiry_date || siblingExpiry expiry_date: existing.expiry_date || siblingExpiry
}); });
} }
// Return empty default state if not found, with inferred metadata // Return empty default state if not found, with inferred metadata
return NextResponse.json({ return NextResponse.json({
filename: safeFilename, filename: safeFilename,
no_sku: inferredSku, no_sku: inferredSku,
nama_item: inferredNamaItem, nama_item: inferredNamaItem,
expiry_date: siblingExpiry, expiry_date: siblingExpiry,
top1_confidence: null, top1_confidence: null,
notes: "", notes: "",
saved_at: "" saved_at: ""
}); });
} catch (err: unknown) { } catch (err: unknown) {
console.error("Error in GET manual-label-scan:", err); console.error("Error in GET manual-label-scan:", err);
const message = err instanceof Error ? err.message : "Failed to load product label"; const message = err instanceof Error ? err.message : "Failed to load product label";
return errorResponse(500, message); return errorResponse(500, message);
} }
} }
export async function POST(req: NextRequest) { export async function POST(req: NextRequest) {
try { try {
const body = await req.json(); const body = await req.json();
const { filename, image } = body; const { filename, image } = body;
let safeFilename = sanitizeFilename(filename || "unknown.jpg"); let safeFilename = sanitizeFilename(filename || "unknown.jpg");
if (image && image.startsWith("data:image/")) { if (image && image.startsWith("data:image/")) {
const base64Data = image.replace(/^data:image\/\w+;base64,/, ""); const base64Data = image.replace(/^data:image\/\w+;base64,/, "");
const buffer = Buffer.from(base64Data, "base64"); const buffer = Buffer.from(base64Data, "base64");
const hash = crypto.createHash("md5").update(buffer).digest("hex"); const hash = crypto.createHash("md5").update(buffer).digest("hex");
const ext = image.match(/^data:image\/(\w+);base64,/)?.[1] || "jpg"; const ext = image.match(/^data:image\/(\w+);base64,/)?.[1] || "jpg";
safeFilename = `${hash}.${ext}`; safeFilename = `${hash}.${ext}`;
const saveDir = path.join(process.cwd(), "..", "sources", "product-test-images"); const saveDir = path.join(process.cwd(), "..", "sources", "product-test-images");
if (!fs.existsSync(saveDir)) { if (!fs.existsSync(saveDir)) {
fs.mkdirSync(saveDir, { recursive: true }); fs.mkdirSync(saveDir, { recursive: true });
} }
fs.writeFileSync(path.join(saveDir, safeFilename), buffer); fs.writeFileSync(path.join(saveDir, safeFilename), buffer);
} }
if (!safeFilename || safeFilename === "unknown.jpg") { if (!safeFilename || safeFilename === "unknown.jpg") {
return errorResponse(400, "Filename or valid image is required in request body"); return errorResponse(400, "Filename or valid image is required in request body");
} }
const labels = readLabels(); const labels = readLabels();
const index = labels.findIndex((l) => l.filename === safeFilename); const index = labels.findIndex((l) => l.filename === safeFilename);
const entry: ProductScanLabel = { const entry: ProductScanLabel = {
filename: safeFilename, filename: safeFilename,
no_sku: body.no_sku || "", no_sku: body.no_sku || "",
nama_item: body.nama_item || "", nama_item: body.nama_item || "",
expiry_date: normalizeDateString(body.expiry_date || ""), expiry_date: normalizeDateString(body.expiry_date || ""),
top1_confidence: typeof body.top1_confidence === "number" ? body.top1_confidence : null, top1_confidence: typeof body.top1_confidence === "number" ? body.top1_confidence : null,
notes: body.notes || "", notes: body.notes || "",
saved_at: new Date().toISOString() saved_at: new Date().toISOString()
}; };
if (index >= 0) { if (index >= 0) {
labels[index] = entry; labels[index] = entry;
} else { } else {
labels.push(entry); labels.push(entry);
} }
writeLabels(labels); writeLabels(labels);
return NextResponse.json({ success: true, filePath: LABELS_PATH, entry }); return NextResponse.json({ success: true, filePath: LABELS_PATH, entry });
} catch (err: unknown) { } catch (err: unknown) {
console.error("Error in POST manual-label-scan:", err); console.error("Error in POST manual-label-scan:", err);
const message = err instanceof Error ? err.message : "Failed to save product label"; const message = err instanceof Error ? err.message : "Failed to save product label";
return errorResponse(500, message); return errorResponse(500, message);
} }
} }
export async function DELETE(req: NextRequest) { export async function DELETE(req: NextRequest) {
try { try {
const { searchParams } = new URL(req.url); const { searchParams } = new URL(req.url);
const filename = searchParams.get("filename"); const filename = searchParams.get("filename");
if (!filename) return errorResponse(400, "Filename parameter is required"); if (!filename) return errorResponse(400, "Filename parameter is required");
const safeFilename = sanitizeFilename(filename); const safeFilename = sanitizeFilename(filename);
const labels = readLabels(); const labels = readLabels();
const filtered = labels.filter((l) => l.filename !== safeFilename); const filtered = labels.filter((l) => l.filename !== safeFilename);
writeLabels(filtered); writeLabels(filtered);
return NextResponse.json({ success: true }); return NextResponse.json({ success: true });
} catch (err: unknown) { } catch (err: unknown) {
const message = err instanceof Error ? err.message : "Failed to delete label"; const message = err instanceof Error ? err.message : "Failed to delete label";
return errorResponse(500, message); return errorResponse(500, message);
} }
} }
@@ -1,147 +1,147 @@
import { NextRequest, NextResponse } from "next/server"; import { NextRequest, NextResponse } from "next/server";
import { query } from "../../../db"; import { query } from "../../../db";
import fs from "fs"; import fs from "fs";
import path from "path"; import path from "path";
import { errorResponse } from "@/utils/api-error"; import { errorResponse } from "@/utils/api-error";
const LABELS_PATH = path.join(process.cwd(), "..", "sources", "manual_labels.json"); const LABELS_PATH = path.join(process.cwd(), "..", "sources", "manual_labels.json");
function readLabels(): any[] { function readLabels(): any[] {
if (!fs.existsSync(LABELS_PATH)) { if (!fs.existsSync(LABELS_PATH)) {
return []; return [];
} }
const raw = fs.readFileSync(LABELS_PATH, "utf8"); const raw = fs.readFileSync(LABELS_PATH, "utf8");
return raw.trim() ? JSON.parse(raw) : []; return raw.trim() ? JSON.parse(raw) : [];
} }
function writeLabels(labels: any[]) { function writeLabels(labels: any[]) {
fs.writeFileSync(LABELS_PATH, JSON.stringify(labels, null, 2), "utf8"); fs.writeFileSync(LABELS_PATH, JSON.stringify(labels, null, 2), "utf8");
} }
const SCALAR_FIELDS = ["noPO", "noSO", "noDO", "tanggal", "customer", "store", "alamat", "plat"] as const; const SCALAR_FIELDS = ["noPO", "noSO", "noDO", "tanggal", "customer", "store", "alamat", "plat"] as const;
// Fetches the latest automated parser result for a filename, in the same // Fetches the latest automated parser result for a filename, in the same
// shape as a manual_labels.json entry, so it can be used as fill-in data. // shape as a manual_labels.json entry, so it can be used as fill-in data.
async function fetchLatestParsed(safeFilename: string): Promise<Record<string, any> | null> { async function fetchLatestParsed(safeFilename: string): Promise<Record<string, any> | null> {
try { try {
const docRes = await query( const docRes = await query(
"SELECT id, metadata FROM documents WHERE filename = $1", "SELECT id, metadata FROM documents WHERE filename = $1",
[safeFilename] [safeFilename]
); );
if (!docRes.rowCount || docRes.rowCount === 0) return null; if (!docRes.rowCount || docRes.rowCount === 0) return null;
const doc = docRes.rows[0]; const doc = docRes.rows[0];
const meta = doc.metadata || {}; const meta = doc.metadata || {};
const itemsRes = await query( const itemsRes = await query(
"SELECT kode_barang, nama_barang, banyak, jumlah FROM ocr_items WHERE document_id = $1 ORDER BY row_index", "SELECT kode_barang, nama_barang, banyak, jumlah FROM ocr_items WHERE document_id = $1 ORDER BY row_index",
[doc.id] [doc.id]
); );
return { return {
noPO: meta.noPO || "", noPO: meta.noPO || "",
noSO: meta.noSO || "", noSO: meta.noSO || "",
noDO: meta.noDO || "", noDO: meta.noDO || "",
tanggal: meta.tanggal || "", tanggal: meta.tanggal || "",
customer: meta.customerInfo || "", customer: meta.customerInfo || "",
store: meta.orderUntuk || "", store: meta.orderUntuk || "",
alamat: meta.alamat || "", alamat: meta.alamat || "",
plat: meta.platTruk || "", plat: meta.platTruk || "",
items: itemsRes.rows.map(row => ({ items: itemsRes.rows.map(row => ({
kodeBarang: row.kode_barang || "", kodeBarang: row.kode_barang || "",
namaBarang: row.nama_barang || "", namaBarang: row.nama_barang || "",
banyak: row.banyak || "", banyak: row.banyak || "",
jumlah: row.jumlah || "" jumlah: row.jumlah || ""
})) }))
}; };
} catch (dbErr) { } catch (dbErr) {
console.error("DB fallback failed inside manual-label GET:", dbErr); console.error("DB fallback failed inside manual-label GET:", dbErr);
return null; return null;
} }
} }
export async function GET(req: NextRequest) { export async function GET(req: NextRequest) {
try { try {
const { searchParams } = new URL(req.url); const { searchParams } = new URL(req.url);
const filename = searchParams.get("filename"); const filename = searchParams.get("filename");
if (!filename) { if (!filename) {
return errorResponse(400, "Filename parameter is required"); return errorResponse(400, "Filename parameter is required");
} }
const safeFilename = path.basename(filename); const safeFilename = path.basename(filename);
const labels = readLabels(); const labels = readLabels();
const existing = labels.find(l => l.filename === safeFilename); const existing = labels.find(l => l.filename === safeFilename);
const latest = await fetchLatestParsed(safeFilename); const latest = await fetchLatestParsed(safeFilename);
if (existing) { if (existing) {
// Never overwrite a field the user already corrected manually - only // Never overwrite a field the user already corrected manually - only
// fill in whatever is still blank, using the latest AI/DB parse. // fill in whatever is still blank, using the latest AI/DB parse.
const merged = { ...existing, filename: safeFilename }; const merged = { ...existing, filename: safeFilename };
if (latest) { if (latest) {
for (const field of SCALAR_FIELDS) { for (const field of SCALAR_FIELDS) {
if (!merged[field]) merged[field] = latest[field]; if (!merged[field]) merged[field] = latest[field];
} }
if (!merged.items || merged.items.length === 0) { if (!merged.items || merged.items.length === 0) {
merged.items = latest.items; merged.items = latest.items;
} }
} }
// aiPredicted is the raw AI value for every field, always included // aiPredicted is the raw AI value for every field, always included
// (even when a manual value already exists) so the UI can show what // (even when a manual value already exists) so the UI can show what
// the AI actually predicted next to the current/manual value. // the AI actually predicted next to the current/manual value.
return NextResponse.json({ ...merged, aiPredicted: latest }); return NextResponse.json({ ...merged, aiPredicted: latest });
} }
if (latest) { if (latest) {
return NextResponse.json({ filename, ...latest, aiPredicted: latest }); return NextResponse.json({ filename, ...latest, aiPredicted: latest });
} }
// Return empty default state if not found anywhere // Return empty default state if not found anywhere
return NextResponse.json({ return NextResponse.json({
filename, filename,
noPO: "", noPO: "",
noSO: "", noSO: "",
noDO: "", noDO: "",
tanggal: "", tanggal: "",
customer: "", customer: "",
store: "", store: "",
alamat: "", alamat: "",
plat: "", plat: "",
items: [], items: [],
aiPredicted: null aiPredicted: null
}); });
} catch (err: any) { } catch (err: any) {
console.error("Error in GET manual-label:", err); console.error("Error in GET manual-label:", err);
return errorResponse(500, err.message || "Failed to load manual label"); return errorResponse(500, err.message || "Failed to load manual label");
} }
} }
export async function POST(req: NextRequest) { export async function POST(req: NextRequest) {
try { try {
const body = await req.json(); const body = await req.json();
const { filename } = body; const { filename } = body;
if (!filename) { if (!filename) {
return errorResponse(400, "Filename is required in request body"); return errorResponse(400, "Filename is required in request body");
} }
const safeFilename = path.basename(filename); const safeFilename = path.basename(filename);
const labels = readLabels(); const labels = readLabels();
const index = labels.findIndex(l => l.filename === safeFilename); const index = labels.findIndex(l => l.filename === safeFilename);
const entry = { ...body, filename: safeFilename }; const entry = { ...body, filename: safeFilename };
if (index >= 0) { if (index >= 0) {
labels[index] = entry; labels[index] = entry;
} else { } else {
labels.push(entry); labels.push(entry);
} }
writeLabels(labels); writeLabels(labels);
return NextResponse.json({ success: true, filePath: LABELS_PATH }); return NextResponse.json({ success: true, filePath: LABELS_PATH });
} catch (err: any) { } catch (err: any) {
console.error("Error in POST manual-label:", err); console.error("Error in POST manual-label:", err);
return errorResponse(500, err.message || "Failed to save manual label"); return errorResponse(500, err.message || "Failed to save manual label");
} }
} }
File diff suppressed because it is too large. Load diff
@@ -1,54 +1,54 @@
import { NextRequest, NextResponse } from "next/server"; import { NextRequest, NextResponse } from "next/server";
import fs from "fs"; import fs from "fs";
import path from "path"; import path from "path";
import { errorResponse } from "@/utils/api-error"; import { errorResponse } from "@/utils/api-error";
export const dynamic = "force-dynamic"; export const dynamic = "force-dynamic";
export async function GET(req: NextRequest) { export async function GET(req: NextRequest) {
try { try {
const filename = req.nextUrl.searchParams.get("filename"); const filename = req.nextUrl.searchParams.get("filename");
const dirPath = path.join(process.cwd(), "..", "sources", "product-test-images-fixed"); const dirPath = path.join(process.cwd(), "..", "sources", "product-test-images-fixed");
// File serving mode // File serving mode
if (filename) { if (filename) {
const safeFile = path.basename(filename); const safeFile = path.basename(filename);
const filePath = path.join(dirPath, safeFile); const filePath = path.join(dirPath, safeFile);
if (!fs.existsSync(filePath)) { if (!fs.existsSync(filePath)) {
return errorResponse(404, "File not found"); return errorResponse(404, "File not found");
} }
const ext = path.extname(safeFile).toLowerCase(); const ext = path.extname(safeFile).toLowerCase();
let contentType = "application/octet-stream"; let contentType = "application/octet-stream";
if (ext === ".jpg" || ext === ".jpeg") contentType = "image/jpeg"; if (ext === ".jpg" || ext === ".jpeg") contentType = "image/jpeg";
else if (ext === ".png") contentType = "image/png"; else if (ext === ".png") contentType = "image/png";
else if (ext === ".webp") contentType = "image/webp"; else if (ext === ".webp") contentType = "image/webp";
const fileBuffer = fs.readFileSync(filePath); const fileBuffer = fs.readFileSync(filePath);
return new Response(fileBuffer, { return new Response(fileBuffer, {
headers: { headers: {
"Content-Type": contentType, "Content-Type": contentType,
"Cache-Control": "public, max-age=31536000, immutable" "Cache-Control": "public, max-age=31536000, immutable"
} }
}); });
} }
// List mode // List mode
if (!fs.existsSync(dirPath)) { if (!fs.existsSync(dirPath)) {
return NextResponse.json({ files: [] }); return NextResponse.json({ files: [] });
} }
const files = fs.readdirSync(dirPath).filter((file) => { const files = fs.readdirSync(dirPath).filter((file) => {
const ext = path.extname(file).toLowerCase(); const ext = path.extname(file).toLowerCase();
return [".jpg", ".jpeg", ".png", ".webp"].includes(ext); return [".jpg", ".jpeg", ".png", ".webp"].includes(ext);
}); });
files.sort(); files.sort();
return NextResponse.json({ files }); return NextResponse.json({ files });
} catch (error: unknown) { } catch (error: unknown) {
console.error("Error in product-images API:", error); console.error("Error in product-images API:", error);
const message = error instanceof Error ? error.message : "Internal server error"; const message = error instanceof Error ? error.message : "Internal server error";
return errorResponse(500, message); return errorResponse(500, message);
} }
} }
@@ -1,86 +1,86 @@
import { NextRequest, NextResponse } from "next/server"; import { NextRequest, NextResponse } from "next/server";
import fs from "fs"; import fs from "fs";
import path from "path"; import path from "path";
import { errorResponse } from "@/utils/api-error"; import { errorResponse } from "@/utils/api-error";
// Serves the most recent accuracy-check-scan.mts detail dump // Serves the most recent accuracy-check-scan.mts detail dump
// (sources/product_scan_detail_*.json) so the manual-label-scan page can show // (sources/product_scan_detail_*.json) so the manual-label-scan page can show
// what the AI actually predicted for a given Validation Set image by default, // what the AI actually predicted for a given Validation Set image by default,
// without re-running the pipeline live for every image browsed. This is the // without re-running the pipeline live for every image browsed. This is the
// same predicted value the accuracy harness scores against ground truth - // same predicted value the accuracy harness scores against ground truth -
// not a fresh scan, so it reflects the last batch test run. // not a fresh scan, so it reflects the last batch test run.
const SOURCES_DIR = path.join(process.cwd(), "..", "sources"); const SOURCES_DIR = path.join(process.cwd(), "..", "sources");
interface DetailCheck { interface DetailCheck {
field: string; field: string;
match: boolean; match: boolean;
expected: string; expected: string;
predicted: string; predicted: string;
} }
interface DetailValidationItem { interface DetailValidationItem {
filename: string; filename: string;
method?: string; method?: string;
confidence?: number; confidence?: number;
checks: DetailCheck[]; checks: DetailCheck[];
} }
interface DetailDump { interface DetailDump {
timestamp: string; timestamp: string;
validation: DetailValidationItem[]; validation: DetailValidationItem[];
} }
function findLatestDump(): { path: string; data: DetailDump } | null { function findLatestDump(): { path: string; data: DetailDump } | null {
if (!fs.existsSync(SOURCES_DIR)) return null; if (!fs.existsSync(SOURCES_DIR)) return null;
const candidates = fs const candidates = fs
.readdirSync(SOURCES_DIR) .readdirSync(SOURCES_DIR)
.filter((f) => /^product_scan_detail_.*\.json$/.test(f)) .filter((f) => /^product_scan_detail_.*\.json$/.test(f))
.map((f) => { .map((f) => {
const p = path.join(SOURCES_DIR, f); const p = path.join(SOURCES_DIR, f);
return { path: p, mtime: fs.statSync(p).mtimeMs }; return { path: p, mtime: fs.statSync(p).mtimeMs };
}) })
.sort((a, b) => b.mtime - a.mtime); .sort((a, b) => b.mtime - a.mtime);
if (candidates.length === 0) return null; if (candidates.length === 0) return null;
const latest = candidates[0]; const latest = candidates[0];
const data = JSON.parse(fs.readFileSync(latest.path, "utf8")); const data = JSON.parse(fs.readFileSync(latest.path, "utf8"));
return { path: latest.path, data }; return { path: latest.path, data };
} }
export async function GET(req: NextRequest) { export async function GET(req: NextRequest) {
try { try {
const { searchParams } = new URL(req.url); const { searchParams } = new URL(req.url);
const filename = searchParams.get("filename"); const filename = searchParams.get("filename");
const latest = findLatestDump(); const latest = findLatestDump();
if (!latest) { if (!latest) {
return NextResponse.json({ available: false }); return NextResponse.json({ available: false });
} }
if (!filename) { if (!filename) {
return NextResponse.json({ available: true, timestamp: latest.data.timestamp }); return NextResponse.json({ available: true, timestamp: latest.data.timestamp });
} }
const item = latest.data.validation.find((v) => v.filename === filename); const item = latest.data.validation.find((v) => v.filename === filename);
if (!item) { if (!item) {
return NextResponse.json({ available: true, timestamp: latest.data.timestamp, found: false }); return NextResponse.json({ available: true, timestamp: latest.data.timestamp, found: false });
} }
const byField = Object.fromEntries(item.checks.map((c) => [c.field, c])); const byField = Object.fromEntries(item.checks.map((c) => [c.field, c]));
return NextResponse.json({ return NextResponse.json({
available: true, available: true,
found: true, found: true,
timestamp: latest.data.timestamp, timestamp: latest.data.timestamp,
method: item.method, method: item.method,
confidence: item.confidence, confidence: item.confidence,
no_sku: byField.no_sku?.predicted, no_sku: byField.no_sku?.predicted,
nama_item: byField.nama_item?.predicted, nama_item: byField.nama_item?.predicted,
expiry_date: byField.expiry_date?.predicted expiry_date: byField.expiry_date?.predicted
}); });
} catch (err: unknown) { } catch (err: unknown) {
console.error("Error in product-scan-results API:", err); console.error("Error in product-scan-results API:", err);
const message = err instanceof Error ? err.message : "Internal server error"; const message = err instanceof Error ? err.message : "Internal server error";
return errorResponse(500, message); return errorResponse(500, message);
} }
} }
@@ -1,50 +1,50 @@
import { NextRequest, NextResponse } from "next/server"; import { NextRequest, NextResponse } from "next/server";
import fs from "fs"; import fs from "fs";
import path from "path"; import path from "path";
import { errorResponse } from "@/utils/api-error"; import { errorResponse } from "@/utils/api-error";
export const dynamic = "force-dynamic"; export const dynamic = "force-dynamic";
export async function GET(req: NextRequest) { export async function GET(req: NextRequest) {
try { try {
const pfmDir = path.join(process.cwd(), "public", "produk-pfm", "foto-kemasan-v2"); const pfmDir = path.join(process.cwd(), "public", "produk-pfm", "foto-kemasan-v2");
if (!fs.existsSync(pfmDir)) { if (!fs.existsSync(pfmDir)) {
return NextResponse.json({ products: [] }); return NextResponse.json({ products: [] });
} }
const entries = fs.readdirSync(pfmDir, { withFileTypes: true }); const entries = fs.readdirSync(pfmDir, { withFileTypes: true });
const products = []; const products = [];
const ignoredNames = ["models", "runs", "yolo_dataset", ".venv", ".venv-api"]; const ignoredNames = ["models", "runs", "yolo_dataset", ".venv", ".venv-api"];
for (const entry of entries) { for (const entry of entries) {
if (entry.isDirectory() && !ignoredNames.includes(entry.name)) { if (entry.isDirectory() && !ignoredNames.includes(entry.name)) {
const productDirPath = path.join(pfmDir, entry.name); const productDirPath = path.join(pfmDir, entry.name);
const files = fs.readdirSync(productDirPath); const files = fs.readdirSync(productDirPath);
// Filter image files // Filter image files
const imageExtensions = [".jpg", ".jpeg", ".png", ".webp", ".bmp"]; const imageExtensions = [".jpg", ".jpeg", ".png", ".webp", ".bmp"];
const images = files.filter(f => const images = files.filter(f =>
imageExtensions.includes(path.extname(f).toLowerCase()) imageExtensions.includes(path.extname(f).toLowerCase())
); );
if (images.length > 0) { if (images.length > 0) {
products.push({ products.push({
productName: entry.name, productName: entry.name,
images: images.map(img => `/produk-pfm/foto-kemasan-v2/${entry.name}/${img}`), images: images.map(img => `/produk-pfm/foto-kemasan-v2/${entry.name}/${img}`),
thumbs: images.map(img => `/produk-pfm/thumbs/${entry.name}/${img}`) thumbs: images.map(img => `/produk-pfm/thumbs/${entry.name}/${img}`)
}); });
} }
} }
} }
// Sort products by name // Sort products by name
products.sort((a, b) => a.productName.localeCompare(b.productName)); products.sort((a, b) => a.productName.localeCompare(b.productName));
return NextResponse.json({ products }); return NextResponse.json({ products });
} catch (error: unknown) { } catch (error: unknown) {
console.error("Error fetching produk PFM:", error); console.error("Error fetching produk PFM:", error);
const message = error instanceof Error ? error.message : "Internal server error"; const message = error instanceof Error ? error.message : "Internal server error";
return errorResponse(500, message); return errorResponse(500, message);
} }
} }
+56 -56
View File
@@ -1,56 +1,56 @@
import { NextRequest, NextResponse } from "next/server"; import { NextRequest, NextResponse } from "next/server";
import fs from "fs"; import fs from "fs";
import path from "path"; import path from "path";
import { query } from "../../../db"; import { query } from "../../../db";
import crypto from "crypto"; import crypto from "crypto";
import { errorResponse } from "@/utils/api-error"; import { errorResponse } from "@/utils/api-error";
const UPLOADS_DIR = "/uploads"; const UPLOADS_DIR = "/uploads";
const PUBLIC_DIR = path.join(process.cwd(), "public/do-pfm"); const PUBLIC_DIR = path.join(process.cwd(), "public/do-pfm");
export async function POST(req: NextRequest) { export async function POST(req: NextRequest) {
try { try {
const { filename, image } = await req.json(); const { filename, image } = await req.json();
if (!filename || !image) { if (!filename || !image) {
return errorResponse(400, "Filename and image base64 data are required"); return errorResponse(400, "Filename and image base64 data are required");
} }
const safeFile = path.basename(filename); const safeFile = path.basename(filename);
const isSample = fs.existsSync(path.join(PUBLIC_DIR, safeFile)); const isSample = fs.existsSync(path.join(PUBLIC_DIR, safeFile));
const filePath = isSample const filePath = isSample
? path.join(PUBLIC_DIR, safeFile) ? path.join(PUBLIC_DIR, safeFile)
: path.join(UPLOADS_DIR, safeFile); : path.join(UPLOADS_DIR, safeFile);
const base64Data = image.replace(/^data:image\/\w+;base64,/, ""); const base64Data = image.replace(/^data:image\/\w+;base64,/, "");
const buffer = Buffer.from(base64Data, "base64"); const buffer = Buffer.from(base64Data, "base64");
// Write file to disk // Write file to disk
fs.writeFileSync(filePath, buffer); fs.writeFileSync(filePath, buffer);
console.log(`Rotated file saved successfully at ${filePath}`); console.log(`Rotated file saved successfully at ${filePath}`);
// Update database fields // Update database fields
const fileHash = crypto.createHash("sha256").update(buffer).digest("hex"); const fileHash = crypto.createHash("sha256").update(buffer).digest("hex");
const stats = fs.statSync(filePath); const stats = fs.statSync(filePath);
// Update document to unparsed state since layout changes // Update document to unparsed state since layout changes
await query( await query(
"UPDATE documents SET size = $1, file_hash = $2, parsed = false, layout_parsing_result = NULL WHERE filename = $3", "UPDATE documents SET size = $1, file_hash = $2, parsed = false, layout_parsing_result = NULL WHERE filename = $3",
[stats.size, fileHash, filename] [stats.size, fileHash, filename]
); );
// Clear old items for this document // Clear old items for this document
const docRes = await query("SELECT id FROM documents WHERE filename = $1", [filename]); const docRes = await query("SELECT id FROM documents WHERE filename = $1", [filename]);
if (docRes.rowCount && docRes.rowCount > 0) { if (docRes.rowCount && docRes.rowCount > 0) {
const docId = docRes.rows[0].id; const docId = docRes.rows[0].id;
await query("DELETE FROM ocr_items WHERE document_id = $1", [docId]); await query("DELETE FROM ocr_items WHERE document_id = $1", [docId]);
} }
return NextResponse.json({ success: true }); return NextResponse.json({ success: true });
} catch (error: unknown) { } catch (error: unknown) {
console.error("Error rotating file:", error); console.error("Error rotating file:", error);
const message = error instanceof Error ? error.message : "Internal server error"; const message = error instanceof Error ? error.message : "Internal server error";
return errorResponse(500, message); return errorResponse(500, message);
} }
} }
@@ -1,31 +1,31 @@
import { NextRequest, NextResponse } from "next/server"; import { NextRequest, NextResponse } from "next/server";
import { errorResponse } from "@/utils/api-error"; import { errorResponse } from "@/utils/api-error";
import { classifyAndMatchProduct, ClassifierError } from "@/utils/product-scan"; import { classifyAndMatchProduct, ClassifierError } from "@/utils/product-scan";
export const dynamic = "force-dynamic"; export const dynamic = "force-dynamic";
export async function POST(req: NextRequest) { export async function POST(req: NextRequest) {
try { try {
const body = await req.json(); const body = await req.json();
const image_base64 = body.image_base64 || body.image; const image_base64 = body.image_base64 || body.image;
if (!image_base64) { if (!image_base64) {
return errorResponse(400, "Image is required"); return errorResponse(400, "Image is required");
} }
const result = await classifyAndMatchProduct(image_base64); const result = await classifyAndMatchProduct(image_base64);
return NextResponse.json({ return NextResponse.json({
classification: result.classification, classification: result.classification,
ocr: result.ocr, ocr: result.ocr,
possibleMatches: result.possibleMatches possibleMatches: result.possibleMatches
}); });
} catch (error: unknown) { } catch (error: unknown) {
console.error("Error in scan-pfm API route:", error); console.error("Error in scan-pfm API route:", error);
if (error instanceof ClassifierError) { if (error instanceof ClassifierError) {
return errorResponse(error.status, error.message); return errorResponse(error.status, error.message);
} }
const message = error instanceof Error ? error.message : "Internal server error"; const message = error instanceof Error ? error.message : "Internal server error";
return errorResponse(500, message); return errorResponse(500, message);
} }
} }
+24 -24
View File
@@ -1,24 +1,24 @@
import { NextRequest, NextResponse } from "next/server"; import { NextRequest, NextResponse } from "next/server";
import { query } from "../../../db"; import { query } from "../../../db";
import { errorResponse } from "@/utils/api-error"; import { errorResponse } from "@/utils/api-error";
export const dynamic = "force-dynamic"; export const dynamic = "force-dynamic";
export async function GET(req: NextRequest) { export async function GET(req: NextRequest) {
try { try {
const res = await query( const res = await query(
"SELECT no_sku, nama_item FROM sku_master ORDER BY no_sku" "SELECT no_sku, nama_item FROM sku_master ORDER BY no_sku"
); );
const skus = res.rows.map(row => ({ const skus = res.rows.map(row => ({
no_sku: row.no_sku, no_sku: row.no_sku,
nama_item: row.nama_item nama_item: row.nama_item
})); }));
return NextResponse.json({ skus }); return NextResponse.json({ skus });
} catch (error: unknown) { } catch (error: unknown) {
console.error("Error in SKUs API route:", error); console.error("Error in SKUs API route:", error);
const message = error instanceof Error ? error.message : "Internal server error"; const message = error instanceof Error ? error.message : "Internal server error";
return errorResponse(500, message); return errorResponse(500, message);
} }
} }
+25 -25
View File
@@ -1,25 +1,25 @@
import { NextRequest, NextResponse } from "next/server"; import { NextRequest, NextResponse } from "next/server";
import { query } from "../../../db"; import { query } from "../../../db";
import { errorResponse } from "@/utils/api-error"; import { errorResponse } from "@/utils/api-error";
export const dynamic = "force-dynamic"; export const dynamic = "force-dynamic";
export async function GET(req: NextRequest) { export async function GET(req: NextRequest) {
try { try {
const res = await query( const res = await query(
"SELECT kode_toko, nama_toko, alamat FROM store_master ORDER BY nama_toko" "SELECT kode_toko, nama_toko, alamat FROM store_master ORDER BY nama_toko"
); );
const stores = res.rows.map(row => ({ const stores = res.rows.map(row => ({
kodeToko: row.kode_toko, kodeToko: row.kode_toko,
namaToko: row.nama_toko, namaToko: row.nama_toko,
alamat: row.alamat alamat: row.alamat
})); }));
return NextResponse.json({ stores }); return NextResponse.json({ stores });
} catch (error: unknown) { } catch (error: unknown) {
console.error("Error in stores API route:", error); console.error("Error in stores API route:", error);
const message = error instanceof Error ? error.message : "Internal server error"; const message = error instanceof Error ? error.message : "Internal server error";
return errorResponse(500, message); return errorResponse(500, message);
} }
} }
@@ -1,233 +1,233 @@
import { NextRequest, NextResponse } from "next/server"; import { NextRequest, NextResponse } from "next/server";
import { query } from "../../../db"; import { query } from "../../../db";
import { correctVisualDigits } from "../../../utils/parser"; import { correctVisualDigits } from "../../../utils/parser";
export const dynamic = "force-dynamic"; export const dynamic = "force-dynamic";
function levenshteinDistance(s1: string, s2: string): number { function levenshteinDistance(s1: string, s2: string): number {
const len1 = s1.length; const len1 = s1.length;
const len2 = s2.length; const len2 = s2.length;
const matrix = Array.from({ length: len1 + 1 }, () => new Array(len2 + 1).fill(0)); const matrix = Array.from({ length: len1 + 1 }, () => new Array(len2 + 1).fill(0));
for (let i = 0; i <= len1; i++) matrix[i][0] = i; for (let i = 0; i <= len1; i++) matrix[i][0] = i;
for (let j = 0; j <= len2; j++) matrix[0][j] = j; for (let j = 0; j <= len2; j++) matrix[0][j] = j;
for (let i = 1; i <= len1; i++) { for (let i = 1; i <= len1; i++) {
for (let j = 1; j <= len2; j++) { for (let j = 1; j <= len2; j++) {
const cost = s1[i - 1] === s2[j - 1] ? 0 : 1; const cost = s1[i - 1] === s2[j - 1] ? 0 : 1;
matrix[i][j] = Math.min( matrix[i][j] = Math.min(
matrix[i - 1][j] + 1, // deletion matrix[i - 1][j] + 1, // deletion
matrix[i][j - 1] + 1, // insertion matrix[i][j - 1] + 1, // insertion
matrix[i - 1][j - 1] + cost // substitution matrix[i - 1][j - 1] + cost // substitution
); );
} }
} }
return matrix[len1][len2]; return matrix[len1][len2];
} }
function getStringSimilarity(s1: string, s2: string): number { function getStringSimilarity(s1: string, s2: string): number {
const clean1 = s1.toLowerCase().replace(/[^a-z0-9]/g, ''); const clean1 = s1.toLowerCase().replace(/[^a-z0-9]/g, '');
const clean2 = s2.toLowerCase().replace(/[^a-z0-9]/g, ''); const clean2 = s2.toLowerCase().replace(/[^a-z0-9]/g, '');
if (!clean1 || !clean2) return 0; if (!clean1 || !clean2) return 0;
const distance = levenshteinDistance(clean1, clean2); const distance = levenshteinDistance(clean1, clean2);
const maxLength = Math.max(clean1.length, clean2.length); const maxLength = Math.max(clean1.length, clean2.length);
return (maxLength - distance) / maxLength; return (maxLength - distance) / maxLength;
} }
// Re-implement cleanDateValue directly so we don't have to deal with exports issues if any // Re-implement cleanDateValue directly so we don't have to deal with exports issues if any
const MONTHS_MAP: Record<string, string> = { const MONTHS_MAP: Record<string, string> = {
january: "January", januari: "January", janov: "January", jan: "January", january: "January", januari: "January", janov: "January", jan: "January",
february: "February", februari: "February", feb: "February", february: "February", februari: "February", feb: "February",
march: "March", maret: "March", mar: "March", march: "March", maret: "March", mar: "March",
april: "April", apr: "April", april: "April", apr: "April",
may: "May", mei: "May", may: "May", mei: "May",
june: "June", juni: "June", jun: "June", june: "June", juni: "June", jun: "June",
july: "July", juli: "July", jul: "July", july: "July", juli: "July", jul: "July",
august: "August", agustus: "August", agt: "August", ags: "August", aug: "August", august: "August", agustus: "August", agt: "August", ags: "August", aug: "August",
september: "September", sept: "September", sep: "September", september: "September", sept: "September", sep: "September",
oktober: "October", october: "October", okt: "October", oct: "October", oktober: "October", october: "October", okt: "October", oct: "October",
november: "November", nopember: "November", nov: "November", november: "November", nopember: "November", nov: "November",
desember: "December", december: "December", des: "December", dec: "December" desember: "December", december: "December", des: "December", dec: "December"
}; };
function cleanDateValue(raw: string): string { function cleanDateValue(raw: string): string {
if (!raw) return "Not Found"; if (!raw) return "Not Found";
const cleaned = raw.trim(); const cleaned = raw.trim();
if (cleaned === "Not Found" || cleaned === "") return "Not Found"; if (cleaned === "Not Found" || cleaned === "") return "Not Found";
const today = new Date(); const today = new Date();
let day: number | null = null; let day: number | null = null;
let monthStr: string | null = null; let monthStr: string | null = null;
let year: number | null = null; let year: number | null = null;
const yearMatch = cleaned.match(/\b(20\d{2})\b/); const yearMatch = cleaned.match(/\b(20\d{2})\b/);
if (yearMatch) { if (yearMatch) {
const parsedYear = parseInt(yearMatch[1], 10); const parsedYear = parseInt(yearMatch[1], 10);
if (parsedYear >= 2010 && parsedYear <= 2035) { if (parsedYear >= 2010 && parsedYear <= 2035) {
year = parsedYear; year = parsedYear;
} }
} }
const lowerRaw = cleaned.toLowerCase(); const lowerRaw = cleaned.toLowerCase();
const monthsKeys = Object.keys(MONTHS_MAP); const monthsKeys = Object.keys(MONTHS_MAP);
monthsKeys.sort((a, b) => b.length - a.length); monthsKeys.sort((a, b) => b.length - a.length);
for (const key of monthsKeys) { for (const key of monthsKeys) {
if (lowerRaw.includes(key)) { if (lowerRaw.includes(key)) {
monthStr = MONTHS_MAP[key] || null; monthStr = MONTHS_MAP[key] || null;
break; break;
} }
} }
let textForDay = cleaned; let textForDay = cleaned;
if (year) { if (year) {
textForDay = textForDay.replace(year.toString(), ""); textForDay = textForDay.replace(year.toString(), "");
} }
const dayMatches = textForDay.match(/\b(\d{1,2})\b/g); const dayMatches = textForDay.match(/\b(\d{1,2})\b/g);
if (dayMatches) { if (dayMatches) {
for (const matchStr of dayMatches) { for (const matchStr of dayMatches) {
const parsedDay = parseInt(matchStr, 10); const parsedDay = parseInt(matchStr, 10);
if (parsedDay >= 1 && parsedDay <= 31) { if (parsedDay >= 1 && parsedDay <= 31) {
day = parsedDay; day = parsedDay;
break; break;
} }
} }
} }
const currentYear = today.getFullYear(); const currentYear = today.getFullYear();
const currentMonthNames = ["January", "February", "March", "April", "May", "June", "July", "August", "September", "October", "November", "December"]; const currentMonthNames = ["January", "February", "March", "April", "May", "June", "July", "August", "September", "October", "November", "December"];
const currentMonth = currentMonthNames[today.getMonth()]; const currentMonth = currentMonthNames[today.getMonth()];
const currentDay = today.getDate(); const currentDay = today.getDate();
const finalDay = day !== null ? day : currentDay; const finalDay = day !== null ? day : currentDay;
const finalMonth = monthStr !== null ? monthStr : currentMonth; const finalMonth = monthStr !== null ? monthStr : currentMonth;
const finalYear = year !== null ? year : currentYear; const finalYear = year !== null ? year : currentYear;
return `${finalDay} ${finalMonth} ${finalYear}`; return `${finalDay} ${finalMonth} ${finalYear}`;
} }
export async function GET(req: NextRequest) { export async function GET(req: NextRequest) {
const results: string[] = []; const results: string[] = [];
let passed = true; let passed = true;
const assert = (condition: boolean, desc: string) => { const assert = (condition: boolean, desc: string) => {
if (condition) { if (condition) {
results.push(`[PASS] ${desc}`); results.push(`[PASS] ${desc}`);
} else { } else {
results.push(`[FAIL] ${desc}`); results.push(`[FAIL] ${desc}`);
passed = false; passed = false;
} }
}; };
// 1. Test Visual Digit Correction // 1. Test Visual Digit Correction
const so1 = correctVisualDigits("16O29B7162"); const so1 = correctVisualDigits("16O29B7162");
assert(so1 === "1602987162", `correctVisualDigits("16O29B7162") -> got "${so1}", expected "1602987162"`); assert(so1 === "1602987162", `correctVisualDigits("16O29B7162") -> got "${so1}", expected "1602987162"`);
const do1 = correctVisualDigits("1602l87"); const do1 = correctVisualDigits("1602l87");
assert(do1 === "1602187", `correctVisualDigits("1602l87") -> got "${do1}", expected "1602187"`); assert(do1 === "1602187", `correctVisualDigits("1602l87") -> got "${do1}", expected "1602187"`);
const so2 = correctVisualDigits("16O29B7162-OK"); const so2 = correctVisualDigits("16O29B7162-OK");
assert(so2 === "1602987162", `correctVisualDigits("16O29B7162-OK") -> got "${so2}", expected "1602987162"`); assert(so2 === "1602987162", `correctVisualDigits("16O29B7162-OK") -> got "${so2}", expected "1602987162"`);
// 2. Test Date Lenient Parsing & Fallback Auto-Fill // 2. Test Date Lenient Parsing & Fallback Auto-Fill
const today = new Date(); const today = new Date();
const currentMonthNames = ["January", "February", "March", "April", "May", "June", "July", "August", "September", "October", "November", "December"]; const currentMonthNames = ["January", "February", "March", "April", "May", "June", "July", "August", "September", "October", "November", "December"];
const currentMonth = currentMonthNames[today.getMonth()]; const currentMonth = currentMonthNames[today.getMonth()];
const currentDay = today.getDate(); const currentDay = today.getDate();
const currentYear = today.getFullYear(); const currentYear = today.getFullYear();
const d1 = cleanDateValue("30-Hv-2026"); const d1 = cleanDateValue("30-Hv-2026");
assert(d1 === `30 ${currentMonth} 2026`, `cleanDateValue("30-Hv-2026") -> got "${d1}", expected "30 ${currentMonth} 2026"`); assert(d1 === `30 ${currentMonth} 2026`, `cleanDateValue("30-Hv-2026") -> got "${d1}", expected "30 ${currentMonth} 2026"`);
const d2 = cleanDateValue("Hv-Jan-2026"); const d2 = cleanDateValue("Hv-Jan-2026");
assert(d2 === `${currentDay} January 2026`, `cleanDateValue("Hv-Jan-2026") -> got "${d2}", expected "${currentDay} January 2026"`); assert(d2 === `${currentDay} January 2026`, `cleanDateValue("Hv-Jan-2026") -> got "${d2}", expected "${currentDay} January 2026"`);
const d3 = cleanDateValue("30-Jan"); const d3 = cleanDateValue("30-Jan");
assert(d3 === `30 January ${currentYear}`, `cleanDateValue("30-Jan") -> got "${d3}", expected "30 January ${currentYear}"`); assert(d3 === `30 January ${currentYear}`, `cleanDateValue("30-Jan") -> got "${d3}", expected "30 January ${currentYear}"`);
// 3. Test Two-Way Database SKU Cross-Check // 3. Test Two-Way Database SKU Cross-Check
try { try {
const skuDbRes = await query("SELECT no_sku, nama_item FROM sku_master"); const skuDbRes = await query("SELECT no_sku, nama_item FROM sku_master");
const skuMasterList = skuDbRes.rows.map(row => ({ const skuMasterList = skuDbRes.rows.map(row => ({
no_sku: row.no_sku.toString().trim(), no_sku: row.no_sku.toString().trim(),
nama_item: row.nama_item.toString().trim() nama_item: row.nama_item.toString().trim()
})); }));
// Mock an OCR parsed items list // Mock an OCR parsed items list
const items = [ const items = [
{ {
kodeBarang: "11048006", kodeBarang: "11048006",
namaBarang: "BEBEK PARTING wrong ocr text", namaBarang: "BEBEK PARTING wrong ocr text",
banyak: "10 BAG", banyak: "10 BAG",
jumlah: "100000" jumlah: "100000"
}, },
{ {
kodeBarang: "Not Found", kodeBarang: "Not Found",
namaBarang: "CEKER BERKUKU FROZEN PACK", namaBarang: "CEKER BERKUKU FROZEN PACK",
banyak: "20 KRG", banyak: "20 KRG",
jumlah: "200000" jumlah: "200000"
}, },
{ {
kodeBarang: "Not Found", kodeBarang: "Not Found",
namaBarang: "Tanda Tangan Supit", namaBarang: "Tanda Tangan Supit",
banyak: "Bag. Pengeluaran Barang", banyak: "Bag. Pengeluaran Barang",
jumlah: "Bagian Penjualan" jumlah: "Bagian Penjualan"
} }
]; ];
const checkedItems: typeof items = []; const checkedItems: typeof items = [];
for (const item of items) { for (const item of items) {
const ocrSku = item.kodeBarang ? item.kodeBarang.trim() : ""; const ocrSku = item.kodeBarang ? item.kodeBarang.trim() : "";
const ocrName = item.namaBarang ? item.namaBarang.trim() : ""; const ocrName = item.namaBarang ? item.namaBarang.trim() : "";
const matchedBySku = /^\d{8}$/.test(ocrSku) ? skuMasterList.find(sku => sku.no_sku === ocrSku) : null; const matchedBySku = /^\d{8}$/.test(ocrSku) ? skuMasterList.find(sku => sku.no_sku === ocrSku) : null;
if (matchedBySku) { if (matchedBySku) {
item.kodeBarang = matchedBySku.no_sku; item.kodeBarang = matchedBySku.no_sku;
item.namaBarang = matchedBySku.nama_item; item.namaBarang = matchedBySku.nama_item;
checkedItems.push(item); checkedItems.push(item);
} else { } else {
let bestMatch: typeof skuMasterList[0] | null = null; let bestMatch: typeof skuMasterList[0] | null = null;
let bestScore = 0; let bestScore = 0;
for (const sku of skuMasterList) { for (const sku of skuMasterList) {
const score = getStringSimilarity(sku.nama_item, ocrName); const score = getStringSimilarity(sku.nama_item, ocrName);
if (score > bestScore) { if (score > bestScore) {
bestScore = score; bestScore = score;
bestMatch = sku; bestMatch = sku;
} }
} }
if (bestMatch && bestScore >= 0.6) { if (bestMatch && bestScore >= 0.6) {
item.kodeBarang = bestMatch.no_sku; item.kodeBarang = bestMatch.no_sku;
item.namaBarang = bestMatch.nama_item; item.namaBarang = bestMatch.nama_item;
checkedItems.push(item); checkedItems.push(item);
} else { } else {
if (/^\d{8}$/.test(ocrSku)) { if (/^\d{8}$/.test(ocrSku)) {
checkedItems.push(item); checkedItems.push(item);
} }
} }
} }
} }
// Verify checkedItems length (noise item discarded) // Verify checkedItems length (noise item discarded)
assert(checkedItems.length === 2, `checkedItems length should be 2, got ${checkedItems.length} (noise footer row successfully discarded)`); assert(checkedItems.length === 2, `checkedItems length should be 2, got ${checkedItems.length} (noise footer row successfully discarded)`);
// Verify item 1 description correction // Verify item 1 description correction
assert(checkedItems[0].kodeBarang === "11048006", "Item 1 SKU should remain 11048006"); assert(checkedItems[0].kodeBarang === "11048006", "Item 1 SKU should remain 11048006");
assert(checkedItems[0].namaBarang === "BEBEK PARTING-NEW(*)", `Item 1 name corrected from DB -> got "${checkedItems[0].namaBarang}"`); assert(checkedItems[0].namaBarang === "BEBEK PARTING-NEW(*)", `Item 1 name corrected from DB -> got "${checkedItems[0].namaBarang}"`);
// Verify item 2 SKU fuzzy autocomplete from description // Verify item 2 SKU fuzzy autocomplete from description
assert(checkedItems[1].kodeBarang === "11110059", `Item 2 SKU autocompleted from DB -> got "${checkedItems[1].kodeBarang}"`); assert(checkedItems[1].kodeBarang === "11110059", `Item 2 SKU autocompleted from DB -> got "${checkedItems[1].kodeBarang}"`);
assert(checkedItems[1].namaBarang === "CEKER BERKUKU FROZEN PACK 1 KG(*)", `Item 2 name corrected from DB -> got "${checkedItems[1].namaBarang}"`); assert(checkedItems[1].namaBarang === "CEKER BERKUKU FROZEN PACK 1 KG(*)", `Item 2 name corrected from DB -> got "${checkedItems[1].namaBarang}"`);
} catch (err: any) { } catch (err: any) {
passed = false; passed = false;
results.push(`[ERROR] Database SKU check failed: ${err.message}`); results.push(`[ERROR] Database SKU check failed: ${err.message}`);
} }
return NextResponse.json({ return NextResponse.json({
status: passed ? "success" : "failed", status: passed ? "success" : "failed",
results results
}); });
} }
@@ -1,72 +1,72 @@
import { NextRequest, NextResponse } from "next/server"; import { NextRequest, NextResponse } from "next/server";
import { query } from "../../../db"; import { query } from "../../../db";
import path from "path"; import path from "path";
import { errorResponse } from "@/utils/api-error"; import { errorResponse } from "@/utils/api-error";
export async function POST(req: NextRequest) { export async function POST(req: NextRequest) {
try { try {
const body = await req.json(); const body = await req.json();
const { page, rowIndex, action } = body; const { page, rowIndex, action } = body;
if (!page || rowIndex === undefined || !action) { if (!page || rowIndex === undefined || !action) {
return errorResponse(400, "Missing required fields"); return errorResponse(400, "Missing required fields");
} }
const safeFile = path.basename(page); const safeFile = path.basename(page);
// Get document ID // Get document ID
const docRes = await query("SELECT id FROM documents WHERE filename = $1", [safeFile]); const docRes = await query("SELECT id FROM documents WHERE filename = $1", [safeFile]);
if (!docRes.rowCount || docRes.rowCount === 0) { if (!docRes.rowCount || docRes.rowCount === 0) {
return errorResponse(404, "Document not found in database"); return errorResponse(404, "Document not found in database");
} }
const docId = docRes.rows[0].id; const docId = docRes.rows[0].id;
if (action === "edit") { if (action === "edit") {
const { field, value } = body; const { field, value } = body;
if (!field || value === undefined) { if (!field || value === undefined) {
return errorResponse(400, "Missing edit parameters"); return errorResponse(400, "Missing edit parameters");
} }
// Map UI field names to database columns // Map UI field names to database columns
let colName = ""; let colName = "";
if (field === "kodeBarang") { if (field === "kodeBarang") {
colName = "kode_barang"; colName = "kode_barang";
} else if (field === "banyak") { } else if (field === "banyak") {
colName = "banyak"; colName = "banyak";
} else if (field === "jumlah") { } else if (field === "jumlah") {
colName = "jumlah"; colName = "jumlah";
} else { } else {
return errorResponse(400, "Invalid field name"); return errorResponse(400, "Invalid field name");
} }
await query( await query(
`UPDATE ocr_items `UPDATE ocr_items
SET ${colName} = $1 SET ${colName} = $1
WHERE document_id = $2 AND row_index = $3`, WHERE document_id = $2 AND row_index = $3`,
[value, docId, rowIndex] [value, docId, rowIndex]
); );
return NextResponse.json({ success: true }); return NextResponse.json({ success: true });
} else if (action === "flag") { } else if (action === "flag") {
const { isFlagged, remark } = body; const { isFlagged, remark } = body;
if (isFlagged === undefined || remark === undefined) { if (isFlagged === undefined || remark === undefined) {
return errorResponse(400, "Missing flag parameters"); return errorResponse(400, "Missing flag parameters");
} }
await query( await query(
`UPDATE ocr_items `UPDATE ocr_items
SET is_flagged = $1, remark = $2 SET is_flagged = $1, remark = $2
WHERE document_id = $3 AND row_index = $4`, WHERE document_id = $3 AND row_index = $4`,
[!!isFlagged, remark, docId, rowIndex] [!!isFlagged, remark, docId, rowIndex]
); );
return NextResponse.json({ success: true }); return NextResponse.json({ success: true });
} else { } else {
return errorResponse(400, "Invalid action"); return errorResponse(400, "Invalid action");
} }
} catch (error: unknown) { } catch (error: unknown) {
console.error("Error in update-row API route:", error); console.error("Error in update-row API route:", error);
const message = error instanceof Error ? error.message : "Internal server error"; const message = error instanceof Error ? error.message : "Internal server error";
return errorResponse(500, message); return errorResponse(500, message);
} }
} }
+241 -241
View File
@@ -1,241 +1,241 @@
import { NextRequest, NextResponse } from "next/server"; import { NextRequest, NextResponse } from "next/server";
import fs from "fs"; import fs from "fs";
import path from "path"; import path from "path";
import crypto from "crypto"; import crypto from "crypto";
import { query, resolveStoreFromText } from "../../../db"; import { query, resolveStoreFromText } from "../../../db";
import { parseDOMetadata, sanitizeParsedMetadata } from "../../../utils/parser"; import { parseDOMetadata, sanitizeParsedMetadata } from "../../../utils/parser";
import { startActiveLog, getActiveLog, clearActiveLog } from "../../../utils/active-log"; import { startActiveLog, getActiveLog, clearActiveLog } from "../../../utils/active-log";
import { errorResponse } from "@/utils/api-error"; import { errorResponse } from "@/utils/api-error";
const UPLOADS_DIR = "/uploads"; const UPLOADS_DIR = "/uploads";
export async function POST(req: NextRequest) { export async function POST(req: NextRequest) {
try { try {
// Ensure uploads directory exists // Ensure uploads directory exists
if (!fs.existsSync(UPLOADS_DIR)) { if (!fs.existsSync(UPLOADS_DIR)) {
fs.mkdirSync(UPLOADS_DIR, { recursive: true }); fs.mkdirSync(UPLOADS_DIR, { recursive: true });
} }
const formData = await req.formData(); const formData = await req.formData();
const file = formData.get("file") as Blob | null; const file = formData.get("file") as Blob | null;
if (!file) { if (!file) {
return errorResponse(400, "No file uploaded"); return errorResponse(400, "No file uploaded");
} }
const originalName = file instanceof File ? file.name : "document.jpg"; const originalName = file instanceof File ? file.name : "document.jpg";
// Sanitize filename to avoid directory traversal // Sanitize filename to avoid directory traversal
const safeName = path.basename(originalName).replace(/\s+/g, "_"); const safeName = path.basename(originalName).replace(/\s+/g, "_");
const filename = `${Date.now()}-${safeName}`; const filename = `${Date.now()}-${safeName}`;
const filePath = path.join(UPLOADS_DIR, filename); const filePath = path.join(UPLOADS_DIR, filename);
// Save file // Save file
const arrayBuffer = await file.arrayBuffer(); const arrayBuffer = await file.arrayBuffer();
const buffer = Buffer.from(arrayBuffer); const buffer = Buffer.from(arrayBuffer);
// Compute hash to check for duplicate content // Compute hash to check for duplicate content
const fileHash = crypto.createHash("sha256").update(buffer).digest("hex"); const fileHash = crypto.createHash("sha256").update(buffer).digest("hex");
fs.writeFileSync(filePath, buffer); fs.writeFileSync(filePath, buffer);
// Convert to base64 for pipeline API // Convert to base64 for pipeline API
const b64 = buffer.toString("base64"); const b64 = buffer.toString("base64");
// Form payload // Form payload
const payload = { const payload = {
file: b64, file: b64,
matchHistoryJob: false, matchHistoryJob: false,
useLayoutDetection: true, useLayoutDetection: true,
fileType: 1, fileType: 1,
useDocUnwarping: false, useDocUnwarping: false,
useDocOrientationClassify: true useDocOrientationClassify: true
}; };
const pipelineUrl = process.env.PIPELINE_URL || "http://paddleocr-pipeline-api:8090/layout-parsing"; const pipelineUrl = process.env.PIPELINE_URL || "http://paddleocr-pipeline-api:8090/layout-parsing";
console.log(`Forwarding uploaded file ${filename} to pipeline: ${pipelineUrl}`); console.log(`Forwarding uploaded file ${filename} to pipeline: ${pipelineUrl}`);
startActiveLog(filename); startActiveLog(filename);
const response = await fetch(pipelineUrl, { const response = await fetch(pipelineUrl, {
method: "POST", method: "POST",
headers: { headers: {
"Content-Type": "application/json" "Content-Type": "application/json"
}, },
body: JSON.stringify(payload) body: JSON.stringify(payload)
}); });
if (!response.ok) { if (!response.ok) {
clearActiveLog(filename); clearActiveLog(filename);
const errText = await response.text(); const errText = await response.text();
return errorResponse(response.status, `Pipeline API error: ${errText}`); return errorResponse(response.status, `Pipeline API error: ${errText}`);
} }
let data = await response.json(); let data = await response.json();
// Check if the image is not straight (tilt > 1.0 degree) // Check if the image is not straight (tilt > 1.0 degree)
const tilt = calculateAverageTilt(data); const tilt = calculateAverageTilt(data);
let unwarped = false; let unwarped = false;
if (tilt > 1.0) { if (tilt > 1.0) {
console.log(`Uploaded document ${filename} is not straight (average tilt: ${tilt.toFixed(2)} deg). Re-running with unwarping and orientation classification enabled...`); console.log(`Uploaded document ${filename} is not straight (average tilt: ${tilt.toFixed(2)} deg). Re-running with unwarping and orientation classification enabled...`);
const unwarpPayload = { const unwarpPayload = {
...payload, ...payload,
useDocUnwarping: true, useDocUnwarping: true,
useDocOrientationClassify: true useDocOrientationClassify: true
}; };
const unwarpResponse = await fetch(pipelineUrl, { const unwarpResponse = await fetch(pipelineUrl, {
method: "POST", method: "POST",
headers: { headers: {
"Content-Type": "application/json" "Content-Type": "application/json"
}, },
body: JSON.stringify(unwarpPayload) body: JSON.stringify(unwarpPayload)
}); });
if (unwarpResponse.ok) { if (unwarpResponse.ok) {
data = await unwarpResponse.json(); data = await unwarpResponse.json();
console.log(`Document unwarped successfully.`); console.log(`Document unwarped successfully.`);
unwarped = true; unwarped = true;
} else { } else {
console.error(`Unwarping failed with status ${unwarpResponse.status}`); console.error(`Unwarping failed with status ${unwarpResponse.status}`);
} }
} }
// Save JSON extraction result // Save JSON extraction result
const jsonPath = `${filePath}.json`; const jsonPath = `${filePath}.json`;
fs.writeFileSync(jsonPath, JSON.stringify(data, null, 2)); fs.writeFileSync(jsonPath, JSON.stringify(data, null, 2));
// Save to PostgreSQL database // Save to PostgreSQL database
try { try {
const pipelineResult = data.result || data; const pipelineResult = data.result || data;
pipelineResult.pipeline_info = { pipelineResult.pipeline_info = {
tilt, tilt,
unwarped, unwarped,
original_tilt: tilt original_tilt: tilt
}; };
const page0 = pipelineResult?.layoutParsingResults?.[0] || {}; const page0 = pipelineResult?.layoutParsingResults?.[0] || {};
const markdownText = page0?.markdown?.text || ""; const markdownText = page0?.markdown?.text || "";
const docMetadata = parseDOMetadata(markdownText); const docMetadata = parseDOMetadata(markdownText);
// Resolve store information using master database // Resolve store information using master database
const resolvedStore = await resolveStoreFromText(markdownText); const resolvedStore = await resolveStoreFromText(markdownText);
(docMetadata as any).orderUntuk = resolvedStore.orderUntuk; (docMetadata as any).orderUntuk = resolvedStore.orderUntuk;
(docMetadata as any).alamat = resolvedStore.alamat; (docMetadata as any).alamat = resolvedStore.alamat;
// Stage 2 Filtering: Sanitize parsed metadata // Stage 2 Filtering: Sanitize parsed metadata
const sanitizedMetadata = sanitizeParsedMetadata(docMetadata as any); const sanitizedMetadata = sanitizeParsedMetadata(docMetadata as any);
// Construct client response representation // Construct client response representation
const wrappedResult = { const wrappedResult = {
errorCode: 0, errorCode: 0,
errorMsg: "Success", errorMsg: "Success",
result: pipelineResult result: pipelineResult
}; };
const clientResponse = { const clientResponse = {
filename, filename,
result: wrappedResult result: wrappedResult
}; };
// Retrieve and finalize active log data // Retrieve and finalize active log data
const activeLog = getActiveLog(filename); const activeLog = getActiveLog(filename);
let logsPayload: any = null; let logsPayload: any = null;
if (activeLog && activeLog.filename === filename) { if (activeLog && activeLog.filename === filename) {
activeLog.ocr_raw = pipelineResult; activeLog.ocr_raw = pipelineResult;
activeLog.stage_1_output = docMetadata; activeLog.stage_1_output = docMetadata;
activeLog.stage_2_output = sanitizedMetadata; activeLog.stage_2_output = sanitizedMetadata;
activeLog.frontend_response = clientResponse; activeLog.frontend_response = clientResponse;
activeLog.pipeline_info = { activeLog.pipeline_info = {
tilt, tilt,
unwarped, unwarped,
original_tilt: tilt original_tilt: tilt
}; };
logsPayload = { ...activeLog }; logsPayload = { ...activeLog };
} }
clearActiveLog(filename); clearActiveLog(filename);
const insertDocRes = await query(` const insertDocRes = await query(`
INSERT INTO documents (filename, upload_time, size, parsed, metadata, layout_parsing_result, is_sample, file_hash, processing_logs) INSERT INTO documents (filename, upload_time, size, parsed, metadata, layout_parsing_result, is_sample, file_hash, processing_logs)
VALUES ($1, $2, $3, $4, $5, $6, $7, $8, $9) VALUES ($1, $2, $3, $4, $5, $6, $7, $8, $9)
RETURNING id RETURNING id
`, [ `, [
filename, filename,
new Date(), new Date(),
buffer.length, buffer.length,
true, true,
JSON.stringify(sanitizedMetadata), JSON.stringify(sanitizedMetadata),
JSON.stringify(pipelineResult), JSON.stringify(pipelineResult),
false, false,
fileHash, fileHash,
logsPayload ? JSON.stringify(logsPayload) : null logsPayload ? JSON.stringify(logsPayload) : null
]); ]);
const docId = insertDocRes.rows[0].id; const docId = insertDocRes.rows[0].id;
for (let i = 0; i < docMetadata.items.length; i++) { for (let i = 0; i < docMetadata.items.length; i++) {
const item = docMetadata.items[i]; const item = docMetadata.items[i];
await query(` await query(`
INSERT INTO ocr_items ( INSERT INTO ocr_items (
document_id, row_index, document_id, row_index,
kode_barang_original, kode_barang, kode_barang_original, kode_barang,
nama_barang, nama_barang,
banyak_original, banyak, banyak_original, banyak,
jumlah_original, jumlah, jumlah_original, jumlah,
is_flagged, remark is_flagged, remark
) )
VALUES ($1, $2, $3, $3, $4, $5, $5, $6, $6, false, '') VALUES ($1, $2, $3, $3, $4, $5, $5, $6, $6, false, '')
ON CONFLICT DO NOTHING ON CONFLICT DO NOTHING
`, [ `, [
docId, docId,
i, i,
item.kodeBarang, item.kodeBarang,
item.namaBarang, item.namaBarang,
item.banyak, item.banyak,
item.jumlah item.jumlah
]); ]);
} }
} catch (dbErr) { } catch (dbErr) {
console.error("Database save failed during upload (falling back to file):", dbErr); console.error("Database save failed during upload (falling back to file):", dbErr);
} }
const wrappedResult = { const wrappedResult = {
errorCode: 0, errorCode: 0,
errorMsg: "Success", errorMsg: "Success",
result: data.result || data result: data.result || data
}; };
return NextResponse.json({ return NextResponse.json({
filename, filename,
result: wrappedResult result: wrappedResult
}); });
} catch (error: unknown) { } catch (error: unknown) {
console.error("Error in upload API route:", error); console.error("Error in upload API route:", error);
const message = error instanceof Error ? error.message : "Internal server error"; const message = error instanceof Error ? error.message : "Internal server error";
return errorResponse(500, message); return errorResponse(500, message);
} }
} }
function getBlockAngle(points: number[][]) { function getBlockAngle(points: number[][]) {
if (!points || points.length < 2) return 0; if (!points || points.length < 2) return 0;
const p0 = points[0]; const p0 = points[0];
const p1 = points[1]; const p1 = points[1];
const dx = p1[0] - p0[0]; const dx = p1[0] - p0[0];
const dy = p1[1] - p0[1]; const dy = p1[1] - p0[1];
let angle = Math.atan2(dy, dx) * 180 / Math.PI; let angle = Math.atan2(dy, dx) * 180 / Math.PI;
if (angle < -45) angle = 90 + angle; if (angle < -45) angle = 90 + angle;
if (angle > 45) angle = angle - 90; if (angle > 45) angle = angle - 90;
return Math.abs(angle); return Math.abs(angle);
} }
function calculateAverageTilt(data: any): number { function calculateAverageTilt(data: any): number {
const results = data?.result?.layoutParsingResults || data?.layoutParsingResults || []; const results = data?.result?.layoutParsingResults || data?.layoutParsingResults || [];
if (results.length === 0) return 0; if (results.length === 0) return 0;
const list = results[0]?.prunedResult?.parsing_res_list || []; const list = results[0]?.prunedResult?.parsing_res_list || [];
if (list.length === 0) return 0; if (list.length === 0) return 0;
const angles: number[] = []; const angles: number[] = [];
for (const block of list) { for (const block of list) {
if (block.block_polygon_points) { if (block.block_polygon_points) {
angles.push(getBlockAngle(block.block_polygon_points)); angles.push(getBlockAngle(block.block_polygon_points));
} }
} }
if (angles.length === 0) return 0; if (angles.length === 0) return 0;
return angles.reduce((sum, a) => sum + a, 0) / angles.length; return angles.reduce((sum, a) => sum + a, 0) / angles.length;
} }
@@ -1,75 +1,75 @@
import { NextRequest, NextResponse } from "next/server"; import { NextRequest, NextResponse } from "next/server";
import bcrypt from "bcryptjs"; import bcrypt from "bcryptjs";
import { errorResponse } from "@/utils/api-error"; import { errorResponse } from "@/utils/api-error";
import { signAccountToken } from "@/utils/auth"; import { signAccountToken } from "@/utils/auth";
import { query } from "../../../../../db"; import { query } from "../../../../../db";
const corsHeaders = { const corsHeaders = {
"Access-Control-Allow-Origin": "*", "Access-Control-Allow-Origin": "*",
"Access-Control-Allow-Methods": "GET, POST, PUT, DELETE, OPTIONS", "Access-Control-Allow-Methods": "GET, POST, PUT, DELETE, OPTIONS",
"Access-Control-Allow-Headers": "Content-Type, Authorization" "Access-Control-Allow-Headers": "Content-Type, Authorization"
}; };
export async function OPTIONS() { export async function OPTIONS() {
return new NextResponse(null, { status: 204, headers: corsHeaders }); return new NextResponse(null, { status: 204, headers: corsHeaders });
} }
export async function POST(req: NextRequest) { export async function POST(req: NextRequest) {
try { try {
const body = await req.json(); const body = await req.json();
const { username, password } = body; const { username, password } = body;
if (!username || !password) { if (!username || !password) {
return errorResponse(401, "Invalid username or password", { headers: corsHeaders }); return errorResponse(401, "Invalid username or password", { headers: corsHeaders });
} }
// Each account is assigned exactly one store (kode_toko) - the token // Each account is assigned exactly one store (kode_toko) - the token
// carries that assignment so store name/address never need OCR // carries that assignment so store name/address never need OCR
// detection later; whichever account uploads, its own store is used. // detection later; whichever account uploads, its own store is used.
const accountRes = await query( const accountRes = await query(
`SELECT a.id, a.username, a.password, a.role, a.is_active, `SELECT a.id, a.username, a.password, a.role, a.is_active,
s.kode_toko, s.nama_toko, s.alamat s.kode_toko, s.nama_toko, s.alamat
FROM accounts a FROM accounts a
LEFT JOIN store_master s ON a.kode_toko = s.kode_toko LEFT JOIN store_master s ON a.kode_toko = s.kode_toko
WHERE a.username = $1`, WHERE a.username = $1`,
[username] [username]
); );
if (accountRes.rowCount && accountRes.rowCount > 0 && bcrypt.compareSync(password, accountRes.rows[0].password)) { if (accountRes.rowCount && accountRes.rowCount > 0 && bcrypt.compareSync(password, accountRes.rows[0].password)) {
const account = accountRes.rows[0]; const account = accountRes.rows[0];
if (!account.is_active) { if (!account.is_active) {
return errorResponse(401, "Account is disabled", { headers: corsHeaders }); return errorResponse(401, "Account is disabled", { headers: corsHeaders });
} }
const token = signAccountToken({ const token = signAccountToken({
accountId: account.id, accountId: account.id,
username: account.username, username: account.username,
kodeToko: account.kode_toko, kodeToko: account.kode_toko,
role: account.role role: account.role
}); });
return NextResponse.json({ return NextResponse.json({
status: "success", status: "success",
message: "Login successful", message: "Login successful",
data: { data: {
token, token,
profile: { profile: {
username: account.username, username: account.username,
role: account.role, role: account.role,
is_active: account.is_active, is_active: account.is_active,
kodeToko: account.kode_toko, kodeToko: account.kode_toko,
namaToko: account.nama_toko, namaToko: account.nama_toko,
alamat: account.alamat alamat: account.alamat
} }
} }
}, { headers: corsHeaders }); }, { headers: corsHeaders });
} }
return errorResponse(401, "Invalid username or password", { headers: corsHeaders }); return errorResponse(401, "Invalid username or password", { headers: corsHeaders });
} catch (error: unknown) { } catch (error: unknown) {
console.error("Error in login API route:", error); console.error("Error in login API route:", error);
const message = error instanceof Error ? error.message : "Internal server error"; const message = error instanceof Error ? error.message : "Internal server error";
return errorResponse(500, message, { headers: corsHeaders }); return errorResponse(500, message, { headers: corsHeaders });
} }
} }
@@ -1,67 +1,67 @@
import { NextRequest, NextResponse } from "next/server"; import { NextRequest, NextResponse } from "next/server";
import { errorResponse } from "@/utils/api-error"; import { errorResponse } from "@/utils/api-error";
import { getAccountFromAuthHeader } from "@/utils/auth"; import { getAccountFromAuthHeader } from "@/utils/auth";
import { query } from "../../../../../db"; import { query } from "../../../../../db";
const corsHeaders = { const corsHeaders = {
"Access-Control-Allow-Origin": "*", "Access-Control-Allow-Origin": "*",
"Access-Control-Allow-Methods": "GET, POST, PUT, DELETE, OPTIONS", "Access-Control-Allow-Methods": "GET, POST, PUT, DELETE, OPTIONS",
"Access-Control-Allow-Headers": "Content-Type, Authorization" "Access-Control-Allow-Headers": "Content-Type, Authorization"
}; };
export async function OPTIONS() { export async function OPTIONS() {
return new NextResponse(null, { status: 204, headers: corsHeaders }); return new NextResponse(null, { status: 204, headers: corsHeaders });
} }
export async function GET(req: NextRequest) { export async function GET(req: NextRequest) {
try { try {
const authHeader = req.headers.get("authorization"); const authHeader = req.headers.get("authorization");
const tokenPayload = getAccountFromAuthHeader(authHeader); const tokenPayload = getAccountFromAuthHeader(authHeader);
if (!tokenPayload) { if (!tokenPayload) {
return errorResponse(401, "Unauthorized", { headers: corsHeaders }); return errorResponse(401, "Unauthorized", { headers: corsHeaders });
} }
const accountRes = await query( const accountRes = await query(
`SELECT a.id, a.username, a.role, a.is_active, `SELECT a.id, a.username, a.role, a.is_active,
s.kode_toko, s.nama_toko, s.alamat s.kode_toko, s.nama_toko, s.alamat
FROM accounts a FROM accounts a
LEFT JOIN store_master s ON a.kode_toko = s.kode_toko LEFT JOIN store_master s ON a.kode_toko = s.kode_toko
WHERE a.id = $1`, WHERE a.id = $1`,
[tokenPayload.accountId] [tokenPayload.accountId]
); );
if (accountRes.rowCount && accountRes.rowCount > 0) { if (accountRes.rowCount && accountRes.rowCount > 0) {
const account = accountRes.rows[0]; const account = accountRes.rows[0];
if (!account.is_active) { if (!account.is_active) {
return errorResponse(401, "Account is disabled", { headers: corsHeaders }); return errorResponse(401, "Account is disabled", { headers: corsHeaders });
} }
// We extract the token exactly as passed in to echo it back in the same shape as login // We extract the token exactly as passed in to echo it back in the same shape as login
const token = authHeader?.slice("Bearer ".length).trim(); const token = authHeader?.slice("Bearer ".length).trim();
return NextResponse.json({ return NextResponse.json({
status: "success", status: "success",
message: "Profile retrieved successfully", message: "Profile retrieved successfully",
data: { data: {
token, token,
profile: { profile: {
username: account.username, username: account.username,
role: account.role, role: account.role,
is_active: account.is_active, is_active: account.is_active,
kodeToko: account.kode_toko, kodeToko: account.kode_toko,
namaToko: account.nama_toko, namaToko: account.nama_toko,
alamat: account.alamat alamat: account.alamat
} }
} }
}, { headers: corsHeaders }); }, { headers: corsHeaders });
} }
return errorResponse(401, "Account not found", { headers: corsHeaders }); return errorResponse(401, "Account not found", { headers: corsHeaders });
} catch (error: unknown) { } catch (error: unknown) {
console.error("Error in auth/me API route:", error); console.error("Error in auth/me API route:", error);
const message = error instanceof Error ? error.message : "Internal server error"; const message = error instanceof Error ? error.message : "Internal server error";
return errorResponse(500, message, { headers: corsHeaders }); return errorResponse(500, message, { headers: corsHeaders });
} }
} }
@@ -1,228 +1,228 @@
import { NextRequest, NextResponse } from "next/server"; import { NextRequest, NextResponse } from "next/server";
import { query, withTransaction } from "../../../../../db"; import { query, withTransaction } from "../../../../../db";
import { errorResponse } from "@/utils/api-error"; import { errorResponse } from "@/utils/api-error";
import { getAccountFromAuthHeader } from "@/utils/auth"; import { getAccountFromAuthHeader } from "@/utils/auth";
import { mapDocumentRow } from "@/utils/document-mapper"; import { mapDocumentRow } from "@/utils/document-mapper";
const corsHeaders = { const corsHeaders = {
"Access-Control-Allow-Origin": "*", "Access-Control-Allow-Origin": "*",
"Access-Control-Allow-Methods": "GET, POST, PUT, DELETE, OPTIONS", "Access-Control-Allow-Methods": "GET, POST, PUT, DELETE, OPTIONS",
"Access-Control-Allow-Headers": "Content-Type, Authorization" "Access-Control-Allow-Headers": "Content-Type, Authorization"
}; };
export async function OPTIONS() { export async function OPTIONS() {
return new NextResponse(null, { status: 204, headers: corsHeaders }); return new NextResponse(null, { status: 204, headers: corsHeaders });
} }
export async function GET( export async function GET(
req: NextRequest, req: NextRequest,
context: { params: Promise<{ id: string }> } context: { params: Promise<{ id: string }> }
) { ) {
try { try {
const account = getAccountFromAuthHeader(req.headers.get("authorization")); const account = getAccountFromAuthHeader(req.headers.get("authorization"));
if (!account) { if (!account) {
return errorResponse(401, "Unauthorized", { headers: corsHeaders }); return errorResponse(401, "Unauthorized", { headers: corsHeaders });
} }
const params = await context.params; const params = await context.params;
const docId = parseInt(params.id); const docId = parseInt(params.id);
if (isNaN(docId)) { if (isNaN(docId)) {
return errorResponse(400, "Invalid document ID", { headers: corsHeaders }); return errorResponse(400, "Invalid document ID", { headers: corsHeaders });
} }
// Deliberately not filtering on `parsed = true` here (unlike the list route) - // Deliberately not filtering on `parsed = true` here (unlike the list route) -
// the whole point of this endpoint is to let the poller see pending/failed // the whole point of this endpoint is to let the poller see pending/failed
// documents, not just done ones. // documents, not just done ones.
const docRes = await query(` const docRes = await query(`
SELECT id, filename, upload_time, parsed, is_sample, metadata, latitude, longitude, kode_toko, scan_mode, parse_error, confirmed SELECT id, filename, upload_time, parsed, is_sample, metadata, latitude, longitude, kode_toko, scan_mode, parse_error, confirmed
FROM documents FROM documents
WHERE id = $1 WHERE id = $1
`, [docId]); `, [docId]);
if (!docRes.rowCount || docRes.rowCount === 0) { if (!docRes.rowCount || docRes.rowCount === 0) {
return errorResponse(404, "Document not found", { headers: corsHeaders }); return errorResponse(404, "Document not found", { headers: corsHeaders });
} }
const doc = docRes.rows[0]; const doc = docRes.rows[0];
if (account.role !== 'admin' && doc.kode_toko !== account.kodeToko) { if (account.role !== 'admin' && doc.kode_toko !== account.kodeToko) {
return errorResponse(403, "Forbidden: You do not have permission to view this document", { headers: corsHeaders }); return errorResponse(403, "Forbidden: You do not have permission to view this document", { headers: corsHeaders });
} }
const itemsRes = await query(` const itemsRes = await query(`
SELECT row_index, kode_barang, nama_barang, banyak, jumlah SELECT row_index, kode_barang, nama_barang, banyak, jumlah
FROM ocr_items FROM ocr_items
WHERE document_id = $1 WHERE document_id = $1
ORDER BY row_index ORDER BY row_index
`, [docId]); `, [docId]);
return NextResponse.json({ return NextResponse.json({
status: "success", status: "success",
data: mapDocumentRow(doc, itemsRes.rows) data: mapDocumentRow(doc, itemsRes.rows)
}, { headers: corsHeaders }); }, { headers: corsHeaders });
} catch (error: unknown) { } catch (error: unknown) {
console.error("Error in get document API v1 route:", error); console.error("Error in get document API v1 route:", error);
const message = error instanceof Error ? error.message : "Internal server error"; const message = error instanceof Error ? error.message : "Internal server error";
return errorResponse(500, message, { headers: corsHeaders }); return errorResponse(500, message, { headers: corsHeaders });
} }
} }
export async function PUT( export async function PUT(
req: NextRequest, req: NextRequest,
context: { params: Promise<{ id: string }> } context: { params: Promise<{ id: string }> }
) { ) {
try { try {
const account = getAccountFromAuthHeader(req.headers.get("authorization")); const account = getAccountFromAuthHeader(req.headers.get("authorization"));
if (!account) { if (!account) {
return errorResponse(401, "Unauthorized", { headers: corsHeaders }); return errorResponse(401, "Unauthorized", { headers: corsHeaders });
} }
const params = await context.params; const params = await context.params;
const { id } = params; const { id } = params;
const docId = parseInt(id); const docId = parseInt(id);
if (isNaN(docId)) { if (isNaN(docId)) {
return errorResponse(400, "Invalid document ID", { headers: corsHeaders }); return errorResponse(400, "Invalid document ID", { headers: corsHeaders });
} }
// Check if document exists // Check if document exists
const checkRes = await query("SELECT id, filename, upload_time, kode_toko FROM documents WHERE id = $1", [docId]); const checkRes = await query("SELECT id, filename, upload_time, kode_toko FROM documents WHERE id = $1", [docId]);
if (!checkRes.rowCount || checkRes.rowCount === 0) { if (!checkRes.rowCount || checkRes.rowCount === 0) {
return errorResponse(404, "Document not found", { headers: corsHeaders }); return errorResponse(404, "Document not found", { headers: corsHeaders });
} }
const doc = checkRes.rows[0]; const doc = checkRes.rows[0];
if (account.role !== 'admin' && doc.kode_toko !== account.kodeToko) { if (account.role !== 'admin' && doc.kode_toko !== account.kodeToko) {
return errorResponse(403, "Forbidden: You do not have permission to modify this document", { headers: corsHeaders }); return errorResponse(403, "Forbidden: You do not have permission to modify this document", { headers: corsHeaders });
} }
const body = await req.json(); const body = await req.json();
const { const {
tanggal, tanggal,
noPo, noPo,
noSo, noSo,
noDo, noDo,
kepadaYth, kepadaYth,
orderUntuk, orderUntuk,
alamat, alamat,
platTruk, platTruk,
namaDriver, namaDriver,
namaPenerima, namaPenerima,
latitude, latitude,
longitude, longitude,
items = [] items = []
} = body; } = body;
// Structuring metadata JSONB to store both formats for full compatibility // Structuring metadata JSONB to store both formats for full compatibility
const metadata = { const metadata = {
// Legacy Next.js web parser format // Legacy Next.js web parser format
tanggal: tanggal || "", tanggal: tanggal || "",
noPO: noPo || "", noPO: noPo || "",
noSO: noSo || "", noSO: noSo || "",
noDO: noDo || doc.filename || "", noDO: noDo || doc.filename || "",
customerInfo: kepadaYth || "", customerInfo: kepadaYth || "",
headerRemark: namaPenerima || "", headerRemark: namaPenerima || "",
// Mobile native app format // Mobile native app format
header: { header: {
tanggal: tanggal || "", tanggal: tanggal || "",
no_po: noPo || "", no_po: noPo || "",
no_so: noSo || "", no_so: noSo || "",
no_do: noDo || "" no_do: noDo || ""
}, },
shipment: { shipment: {
kepada_yth: kepadaYth || "", kepada_yth: kepadaYth || "",
order_untuk: orderUntuk || "", order_untuk: orderUntuk || "",
alamat: alamat || "", alamat: alamat || "",
plat_truk: platTruk || "", plat_truk: platTruk || "",
nama_driver: namaDriver || "", nama_driver: namaDriver || "",
nama_penerima: namaPenerima || "" nama_penerima: namaPenerima || ""
} }
}; };
const latFloat = latitude ? parseFloat(latitude.toString()) : null; const latFloat = latitude ? parseFloat(latitude.toString()) : null;
const lngFloat = longitude ? parseFloat(longitude.toString()) : null; const lngFloat = longitude ? parseFloat(longitude.toString()) : null;
// Update document record. `confirmed = true` is the one and only place // Update document record. `confirmed = true` is the one and only place
// this flips - this PUT is literally "the user tapped Simpan & Konfirmasi" // this flips - this PUT is literally "the user tapped Simpan & Konfirmasi"
// (see docs/api-contract-map.md G11). // (see docs/api-contract-map.md G11).
await query(` await query(`
UPDATE documents UPDATE documents
SET parsed = true, SET parsed = true,
confirmed = true, confirmed = true,
latitude = $2, latitude = $2,
longitude = $3, longitude = $3,
metadata = $4 metadata = $4
WHERE id = $1 WHERE id = $1
`, [docId, latFloat, lngFloat, JSON.stringify(metadata)]); `, [docId, latFloat, lngFloat, JSON.stringify(metadata)]);
// Delete-then-reinsert must be atomic: without a transaction, a failure partway // Delete-then-reinsert must be atomic: without a transaction, a failure partway
// through the insert loop leaves the document with its header already updated // through the insert loop leaves the document with its header already updated
// above but only some (or none) of its items, since the delete has already // above but only some (or none) of its items, since the delete has already
// committed independently. // committed independently.
await withTransaction(async (client) => { await withTransaction(async (client) => {
await client.query("DELETE FROM ocr_items WHERE document_id = $1", [docId]); await client.query("DELETE FROM ocr_items WHERE document_id = $1", [docId]);
for (let i = 0; i < items.length; i++) { for (let i = 0; i < items.length; i++) {
const item = items[i]; const item = items[i];
const nomorSku = item.nomor_sku || item.nomorSku || ""; const nomorSku = item.nomor_sku || item.nomorSku || "";
const namaBarang = item.nama_barang || item.namaBarang || ""; const namaBarang = item.nama_barang || item.namaBarang || "";
const banyak = item.banyak || ""; const banyak = item.banyak || "";
const jumlah = item.jumlah || ""; const jumlah = item.jumlah || "";
await client.query(` await client.query(`
INSERT INTO ocr_items ( INSERT INTO ocr_items (
document_id, row_index, document_id, row_index,
kode_barang_original, kode_barang, kode_barang_original, kode_barang,
nama_barang, nama_barang,
banyak_original, banyak, banyak_original, banyak,
jumlah_original, jumlah, jumlah_original, jumlah,
is_flagged, remark is_flagged, remark
) VALUES ($1, $2, $3, $3, $4, $5, $5, $6, $6, false, '') ) VALUES ($1, $2, $3, $3, $4, $5, $5, $6, $6, false, '')
`, [docId, i, nomorSku, namaBarang, banyak, jumlah]); `, [docId, i, nomorSku, namaBarang, banyak, jumlah]);
} }
}); });
// Return the updated document mapping // Return the updated document mapping
const mappedData = { const mappedData = {
id: docId.toString(), id: docId.toString(),
filePath: doc.filename, filePath: doc.filename,
createdAt: doc.upload_time.toISOString(), createdAt: doc.upload_time.toISOString(),
header: { header: {
tanggal: tanggal || "", tanggal: tanggal || "",
no_po: noPo || "", no_po: noPo || "",
no_so: noSo || "", no_so: noSo || "",
no_do: noDo || "" no_do: noDo || ""
}, },
shipment: { shipment: {
kepada_yth: kepadaYth || "", kepada_yth: kepadaYth || "",
order_untuk: orderUntuk || "", order_untuk: orderUntuk || "",
alamat: alamat || "", alamat: alamat || "",
plat_truk: platTruk || "", plat_truk: platTruk || "",
nama_driver: namaDriver || "", nama_driver: namaDriver || "",
nama_penerima: namaPenerima || "" nama_penerima: namaPenerima || ""
}, },
items: items.map((item: any) => ({ items: items.map((item: any) => ({
nomor_sku: item.nomor_sku || item.nomorSku || "", nomor_sku: item.nomor_sku || item.nomorSku || "",
nama_barang: item.nama_barang || item.namaBarang || "", nama_barang: item.nama_barang || item.namaBarang || "",
banyak: item.banyak || "", banyak: item.banyak || "",
jumlah: item.jumlah || "" jumlah: item.jumlah || ""
})), })),
latitude: latFloat, latitude: latFloat,
longitude: lngFloat longitude: lngFloat
}; };
return NextResponse.json({ return NextResponse.json({
status: "success", status: "success",
message: "Document updated successfully", message: "Document updated successfully",
data: mappedData data: mappedData
}, { headers: corsHeaders }); }, { headers: corsHeaders });
} catch (error: unknown) { } catch (error: unknown) {
console.error("Error in update document API v1 route:", error); console.error("Error in update document API v1 route:", error);
const message = error instanceof Error ? error.message : "Internal server error"; const message = error instanceof Error ? error.message : "Internal server error";
return errorResponse(500, message, { headers: corsHeaders }); return errorResponse(500, message, { headers: corsHeaders });
} }
} }
@@ -1,66 +1,66 @@
import { NextRequest, NextResponse } from "next/server"; import { NextRequest, NextResponse } from "next/server";
import { query } from "../../../../db"; import { query } from "../../../../db";
import { errorResponse } from "@/utils/api-error"; import { errorResponse } from "@/utils/api-error";
import { getAccountFromAuthHeader } from "@/utils/auth"; import { getAccountFromAuthHeader } from "@/utils/auth";
import { mapDocumentRow } from "@/utils/document-mapper"; import { mapDocumentRow } from "@/utils/document-mapper";
const corsHeaders = { const corsHeaders = {
"Access-Control-Allow-Origin": "*", "Access-Control-Allow-Origin": "*",
"Access-Control-Allow-Methods": "GET, POST, PUT, DELETE, OPTIONS", "Access-Control-Allow-Methods": "GET, POST, PUT, DELETE, OPTIONS",
"Access-Control-Allow-Headers": "Content-Type, Authorization" "Access-Control-Allow-Headers": "Content-Type, Authorization"
}; };
export async function OPTIONS() { export async function OPTIONS() {
return new NextResponse(null, { status: 204, headers: corsHeaders }); return new NextResponse(null, { status: 204, headers: corsHeaders });
} }
export async function GET(req: NextRequest) { export async function GET(req: NextRequest) {
try { try {
const account = getAccountFromAuthHeader(req.headers.get("authorization")); const account = getAccountFromAuthHeader(req.headers.get("authorization"));
if (!account) { if (!account) {
return errorResponse(401, "Unauthorized", { headers: corsHeaders }); return errorResponse(401, "Unauthorized", { headers: corsHeaders });
} }
// Retrieve all custom-uploaded documents // Retrieve all custom-uploaded documents
let docsQuery = ` let docsQuery = `
SELECT id, filename, upload_time, size, parsed, is_sample, metadata, latitude, longitude, scan_mode, parse_error, confirmed SELECT id, filename, upload_time, size, parsed, is_sample, metadata, latitude, longitude, scan_mode, parse_error, confirmed
FROM documents FROM documents
WHERE is_sample = false AND parsed = true AND confirmed = true WHERE is_sample = false AND parsed = true AND confirmed = true
`; `;
const queryParams: any[] = []; const queryParams: any[] = [];
if (account.role !== 'admin') { if (account.role !== 'admin') {
docsQuery += ` AND kode_toko = $1`; docsQuery += ` AND kode_toko = $1`;
queryParams.push(account.kodeToko); queryParams.push(account.kodeToko);
} }
docsQuery += ` ORDER BY upload_time DESC`; docsQuery += ` ORDER BY upload_time DESC`;
const docRes = await query(docsQuery, queryParams); const docRes = await query(docsQuery, queryParams);
const documents = docRes.rows; const documents = docRes.rows;
const mappedList = []; const mappedList = [];
for (const doc of documents) { for (const doc of documents) {
// Retrieve items from ocr_items // Retrieve items from ocr_items
const itemsRes = await query(` const itemsRes = await query(`
SELECT row_index, kode_barang, nama_barang, banyak, jumlah SELECT row_index, kode_barang, nama_barang, banyak, jumlah
FROM ocr_items FROM ocr_items
WHERE document_id = $1 WHERE document_id = $1
ORDER BY row_index ORDER BY row_index
`, [doc.id]); `, [doc.id]);
mappedList.push(mapDocumentRow(doc, itemsRes.rows)); mappedList.push(mapDocumentRow(doc, itemsRes.rows));
} }
return NextResponse.json({ return NextResponse.json({
status: "success", status: "success",
data: mappedList data: mappedList
}, { headers: corsHeaders }); }, { headers: corsHeaders });
} catch (error: unknown) { } catch (error: unknown) {
console.error("Error in list documents API v1 route:", error); console.error("Error in list documents API v1 route:", error);
const message = error instanceof Error ? error.message : "Internal server error"; const message = error instanceof Error ? error.message : "Internal server error";
return errorResponse(500, message, { headers: corsHeaders }); return errorResponse(500, message, { headers: corsHeaders });
} }
} }
@@ -1,187 +1,187 @@
import { NextRequest, NextResponse } from "next/server"; import { NextRequest, NextResponse } from "next/server";
import fs from "fs"; import fs from "fs";
import path from "path"; import path from "path";
import crypto from "crypto"; import crypto from "crypto";
import { query } from "../../../../../db"; import { query } from "../../../../../db";
import { errorResponse } from "@/utils/api-error"; import { errorResponse } from "@/utils/api-error";
import { getAccountFromAuthHeader } from "@/utils/auth"; import { getAccountFromAuthHeader } from "@/utils/auth";
import { mapDocumentRow } from "@/utils/document-mapper"; import { mapDocumentRow } from "@/utils/document-mapper";
const UPLOADS_DIR = "/uploads"; const UPLOADS_DIR = "/uploads";
const corsHeaders = { const corsHeaders = {
"Access-Control-Allow-Origin": "*", "Access-Control-Allow-Origin": "*",
"Access-Control-Allow-Methods": "GET, POST, PUT, DELETE, OPTIONS", "Access-Control-Allow-Methods": "GET, POST, PUT, DELETE, OPTIONS",
"Access-Control-Allow-Headers": "Content-Type, Authorization" "Access-Control-Allow-Headers": "Content-Type, Authorization"
}; };
export async function OPTIONS() { export async function OPTIONS() {
return new NextResponse(null, { status: 204, headers: corsHeaders }); return new NextResponse(null, { status: 204, headers: corsHeaders });
} }
export async function POST(req: NextRequest) { export async function POST(req: NextRequest) {
try { try {
// Ensure uploads directory exists // Ensure uploads directory exists
if (!fs.existsSync(UPLOADS_DIR)) { if (!fs.existsSync(UPLOADS_DIR)) {
fs.mkdirSync(UPLOADS_DIR, { recursive: true }); fs.mkdirSync(UPLOADS_DIR, { recursive: true });
} }
// The account uploading is assigned exactly one store (kode_toko) - pass // The account uploading is assigned exactly one store (kode_toko) - pass
// it through to /api/parse so store name/address are set directly from // it through to /api/parse so store name/address are set directly from
// that assignment instead of being OCR-detected from the document photo. // that assignment instead of being OCR-detected from the document photo.
const account = getAccountFromAuthHeader(req.headers.get("authorization")); const account = getAccountFromAuthHeader(req.headers.get("authorization"));
if (!account) { if (!account) {
return errorResponse(401, "Unauthorized", { headers: corsHeaders }); return errorResponse(401, "Unauthorized", { headers: corsHeaders });
} }
const formData = await req.formData(); const formData = await req.formData();
const file = (formData.get("image") || formData.get("file")) as Blob | null; const file = (formData.get("image") || formData.get("file")) as Blob | null;
const scanMode = formData.get("scan_mode")?.toString() || "DO"; const scanMode = formData.get("scan_mode")?.toString() || "DO";
console.log(`[Upload] Received scan_mode: "${scanMode}"`); console.log(`[Upload] Received scan_mode: "${scanMode}"`);
if (!file) { if (!file) {
return errorResponse(400, "No file uploaded", { headers: corsHeaders }); return errorResponse(400, "No file uploaded", { headers: corsHeaders });
} }
const originalName = file instanceof File ? file.name : "document.jpg"; const originalName = file instanceof File ? file.name : "document.jpg";
const safeName = path.basename(originalName).replace(/\s+/g, "_"); const safeName = path.basename(originalName).replace(/\s+/g, "_");
const filename = `${Date.now()}-${safeName}`; const filename = `${Date.now()}-${safeName}`;
const filePath = path.join(UPLOADS_DIR, filename); const filePath = path.join(UPLOADS_DIR, filename);
// Compute hash before writing/inserting anything, so we can detect a duplicate // Compute hash before writing/inserting anything, so we can detect a duplicate
// upload (e.g. the client retrying after a perceived timeout on a slow OCR pass) // upload (e.g. the client retrying after a perceived timeout on a slow OCR pass)
// without creating a second document row or re-running the pipeline on it. // without creating a second document row or re-running the pipeline on it.
const arrayBuffer = await file.arrayBuffer(); const arrayBuffer = await file.arrayBuffer();
const buffer = Buffer.from(arrayBuffer); const buffer = Buffer.from(arrayBuffer);
const fileHash = crypto.createHash("sha256").update(buffer).digest("hex"); const fileHash = crypto.createHash("sha256").update(buffer).digest("hex");
// Geolocation tags // Geolocation tags
const latVal = formData.get("latitude"); const latVal = formData.get("latitude");
const lngVal = formData.get("longitude"); const lngVal = formData.get("longitude");
const latitude = latVal ? parseFloat(latVal.toString()) : null; const latitude = latVal ? parseFloat(latVal.toString()) : null;
const longitude = lngVal ? parseFloat(lngVal.toString()) : null; const longitude = lngVal ? parseFloat(lngVal.toString()) : null;
// Basic dedup // Basic dedup
const dedupQuery = account?.kodeToko const dedupQuery = account?.kodeToko
? "SELECT id, filename, upload_time, parsed, metadata, latitude, longitude, scan_mode, parse_error, confirmed FROM documents WHERE file_hash = $1 AND kode_toko = $2 ORDER BY upload_time ASC LIMIT 1" ? "SELECT id, filename, upload_time, parsed, metadata, latitude, longitude, scan_mode, parse_error, confirmed FROM documents WHERE file_hash = $1 AND kode_toko = $2 ORDER BY upload_time ASC LIMIT 1"
: "SELECT id, filename, upload_time, parsed, metadata, latitude, longitude, scan_mode, parse_error, confirmed FROM documents WHERE file_hash = $1 AND kode_toko IS NULL ORDER BY upload_time ASC LIMIT 1"; : "SELECT id, filename, upload_time, parsed, metadata, latitude, longitude, scan_mode, parse_error, confirmed FROM documents WHERE file_hash = $1 AND kode_toko IS NULL ORDER BY upload_time ASC LIMIT 1";
const dedupParams = account?.kodeToko ? [fileHash, account.kodeToko] : [fileHash]; const dedupParams = account?.kodeToko ? [fileHash, account.kodeToko] : [fileHash];
const existing = await query(dedupQuery, dedupParams); const existing = await query(dedupQuery, dedupParams);
if (existing.rows.length > 0) { if (existing.rows.length > 0) {
const existingDoc = existing.rows[0]; const existingDoc = existing.rows[0];
console.log(`[Dedup] Identical content already uploaded as document ${existingDoc.id}. Skipping duplicate insert and re-parse.`); console.log(`[Dedup] Identical content already uploaded as document ${existingDoc.id}. Skipping duplicate insert and re-parse.`);
// Return the original document's actual current parse state instead of an // Return the original document's actual current parse state instead of an
// always-empty stub, so a retried upload doesn't look permanently "fresh." // always-empty stub, so a retried upload doesn't look permanently "fresh."
const itemsRes = await query(` const itemsRes = await query(`
SELECT row_index, kode_barang, nama_barang, banyak, jumlah SELECT row_index, kode_barang, nama_barang, banyak, jumlah
FROM ocr_items FROM ocr_items
WHERE document_id = $1 WHERE document_id = $1
ORDER BY row_index ORDER BY row_index
`, [existingDoc.id]); `, [existingDoc.id]);
const mappedData = mapDocumentRow(existingDoc, itemsRes.rows); const mappedData = mapDocumentRow(existingDoc, itemsRes.rows);
// Fall back to this retry's own GPS tag if the original document never got one. // Fall back to this retry's own GPS tag if the original document never got one.
if (mappedData.latitude === null) mappedData.latitude = latitude; if (mappedData.latitude === null) mappedData.latitude = latitude;
if (mappedData.longitude === null) mappedData.longitude = longitude; if (mappedData.longitude === null) mappedData.longitude = longitude;
return NextResponse.json({ return NextResponse.json({
status: "success", status: "success",
message: "Document already uploaded", message: "Document already uploaded",
data: mappedData data: mappedData
}, { status: 201, headers: corsHeaders }); }, { status: 201, headers: corsHeaders });
} }
// Save file // Save file
fs.writeFileSync(filePath, buffer); fs.writeFileSync(filePath, buffer);
let docId: number; let docId: number;
let finalFilename = filename; let finalFilename = filename;
// `confirmed = false`: this row isn't visible via GET /api/v1/documents // `confirmed = false`: this row isn't visible via GET /api/v1/documents
// until the user's editor PUT confirms it (see docs/api-contract-map.md G11). // until the user's editor PUT confirms it (see docs/api-contract-map.md G11).
const insertRes = await query(` const insertRes = await query(`
INSERT INTO documents (filename, upload_time, size, parsed, is_sample, file_hash, latitude, longitude, kode_toko, scan_mode, confirmed) INSERT INTO documents (filename, upload_time, size, parsed, is_sample, file_hash, latitude, longitude, kode_toko, scan_mode, confirmed)
VALUES ($1, $2, $3, $4, $5, $6, $7, $8, $9, $10, $11) VALUES ($1, $2, $3, $4, $5, $6, $7, $8, $9, $10, $11)
RETURNING id RETURNING id
`, [ `, [
filename, filename,
new Date(), new Date(),
buffer.length, buffer.length,
false, false,
false, false,
fileHash, fileHash,
latitude, latitude,
longitude, longitude,
account?.kodeToko || null, account?.kodeToko || null,
scanMode, scanMode,
false false
]); ]);
docId = insertRes.rows[0].id; docId = insertRes.rows[0].id;
// Trigger parsing synchronously to ensure it is processed immediately on receiving the image. // Trigger parsing synchronously to ensure it is processed immediately on receiving the image.
// Bounded well above /api/parse's own per-pass pipeline timeout (2 passes worst case) so a // Bounded well above /api/parse's own per-pass pipeline timeout (2 passes worst case) so a
// wedged GPU container doesn't hang this request forever - it still won't fit under the // wedged GPU container doesn't hang this request forever - it still won't fit under the
// mobile client's 2-minute receive timeout in the worst case, but bounds the hang to a fixed, // mobile client's 2-minute receive timeout in the worst case, but bounds the hang to a fixed,
// known ceiling instead of an indefinite one. // known ceiling instead of an indefinite one.
// //
// /api/parse has its own error handlers that mark the document parsed=true with // /api/parse has its own error handlers that mark the document parsed=true with
// "Not Found" placeholder metadata on a pipeline failure - so those cases already // "Not Found" placeholder metadata on a pipeline failure - so those cases already
// resolve out of "pending". The one gap is this call itself never completing // resolve out of "pending". The one gap is this call itself never completing
// (network error / the 210s abort firing): /api/parse's handlers never even run, // (network error / the 210s abort firing): /api/parse's handlers never even run,
// so the document is otherwise silently stuck at parsed=false forever. Record // so the document is otherwise silently stuck at parsed=false forever. Record
// that case explicitly so GET /api/v1/documents/:id can report parseStatus "failed" // that case explicitly so GET /api/v1/documents/:id can report parseStatus "failed"
// instead of the client burning its own full timeout waiting on "pending". // instead of the client burning its own full timeout waiting on "pending".
try { try {
const parseRes = await fetch("http://127.0.0.1:3000/api/parse", { const parseRes = await fetch("http://127.0.0.1:3000/api/parse", {
method: "POST", method: "POST",
headers: { "Content-Type": "application/json" }, headers: { "Content-Type": "application/json" },
body: JSON.stringify({ filename: finalFilename, kodeToko: account?.kodeToko, scanMode }), body: JSON.stringify({ filename: finalFilename, kodeToko: account?.kodeToko, scanMode }),
signal: AbortSignal.timeout(210_000) signal: AbortSignal.timeout(210_000)
}); });
if (!parseRes.ok) { if (!parseRes.ok) {
await query("UPDATE documents SET parse_error = $1 WHERE id = $2", [`Pipeline error: HTTP ${parseRes.status}`, docId]); await query("UPDATE documents SET parse_error = $1 WHERE id = $2", [`Pipeline error: HTTP ${parseRes.status}`, docId]);
} }
} catch (err) { } catch (err) {
console.error("Error triggering parse synchronously:", err); console.error("Error triggering parse synchronously:", err);
const message = err instanceof Error ? err.message : "Parse request failed"; const message = err instanceof Error ? err.message : "Parse request failed";
await query("UPDATE documents SET parse_error = $1 WHERE id = $2", [message, docId]); await query("UPDATE documents SET parse_error = $1 WHERE id = $2", [message, docId]);
} }
// Return the response structured as DocumentModel.fromJson format // Return the response structured as DocumentModel.fromJson format
const mappedData = { const mappedData = {
id: docId.toString(), id: docId.toString(),
header: { header: {
tanggal: "", tanggal: "",
no_po: "", no_po: "",
no_so: "", no_so: "",
no_do: "" no_do: ""
}, },
shipment: { shipment: {
kepada_yth: "PT.PRIMAFOOD INTERNATIONAL", kepada_yth: "PT.PRIMAFOOD INTERNATIONAL",
order_untuk: "", order_untuk: "",
alamat: "", alamat: "",
plat_truk: "", plat_truk: "",
nama_driver: "", nama_driver: "",
nama_penerima: "" nama_penerima: ""
}, },
items: [] as any[], items: [] as any[],
latitude: latitude, latitude: latitude,
longitude: longitude, longitude: longitude,
createdAt: new Date().toISOString() createdAt: new Date().toISOString()
}; };
return NextResponse.json({ return NextResponse.json({
status: "success", status: "success",
message: "Document uploaded successfully", message: "Document uploaded successfully",
data: mappedData data: mappedData
}, { status: 201, headers: corsHeaders }); }, { status: 201, headers: corsHeaders });
} catch (error: unknown) { } catch (error: unknown) {
console.error("Error in upload API v1 route:", error); console.error("Error in upload API v1 route:", error);
const message = error instanceof Error ? error.message : "Internal server error"; const message = error instanceof Error ? error.message : "Internal server error";
return errorResponse(500, message, { headers: corsHeaders }); return errorResponse(500, message, { headers: corsHeaders });
} }
} }
@@ -1,56 +1,56 @@
import { NextResponse } from "next/server"; import { NextResponse } from "next/server";
import { query } from "../../../../db"; import { query } from "../../../../db";
const corsHeaders = { const corsHeaders = {
"Access-Control-Allow-Origin": "*", "Access-Control-Allow-Origin": "*",
"Access-Control-Allow-Methods": "GET, OPTIONS", "Access-Control-Allow-Methods": "GET, OPTIONS",
"Access-Control-Allow-Headers": "Content-Type, Authorization" "Access-Control-Allow-Headers": "Content-Type, Authorization"
}; };
export async function OPTIONS() { export async function OPTIONS() {
return new NextResponse(null, { status: 204, headers: corsHeaders }); return new NextResponse(null, { status: 204, headers: corsHeaders });
} }
export async function GET() { export async function GET() {
let dbHealthy = false; let dbHealthy = false;
let pipelineHealthy = false; let pipelineHealthy = false;
// Check Database // Check Database
try { try {
const res = await query("SELECT 1 as healthy"); const res = await query("SELECT 1 as healthy");
if (res.rowCount && res.rows[0].healthy === 1) { if (res.rowCount && res.rows[0].healthy === 1) {
dbHealthy = true; dbHealthy = true;
} }
} catch (err) { } catch (err) {
console.error("Health check - DB ping failed:", err); console.error("Health check - DB ping failed:", err);
} }
// Check Pipeline API // Check Pipeline API
try { try {
const pipelineUrl = process.env.PIPELINE_URL; const pipelineUrl = process.env.PIPELINE_URL;
// e.g. http://paddleocr-pipeline-api:8090/layout-parsing // e.g. http://paddleocr-pipeline-api:8090/layout-parsing
if (pipelineUrl) { if (pipelineUrl) {
const healthUrl = new URL("/", pipelineUrl).toString(); const healthUrl = new URL("/", pipelineUrl).toString();
const response = await fetch(healthUrl, { method: "GET", signal: AbortSignal.timeout(3000) }); const response = await fetch(healthUrl, { method: "GET", signal: AbortSignal.timeout(3000) });
// As long as the server responds (even with 404 or 405), it is running. // As long as the server responds (even with 404 or 405), it is running.
if (response.status) { if (response.status) {
pipelineHealthy = true; pipelineHealthy = true;
} }
} else { } else {
console.warn("Health check - PIPELINE_URL not configured in environment"); console.warn("Health check - PIPELINE_URL not configured in environment");
} }
} catch (err) { } catch (err) {
console.error("Health check - Pipeline ping failed:", err); console.error("Health check - Pipeline ping failed:", err);
} }
const isHealthy = dbHealthy && pipelineHealthy; const isHealthy = dbHealthy && pipelineHealthy;
return NextResponse.json({ return NextResponse.json({
status: isHealthy ? "ok" : "error", status: isHealthy ? "ok" : "error",
db: dbHealthy, db: dbHealthy,
pipeline: pipelineHealthy, pipeline: pipelineHealthy,
}, { }, {
status: isHealthy ? 200 : 503, status: isHealthy ? 200 : 503,
headers: corsHeaders headers: corsHeaders
}); });
} }
@@ -1,71 +1,71 @@
import { NextRequest, NextResponse } from "next/server"; import { NextRequest, NextResponse } from "next/server";
import { query } from "@/db"; import { query } from "@/db";
import { errorResponse } from "@/utils/api-error"; import { errorResponse } from "@/utils/api-error";
import { getAccountFromAuthHeader } from "@/utils/auth"; import { getAccountFromAuthHeader } from "@/utils/auth";
export async function PUT( export async function PUT(
req: NextRequest, req: NextRequest,
context: { params: Promise<{ kode: string }> } context: { params: Promise<{ kode: string }> }
) { ) {
try { try {
const account = getAccountFromAuthHeader(req.headers.get("authorization")); const account = getAccountFromAuthHeader(req.headers.get("authorization"));
if (!account || account.role !== 'admin') { if (!account || account.role !== 'admin') {
return errorResponse(403, "Forbidden: Admin access required"); return errorResponse(403, "Forbidden: Admin access required");
} }
const { kode } = await context.params; const { kode } = await context.params;
const body = await req.json(); const body = await req.json();
const { const {
nama_item, nama_item,
jenis_outer, jenis_outer,
standar_jumlah standar_jumlah
} = body; } = body;
const res = await query( const res = await query(
`UPDATE sku_master `UPDATE sku_master
SET nama_item = $1, jenis_outer = $2, standar_jumlah = $3 SET nama_item = $1, jenis_outer = $2, standar_jumlah = $3
WHERE no_sku = $4 RETURNING *`, WHERE no_sku = $4 RETURNING *`,
[ [
nama_item, nama_item,
jenis_outer || '', jenis_outer || '',
String(standar_jumlah || '1'), String(standar_jumlah || '1'),
kode kode
] ]
); );
if (res.rowCount === 0) { if (res.rowCount === 0) {
return errorResponse(404, "SKU not found"); return errorResponse(404, "SKU not found");
} }
return NextResponse.json({ status: "success", data: res.rows[0] }); return NextResponse.json({ status: "success", data: res.rows[0] });
} catch (err: any) { } catch (err: any) {
return errorResponse(500, err.message); return errorResponse(500, err.message);
} }
} }
export async function DELETE( export async function DELETE(
req: NextRequest, req: NextRequest,
context: { params: Promise<{ kode: string }> } context: { params: Promise<{ kode: string }> }
) { ) {
try { try {
const account = getAccountFromAuthHeader(req.headers.get("authorization")); const account = getAccountFromAuthHeader(req.headers.get("authorization"));
if (!account || account.role !== 'admin') { if (!account || account.role !== 'admin') {
return errorResponse(403, "Forbidden: Admin access required"); return errorResponse(403, "Forbidden: Admin access required");
} }
const { kode } = await context.params; const { kode } = await context.params;
const res = await query("DELETE FROM sku_master WHERE no_sku = $1 RETURNING *", [kode]); const res = await query("DELETE FROM sku_master WHERE no_sku = $1 RETURNING *", [kode]);
if (res.rowCount === 0) { if (res.rowCount === 0) {
return errorResponse(404, "SKU not found"); return errorResponse(404, "SKU not found");
} }
return NextResponse.json({ status: "success", message: "SKU deleted successfully" }); return NextResponse.json({ status: "success", message: "SKU deleted successfully" });
} catch (err: any) { } catch (err: any) {
if (err.code === '23503') { // foreign key violation if (err.code === '23503') { // foreign key violation
return errorResponse(409, "Cannot delete SKU because it is referenced in documents"); return errorResponse(409, "Cannot delete SKU because it is referenced in documents");
} }
return errorResponse(500, err.message); return errorResponse(500, err.message);
} }
} }
@@ -1,68 +1,68 @@
import { NextRequest, NextResponse } from "next/server"; import { NextRequest, NextResponse } from "next/server";
import { query } from "@/db"; import { query } from "@/db";
import { errorResponse } from "@/utils/api-error"; import { errorResponse } from "@/utils/api-error";
import { getAccountFromAuthHeader } from "@/utils/auth"; import { getAccountFromAuthHeader } from "@/utils/auth";
export async function GET(req: NextRequest) { export async function GET(req: NextRequest) {
try { try {
// Read access is open to any authenticated account (task 9.2) - the // Read access is open to any authenticated account (task 9.2) - the
// Flutter product editor needs this to populate its SKU dropdown, and // Flutter product editor needs this to populate its SKU dropdown, and
// has no admin role of its own. Writes below stay admin-gated. // has no admin role of its own. Writes below stay admin-gated.
const account = getAccountFromAuthHeader(req.headers.get("authorization")); const account = getAccountFromAuthHeader(req.headers.get("authorization"));
if (!account) { if (!account) {
return errorResponse(401, "Unauthorized"); return errorResponse(401, "Unauthorized");
} }
const res = await query(` const res = await query(`
SELECT no_sku, nama_item, standar_jumlah, berat_kemasan, isi_outer_kg, isi_outer_pac, jenis_outer SELECT no_sku, nama_item, standar_jumlah, berat_kemasan, isi_outer_kg, isi_outer_pac, jenis_outer
FROM sku_master FROM sku_master
ORDER BY no_sku ASC ORDER BY no_sku ASC
`); `);
return NextResponse.json({ status: "success", data: res.rows }); return NextResponse.json({ status: "success", data: res.rows });
} catch (err: any) { } catch (err: any) {
return errorResponse(500, err.message); return errorResponse(500, err.message);
} }
} }
export async function POST(req: NextRequest) { export async function POST(req: NextRequest) {
try { try {
const account = getAccountFromAuthHeader(req.headers.get("authorization")); const account = getAccountFromAuthHeader(req.headers.get("authorization"));
if (!account || account.role !== 'admin') { if (!account || account.role !== 'admin') {
return errorResponse(403, "Forbidden: Admin access required"); return errorResponse(403, "Forbidden: Admin access required");
} }
const body = await req.json(); const body = await req.json();
const { const {
kode_item, kode_item,
no_sku, no_sku,
nama_item, nama_item,
jenis_outer, jenis_outer,
standar_jumlah standar_jumlah
} = body; } = body;
const skuCode = no_sku || kode_item; const skuCode = no_sku || kode_item;
if (!skuCode || !nama_item) { if (!skuCode || !nama_item) {
return errorResponse(400, "no_sku and nama_item are required"); return errorResponse(400, "no_sku and nama_item are required");
} }
await query( await query(
`INSERT INTO sku_master `INSERT INTO sku_master
(no_sku, nama_item, jenis_outer, standar_jumlah) (no_sku, nama_item, jenis_outer, standar_jumlah)
VALUES ($1, $2, $3, $4)`, VALUES ($1, $2, $3, $4)`,
[ [
skuCode, skuCode,
nama_item, nama_item,
jenis_outer || '', jenis_outer || '',
String(standar_jumlah || '1') String(standar_jumlah || '1')
] ]
); );
return NextResponse.json({ status: "success", message: "SKU created successfully" }); return NextResponse.json({ status: "success", message: "SKU created successfully" });
} catch (err: any) { } catch (err: any) {
if (err.code === '23505') { // unique violation if (err.code === '23505') { // unique violation
return errorResponse(409, "SKU with this kode_item already exists"); return errorResponse(409, "SKU with this kode_item already exists");
} }
return errorResponse(500, err.message); return errorResponse(500, err.message);
} }
} }
@@ -1,68 +1,68 @@
import { NextRequest, NextResponse } from "next/server"; import { NextRequest, NextResponse } from "next/server";
import { query, withTransaction } from "@/db"; import { query, withTransaction } from "@/db";
import { errorResponse } from "@/utils/api-error"; import { errorResponse } from "@/utils/api-error";
import { getAccountFromAuthHeader } from "@/utils/auth"; import { getAccountFromAuthHeader } from "@/utils/auth";
export async function PUT( export async function PUT(
req: NextRequest, req: NextRequest,
context: { params: Promise<{ kode: string }> } context: { params: Promise<{ kode: string }> }
) { ) {
try { try {
const account = getAccountFromAuthHeader(req.headers.get("authorization")); const account = getAccountFromAuthHeader(req.headers.get("authorization"));
if (!account || account.role !== 'admin') { if (!account || account.role !== 'admin') {
return errorResponse(403, "Forbidden: Admin access required"); return errorResponse(403, "Forbidden: Admin access required");
} }
const { kode } = await context.params; const { kode } = await context.params;
const body = await req.json(); const body = await req.json();
const { nama_toko, alamat } = body; const { nama_toko, alamat } = body;
const res = await query( const res = await query(
"UPDATE store_master SET nama_toko = $1, alamat = $2 WHERE kode_toko = $3 RETURNING *", "UPDATE store_master SET nama_toko = $1, alamat = $2 WHERE kode_toko = $3 RETURNING *",
[nama_toko, alamat || '', kode] [nama_toko, alamat || '', kode]
); );
if (res.rowCount === 0) { if (res.rowCount === 0) {
return errorResponse(404, "Store not found"); return errorResponse(404, "Store not found");
} }
return NextResponse.json({ status: "success", data: res.rows[0] }); return NextResponse.json({ status: "success", data: res.rows[0] });
} catch (err: any) { } catch (err: any) {
return errorResponse(500, err.message); return errorResponse(500, err.message);
} }
} }
export async function DELETE( export async function DELETE(
req: NextRequest, req: NextRequest,
context: { params: Promise<{ kode: string }> } context: { params: Promise<{ kode: string }> }
) { ) {
try { try {
const account = getAccountFromAuthHeader(req.headers.get("authorization")); const account = getAccountFromAuthHeader(req.headers.get("authorization"));
if (!account || account.role !== 'admin') { if (!account || account.role !== 'admin') {
return errorResponse(403, "Forbidden: Admin access required"); return errorResponse(403, "Forbidden: Admin access required");
} }
const { kode } = await context.params; const { kode } = await context.params;
await withTransaction(async (client) => { await withTransaction(async (client) => {
// Delete associated account first due to FK account -> store_master // Delete associated account first due to FK account -> store_master
await client.query("DELETE FROM accounts WHERE kode_toko = $1", [kode]); await client.query("DELETE FROM accounts WHERE kode_toko = $1", [kode]);
const res = await client.query("DELETE FROM store_master WHERE kode_toko = $1 RETURNING *", [kode]); const res = await client.query("DELETE FROM store_master WHERE kode_toko = $1 RETURNING *", [kode]);
if (res.rowCount === 0) { if (res.rowCount === 0) {
throw new Error("Store not found"); throw new Error("Store not found");
} }
}); });
return NextResponse.json({ status: "success", message: "Store and associated account deleted successfully" }); return NextResponse.json({ status: "success", message: "Store and associated account deleted successfully" });
} catch (err: any) { } catch (err: any) {
if (err.code === '23503') { // foreign key violation (e.g. documents exist) if (err.code === '23503') { // foreign key violation (e.g. documents exist)
return errorResponse(409, "Cannot delete store because it has associated documents"); return errorResponse(409, "Cannot delete store because it has associated documents");
} }
if (err.message === "Store not found") { if (err.message === "Store not found") {
return errorResponse(404, err.message); return errorResponse(404, err.message);
} }
return errorResponse(500, err.message); return errorResponse(500, err.message);
} }
} }
Loaded 100 of 315 files, more files were not shown because too many files have changed in this diff. Show more