feat: consolidate backend and docker-compose setup

This commit is contained in:
Rafhan Mazaya Fathurrahman committed 2026-06-30 21:09:08 +07:00
commit ff3753a745
306 files changed
+35450

No files matched your search

@@ -0,0 +1,44 @@
import os
from PIL import Image
def generate_thumbnails():
src_base = os.path.join("pfm-web-app", "public", "produk-pfm", "foto-kemasan-v2")
dest_base = os.path.join("pfm-web-app", "public", "produk-pfm", "thumbs")
if not os.path.exists(src_base):
print(f"Source directory {src_base} does not exist.")
return
# Supported image extensions
valid_exts = (".jpg", ".jpeg", ".png", ".webp", ".bmp")
count = 0
# Walk through the source directory
for root, dirs, files in os.walk(src_base):
for file in files:
if file.lower().endswith(valid_exts):
src_path = os.path.join(root, file)
# Get the relative path from the source base
rel_path = os.path.relpath(src_path, src_base)
dest_path = os.path.join(dest_base, rel_path)
# Ensure destination directory exists
dest_dir = os.path.dirname(dest_path)
os.makedirs(dest_dir, exist_ok=True)
try:
with Image.open(src_path) as img:
# Resize to maximum dimension of 120px while preserving aspect ratio
img.thumbnail((120, 120))
# Save the thumbnail
img.save(dest_path)
print(f"Generated thumbnail: {dest_path}")
count += 1
except Exception as e:
print(f"Failed to generate thumbnail for {src_path}: {e}")
print(f"Successfully generated {count} thumbnails.")
if __name__ == "__main__":
generate_thumbnails()
+27
View File
@@ -0,0 +1,27 @@
#!/usr/bin/env bash
set -euo pipefail
ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)"
cd "$ROOT"
unset VIRTUAL_ENV
PYTHON="${PYTHON:-3.12}"
VENV=".venv-api"
echo "==> Creating PaddleOCR-VL pipeline API env (${VENV}) with Python ${PYTHON}"
uv venv "${VENV}" --python "${PYTHON}"
echo "==> Installing PaddlePaddle GPU (CUDA 12.6)"
uv pip install --python "${VENV}" \
paddlepaddle-gpu \
-i https://www.paddlepaddle.org.cn/packages/stable/cu126/
echo "==> Installing PaddleOCR doc-parser stack"
uv pip install --python "${VENV}" "paddleocr[doc-parser]>=3.3.0"
echo "==> Installing PaddleX serving dependencies"
uv pip install --python "${VENV}" \
"aiohttp>=3.9" "filetype>=1.2" "fastapi>=0.110" "starlette>=0.36" "uvicorn>=0.16"
echo "==> Done. Start pipeline API with: ./scripts/serve-pipeline.sh"
+28
View File
@@ -0,0 +1,28 @@
#!/usr/bin/env bash
set -euo pipefail
ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)"
cd "$ROOT"
# Avoid picking up another project's active venv
unset VIRTUAL_ENV
PYTHON="${PYTHON:-3.12}"
FLASH_ATTN_WHEEL="${FLASH_ATTN_WHEEL:-}"
echo "==> Using uv with Python ${PYTHON}"
uv python pin "${PYTHON}"
uv sync
if [[ -n "${FLASH_ATTN_WHEEL}" ]]; then
echo "==> Installing prebuilt flash-attn wheel"
uv pip install --python .venv "${FLASH_ATTN_WHEEL}"
else
echo "==> Installing flash-attn (may require a prebuilt wheel; set FLASH_ATTN_WHEEL)"
uv pip install --python .venv "flash-attn==2.8.2"
fi
echo "==> Verifying paddleocr genai_server entrypoint"
uv run paddleocr genai_server --help >/dev/null
echo "==> Done. Start the service with: ./scripts/serve.sh"
+44
View File
@@ -0,0 +1,44 @@
#!/usr/bin/env bash
set -euo pipefail
ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)"
DEMO="${ROOT}/PaddleOCR-VL-1.6_Online_Demo"
cd "$DEMO"
unset VIRTUAL_ENV
if [[ -f "${ROOT}/.env" ]]; then
set -a
# shellcheck disable=SC1091
source "${ROOT}/.env"
set +a
fi
PIPELINE_PORT="${PIPELINE_PORT:-8090}"
GRADIO_PORT="${GRADIO_PORT:-7870}"
export API_URL="${API_URL:-http://127.0.0.1:${PIPELINE_PORT}/layout-parsing}"
echo "==> Starting PaddleOCR-VL-1.6 Online Demo"
echo " API_URL: ${API_URL}"
echo " Gradio: http://127.0.0.1:${GRADIO_PORT}"
if command -v ss >/dev/null 2>&1 && ss -tln | grep -q ":${GRADIO_PORT} "; then
echo "ERROR: Port ${GRADIO_PORT} is already in use." >&2
echo " The demo may already be running — open http://127.0.0.1:${GRADIO_PORT}" >&2
echo " To restart: stop the old process (e.g. fuser -k ${GRADIO_PORT}/tcp) then run this script again." >&2
echo " Or use another port: GRADIO_PORT=7871 ./scripts/run-demo.sh" >&2
exit 1
fi
if [[ ! -d .venv ]]; then
echo "==> Creating demo venv and installing requirements"
uv venv .venv --python 3.12
uv pip install --python .venv -r requirements.txt
fi
export GRADIO_SERVER_PORT="${GRADIO_PORT}"
export GRADIO_SERVER_NAME="127.0.0.1"
export GRADIO_MCP_SERVER="True"
exec uv run --no-project --python .venv/bin/python python app.py
+61
View File
@@ -0,0 +1,61 @@
#!/usr/bin/env bash
set -euo pipefail
ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)"
cd "$ROOT"
unset VIRTUAL_ENV
if [[ -f .env ]]; then
set -a
# shellcheck disable=SC1091
source .env
set +a
fi
PIPELINE_CONFIG="${PIPELINE_CONFIG:-config/pipeline_config_vllm.yaml}"
PIPELINE_HOST="${PIPELINE_HOST:-0.0.0.0}"
PIPELINE_PORT="${PIPELINE_PORT:-8090}"
PIPELINE_DEVICE="${PIPELINE_DEVICE:-gpu:0}"
VENV=".venv-api"
# Check if VLLM_SERVER_URL is provided to override the url in the config
if [[ -n "${VLLM_SERVER_URL:-}" ]]; then
echo "==> Overriding VLLM server URL with ${VLLM_SERVER_URL}"
TEMP_CONFIG=$(mktemp /tmp/pipeline_config_XXXXXX.yaml)
cp "${PIPELINE_CONFIG}" "${TEMP_CONFIG}"
sed -i "s|server_url:.*|server_url: ${VLLM_SERVER_URL}|g" "${TEMP_CONFIG}"
PIPELINE_CONFIG="${TEMP_CONFIG}"
trap 'rm -f "${TEMP_CONFIG}"' EXIT
fi
if [[ ! -d "${VENV}" ]]; then
echo "Missing ${VENV}. Run ./scripts/install-pipeline.sh first." >&2
exit 1
fi
if [[ ! -f "${PIPELINE_CONFIG}" ]]; then
echo "Missing pipeline config: ${PIPELINE_CONFIG}" >&2
exit 1
fi
echo "==> Starting PaddleOCR-VL pipeline API"
echo " config: ${PIPELINE_CONFIG}"
echo " device: ${PIPELINE_DEVICE}"
echo " url: http://${PIPELINE_HOST}:${PIPELINE_PORT}/layout-parsing"
echo " vllm: ${VLLM_SERVER_URL:-http://127.0.0.1:8118/v1}"
echo "==> Starting secondary PFM classifier & OCR server"
"${ROOT}/${VENV}/bin/python" "${ROOT}/config/classify_ocr_server.py" &
# Patch default max_new_tokens in paddlex to avoid context length overflow in vllm
PIPELINE_PY="${ROOT}/${VENV}/lib/python3.12/site-packages/paddlex/inference/pipelines/paddleocr_vl/pipeline.py"
if [[ -f "${PIPELINE_PY}" ]]; then
echo "==> Patching default max_new_tokens in paddlex to 1024"
sed -i 's/vlm_kwargs\["max_new_tokens"\] = 4096/vlm_kwargs\["max_new_tokens"\] = 1024/g' "${PIPELINE_PY}"
fi
exec "${ROOT}/${VENV}/bin/paddlex" --serve \
--pipeline "${PIPELINE_CONFIG}" \
--host "${PIPELINE_HOST}" \
--port "${PIPELINE_PORT}" \
--device "${PIPELINE_DEVICE}"
+38
View File
@@ -0,0 +1,38 @@
#!/usr/bin/env bash
set -euo pipefail
ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)"
cd "$ROOT"
unset VIRTUAL_ENV
if [[ -f .env ]]; then
set -a
# shellcheck disable=SC1091
source .env
set +a
fi
GENAI_HOST="${GENAI_HOST:-0.0.0.0}"
GENAI_PORT="${GENAI_PORT:-8118}"
GENAI_MODEL="${GENAI_MODEL:-PaddleOCR-VL-1.6-0.9B}"
GENAI_BACKEND="${GENAI_BACKEND:-vllm}"
VLLM_CONFIG="${VLLM_CONFIG:-config/vllm_config.yaml}"
if [[ ! -f "${VLLM_CONFIG}" ]]; then
echo "Missing vLLM config: ${VLLM_CONFIG}" >&2
exit 1
fi
echo "==> Starting PaddleOCR-VL vLLM service"
echo " model: ${GENAI_MODEL}"
echo " backend: ${GENAI_BACKEND}"
echo " url: http://${GENAI_HOST}:${GENAI_PORT}/v1"
echo " config: ${VLLM_CONFIG}"
exec uv run paddleocr genai_server \
--model_name "${GENAI_MODEL}" \
--host "${GENAI_HOST}" \
--port "${GENAI_PORT}" \
--backend "${GENAI_BACKEND}" \
--backend_config "${VLLM_CONFIG}"