fix and adjust main dashboard ai insight
This commit is contained in:
1 parent
18790ab8f6
commit
68e0d1a0bb
27 files changed
+360
-107
No files matched your search
@@ -0,0 +1,24 @@
|
||||
FROM python:3.9-slim
|
||||
|
||||
WORKDIR /app
|
||||
|
||||
|
||||
# Copy dependencies
|
||||
COPY requirements.txt .
|
||||
RUN pip install --no-cache-dir -r requirements.txt
|
||||
|
||||
# Pre-download the Hugging Face embedding model during Docker build phase
|
||||
# so it is baked into the image and does not need internet to start
|
||||
RUN python -c "from sentence_transformers import SentenceTransformer; SentenceTransformer('sentence-transformers/paraphrase-multilingual-MiniLM-L12-v2')"
|
||||
|
||||
# Copy source code and vector DB
|
||||
COPY . .
|
||||
|
||||
# Set offline environment variables
|
||||
ENV HF_HUB_OFFLINE=1
|
||||
ENV TRANSFORMERS_OFFLINE=1
|
||||
ENV RAG_PORT=5002
|
||||
|
||||
EXPOSE 5002
|
||||
|
||||
CMD ["python", "rag_service.py"]
|
||||
@@ -0,0 +1,34 @@
|
||||
"""Classify CP 707 extracted chunks as prosa vs tabel for RAG metadata."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
|
||||
_NUMERIC_PIPE_ROW = re.compile(r"^[\d.,]+(\s*\|\s*[\d.,]*)+$")
|
||||
_CHAPTER_RE = re.compile(
|
||||
r"(?:bab|chapter|lampiran)\s*([0-9IVXLC]+)",
|
||||
re.IGNORECASE,
|
||||
)
|
||||
|
||||
|
||||
def classify_chunk_tipe(text: str) -> str:
|
||||
"""Return 'tabel' if the chunk is mostly headerless numeric rows, else 'prosa'."""
|
||||
lines = [ln.strip() for ln in (text or "").splitlines() if ln.strip()]
|
||||
if not lines:
|
||||
return "prosa"
|
||||
numeric = sum(1 for ln in lines if _NUMERIC_PIPE_ROW.match(ln))
|
||||
if numeric >= max(2, len(lines) // 2):
|
||||
return "tabel"
|
||||
return "prosa"
|
||||
|
||||
|
||||
def detect_bab(text: str, source_filename: str = "") -> str:
|
||||
"""Best-effort chapter / lampiran label from chunk text or filename."""
|
||||
haystack = f"{source_filename}\n{text[:800]}"
|
||||
match = _CHAPTER_RE.search(haystack)
|
||||
if match:
|
||||
return match.group(0).strip()
|
||||
lower = source_filename.lower()
|
||||
if "lampiran" in lower:
|
||||
return "lampiran"
|
||||
return ""
|
||||
Binary file not shown.
Binary file not shown.
@@ -0,0 +1,120 @@
|
||||
import os
|
||||
import glob
|
||||
from dotenv import load_dotenv
|
||||
|
||||
# Load environment variables
|
||||
load_dotenv()
|
||||
|
||||
INPUT_DIR = os.getenv("INPUT_DIR", "documents")
|
||||
OUTPUT_TXT_DIR = os.getenv("OUTPUT_TXT_DIR", "extracted_txt")
|
||||
|
||||
def extract_docx(file_path):
|
||||
"""Mengekstrak teks dari file .docx dengan menjaga urutan asli paragraf dan tabel."""
|
||||
import docx
|
||||
from docx.oxml import OxmlElement
|
||||
from docx.text.paragraph import Paragraph
|
||||
from docx.table import Table
|
||||
|
||||
doc = docx.Document(file_path)
|
||||
full_text = []
|
||||
|
||||
# Iterasi semua elemen anak di dalam body document untuk menjaga urutan
|
||||
for element in doc.element.body:
|
||||
tag = element.tag
|
||||
if tag.endswith('p'):
|
||||
para = Paragraph(element, doc)
|
||||
if para.text.strip():
|
||||
full_text.append(para.text)
|
||||
elif tag.endswith('tbl'):
|
||||
table = Table(element, doc)
|
||||
table_text = []
|
||||
for row in table.rows:
|
||||
row_text = []
|
||||
for cell in row.cells:
|
||||
text = cell.text.strip()
|
||||
# Hindari duplikasi text sel gabungan (merged cells) secara berturut-turut
|
||||
if not row_text or row_text[-1] != text:
|
||||
row_text.append(text)
|
||||
if row_text:
|
||||
table_text.append(" | ".join(row_text))
|
||||
if table_text:
|
||||
full_text.append("\n".join(table_text))
|
||||
|
||||
return "\n\n".join(full_text)
|
||||
|
||||
def extract_pdf(file_path):
|
||||
"""Mengekstrak teks dari file .pdf halaman demi halaman."""
|
||||
from pypdf import PdfReader
|
||||
reader = PdfReader(file_path)
|
||||
full_text = []
|
||||
for i, page in enumerate(reader.pages):
|
||||
text = page.extract_text()
|
||||
if text and text.strip():
|
||||
full_text.append(text)
|
||||
return "\n".join(full_text)
|
||||
|
||||
def extract_txt(file_path):
|
||||
"""Membaca file teks dengan encoding UTF-8."""
|
||||
with open(file_path, "r", encoding="utf-8", errors="ignore") as f:
|
||||
return f.read()
|
||||
|
||||
def main():
|
||||
# Pastikan folder input ada
|
||||
if not os.path.exists(INPUT_DIR):
|
||||
print(f"Folder input '{INPUT_DIR}' tidak ditemukan. Membuat folder...")
|
||||
os.makedirs(INPUT_DIR)
|
||||
print(f"Silakan letakkan file dokumen Anda di folder '{INPUT_DIR}' lalu jalankan kembali script ini.")
|
||||
return
|
||||
|
||||
# Buat folder output jika belum ada
|
||||
os.makedirs(OUTPUT_TXT_DIR, exist_ok=True)
|
||||
|
||||
# Cari semua dokumen pendukung
|
||||
supported_extensions = ["*.docx", "*.pdf", "*.txt"]
|
||||
files_to_process = []
|
||||
for ext in supported_extensions:
|
||||
# Cari case-insensitive atau kombinasikan lowercase/uppercase
|
||||
files_to_process.extend(glob.glob(os.path.join(INPUT_DIR, ext)))
|
||||
files_to_process.extend(glob.glob(os.path.join(INPUT_DIR, ext.upper())))
|
||||
|
||||
# Hapus duplikasi jika ada (karena pencarian case-sensitive pada OS tertentu)
|
||||
files_to_process = list(set(files_to_process))
|
||||
|
||||
if not files_to_process:
|
||||
print(f"Tidak ada file .docx, .pdf, atau .txt yang ditemukan di folder '{INPUT_DIR}'.")
|
||||
return
|
||||
|
||||
print(f"Menemukan {len(files_to_process)} file dokumen untuk diekstrak.")
|
||||
|
||||
for file_path in files_to_process:
|
||||
filename = os.path.basename(file_path)
|
||||
base_name, ext = os.path.splitext(filename)
|
||||
output_file_path = os.path.join(OUTPUT_TXT_DIR, f"{base_name}.txt")
|
||||
|
||||
print(f"Mengekstrak: {filename} ... ", end="", flush=True)
|
||||
|
||||
try:
|
||||
ext_lower = ext.lower()
|
||||
if ext_lower == ".docx":
|
||||
text = extract_docx(file_path)
|
||||
elif ext_lower == ".pdf":
|
||||
text = extract_pdf(file_path)
|
||||
elif ext_lower == ".txt":
|
||||
text = extract_txt(file_path)
|
||||
else:
|
||||
print("Format tidak didukung (dilewati)")
|
||||
continue
|
||||
|
||||
# Simpan hasil teks ke file .txt di folder output
|
||||
with open(output_file_path, "w", encoding="utf-8") as f:
|
||||
f.write(text)
|
||||
|
||||
print(f"Selesai! Disimpan ke: {output_file_path}")
|
||||
|
||||
except Exception as e:
|
||||
print(f"GAGAL! Error: {str(e)}")
|
||||
|
||||
print("\nProses ekstraksi selesai seluruhnya!")
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
File diff suppressed because it is too large.
Load diff
@@ -0,0 +1,180 @@
|
||||
import os
|
||||
import glob
|
||||
from pathlib import Path
|
||||
from dotenv import load_dotenv
|
||||
import chromadb
|
||||
from sentence_transformers import SentenceTransformer
|
||||
|
||||
from chunk_classify import classify_chunk_tipe, detect_bab
|
||||
|
||||
# Set offline mode agar sentence-transformers tidak mencoba menghubungi Hugging Face di jaringan on-premise
|
||||
os.environ["HF_HUB_OFFLINE"] = "1"
|
||||
|
||||
# Load environment variables
|
||||
load_dotenv()
|
||||
|
||||
SCRIPT_DIR = Path(__file__).parent
|
||||
OUTPUT_TXT_DIR = os.getenv("OUTPUT_TXT_DIR", str(SCRIPT_DIR / "extracted_txt"))
|
||||
CHROMA_DB_DIR = os.getenv("CHROMA_DB_DIR", str(SCRIPT_DIR / "chroma_db"))
|
||||
EMBEDDING_MODEL_NAME = os.getenv(
|
||||
"EMBEDDING_MODEL_NAME",
|
||||
"sentence-transformers/paraphrase-multilingual-MiniLM-L12-v2",
|
||||
)
|
||||
|
||||
def chunk_text(text, chunk_size=800, chunk_overlap=150):
|
||||
"""Memecah teks menjadi chunk berdasarkan paragraf, baris, atau kata."""
|
||||
paragraphs = text.split("\n\n")
|
||||
chunks = []
|
||||
current_chunk = ""
|
||||
|
||||
for para in paragraphs:
|
||||
para = para.strip()
|
||||
if not para:
|
||||
continue
|
||||
|
||||
# Jika paragraf itu sendiri lebih besar dari chunk_size, bagi berdasarkan baris
|
||||
if len(para) > chunk_size:
|
||||
if current_chunk:
|
||||
chunks.append(current_chunk)
|
||||
current_chunk = ""
|
||||
|
||||
lines = para.split("\n")
|
||||
for line in lines:
|
||||
line = line.strip()
|
||||
if not line:
|
||||
continue
|
||||
|
||||
if len(line) > chunk_size:
|
||||
# Bagi baris panjang berdasarkan kata
|
||||
words = line.split(" ")
|
||||
temp_chunk = ""
|
||||
for word in words:
|
||||
if len(temp_chunk) + len(word) + 1 > chunk_size:
|
||||
if temp_chunk:
|
||||
chunks.append(temp_chunk)
|
||||
overlap_start = max(0, len(temp_chunk) - chunk_overlap)
|
||||
temp_chunk = temp_chunk[overlap_start:].strip()
|
||||
if temp_chunk:
|
||||
temp_chunk += " " + word
|
||||
else:
|
||||
temp_chunk = word
|
||||
else:
|
||||
if temp_chunk:
|
||||
temp_chunk += " " + word
|
||||
else:
|
||||
temp_chunk = word
|
||||
if temp_chunk:
|
||||
current_chunk = temp_chunk
|
||||
else:
|
||||
if len(current_chunk) + len(line) + 1 > chunk_size:
|
||||
chunks.append(current_chunk)
|
||||
overlap_start = max(0, len(current_chunk) - chunk_overlap)
|
||||
current_chunk = current_chunk[overlap_start:].strip()
|
||||
if current_chunk:
|
||||
current_chunk += "\n" + line
|
||||
else:
|
||||
current_chunk = line
|
||||
else:
|
||||
if current_chunk:
|
||||
current_chunk += "\n" + line
|
||||
else:
|
||||
current_chunk = line
|
||||
else:
|
||||
# Pengelompokan paragraf standar
|
||||
if len(current_chunk) + len(para) + 2 > chunk_size:
|
||||
chunks.append(current_chunk)
|
||||
overlap_start = max(0, len(current_chunk) - chunk_overlap)
|
||||
current_chunk = current_chunk[overlap_start:].strip()
|
||||
if current_chunk:
|
||||
current_chunk += "\n\n" + para
|
||||
else:
|
||||
current_chunk = para
|
||||
else:
|
||||
if current_chunk:
|
||||
current_chunk += "\n\n" + para
|
||||
else:
|
||||
current_chunk = para
|
||||
|
||||
if current_chunk:
|
||||
chunks.append(current_chunk)
|
||||
|
||||
return chunks
|
||||
|
||||
def main():
|
||||
if not os.path.exists(OUTPUT_TXT_DIR):
|
||||
print(f"Folder teks terekstrak '{OUTPUT_TXT_DIR}' tidak ditemukan. Jalankan extract_text.py terlebih dahulu.")
|
||||
return
|
||||
|
||||
txt_files = glob.glob(os.path.join(OUTPUT_TXT_DIR, "*.txt"))
|
||||
if not txt_files:
|
||||
print(f"Tidak ada file .txt ditemukan di '{OUTPUT_TXT_DIR}'. Jalankan extract_text.py terlebih dahulu.")
|
||||
return
|
||||
|
||||
# Inisialisasi Model Embedding lokal
|
||||
print(f"Memuat model embedding lokal '{EMBEDDING_MODEL_NAME}'...")
|
||||
model = SentenceTransformer(EMBEDDING_MODEL_NAME)
|
||||
print("Model embedding berhasil dimuat.")
|
||||
|
||||
# Inisialisasi Chroma DB client
|
||||
print(f"Menginisialisasi Chroma DB di folder '{CHROMA_DB_DIR}'...")
|
||||
client = chromadb.PersistentClient(path=CHROMA_DB_DIR)
|
||||
|
||||
# Hapus koleksi lama jika ada untuk menghindari duplikasi data lama saat indeks ulang
|
||||
try:
|
||||
client.delete_collection(name="company_sop")
|
||||
print("Koleksi lama 'company_sop' berhasil dihapus untuk indeks ulang.")
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
collection = client.create_collection(name="company_sop")
|
||||
|
||||
total_chunks = 0
|
||||
|
||||
for file_path in txt_files:
|
||||
filename = os.path.basename(file_path)
|
||||
print(f"\nMemproses chunking & embedding untuk file: {filename}...")
|
||||
|
||||
with open(file_path, "r", encoding="utf-8") as f:
|
||||
text = f.read()
|
||||
|
||||
chunks = chunk_text(text, chunk_size=800, chunk_overlap=150)
|
||||
if not chunks:
|
||||
print(f"File {filename} kosong atau tidak menghasilkan chunk.")
|
||||
continue
|
||||
|
||||
print(f"Menghasilkan {len(chunks)} chunks dari {filename}. Membuat embedding...")
|
||||
|
||||
# Hitung embeddings
|
||||
embeddings = model.encode(chunks)
|
||||
embeddings_list = [emb.tolist() for emb in embeddings]
|
||||
|
||||
# Metadata: tipe=prosa|tabel, bab — query path prefers prosa
|
||||
metadatas = []
|
||||
for i, chunk in enumerate(chunks):
|
||||
tipe = classify_chunk_tipe(chunk)
|
||||
bab = detect_bab(chunk, filename)
|
||||
metadatas.append(
|
||||
{
|
||||
"source": filename,
|
||||
"chunk_index": i,
|
||||
"tipe": tipe,
|
||||
"bab": bab or "",
|
||||
}
|
||||
)
|
||||
ids = [f"{filename}_chunk_{i}" for i in range(len(chunks))]
|
||||
|
||||
# Tambahkan ke Chroma DB
|
||||
collection.add(
|
||||
documents=chunks,
|
||||
embeddings=embeddings_list,
|
||||
metadatas=metadatas,
|
||||
ids=ids
|
||||
)
|
||||
|
||||
total_chunks += len(chunks)
|
||||
print(f"Berhasil menyimpan {len(chunks)} chunks ke Chroma DB.")
|
||||
|
||||
print(f"\nProses pembuatan database selesai! Total {total_chunks} chunks berhasil disimpan di Chroma DB.")
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,165 @@
|
||||
"""
|
||||
CLI / local-dev helper to query Chroma + an OpenAI-compatible LLM (e.g. LM Studio).
|
||||
|
||||
NOT used by the dashboard runtime. Production/lab path is:
|
||||
Django insight_service → HTTP → rag_service.py (/query) → Ollama from Django.
|
||||
Prefer `rag_service.py` + `populate_db.py` when testing what the app actually calls.
|
||||
"""
|
||||
|
||||
import os
|
||||
import sys
|
||||
import requests
|
||||
from dotenv import load_dotenv
|
||||
import chromadb
|
||||
from sentence_transformers import SentenceTransformer
|
||||
|
||||
# Set offline mode agar sentence-transformers tidak mencoba menghubungi Hugging Face di jaringan on-premise
|
||||
os.environ["HF_HUB_OFFLINE"] = "1"
|
||||
|
||||
# Load environment variables
|
||||
load_dotenv()
|
||||
|
||||
CHROMA_DB_DIR = os.getenv("CHROMA_DB_DIR", "chroma_db")
|
||||
EMBEDDING_MODEL_NAME = os.getenv("EMBEDDING_MODEL_NAME", "all-MiniLM-L6-v2")
|
||||
OPENAI_BASE_URL = os.getenv("OPENAI_BASE_URL", "http://localhost:1234/v1")
|
||||
OPENAI_API_KEY = os.getenv("OPENAI_API_KEY", "lm-studio")
|
||||
LLM_MODEL_NAME = os.getenv("LLM_MODEL_NAME", "qwen2.5:7b")
|
||||
|
||||
def main():
|
||||
# 1. Inisialisasi Database Vektor
|
||||
if not os.path.exists(CHROMA_DB_DIR):
|
||||
print(f"Database Chroma DB di '{CHROMA_DB_DIR}' tidak ditemukan. Silakan jalankan extract_text.py dan populate_db.py terlebih dahulu.")
|
||||
return
|
||||
|
||||
print("Menghubungkan ke Chroma DB...")
|
||||
chroma_client = chromadb.PersistentClient(path=CHROMA_DB_DIR)
|
||||
try:
|
||||
collection = chroma_client.get_collection(name="company_sop")
|
||||
except Exception as e:
|
||||
print(f"Koleksi 'company_sop' tidak ditemukan di database. Pastikan populate_db.py sudah dijalankan dengan sukses. Error: {e}")
|
||||
return
|
||||
|
||||
# 2. Inisialisasi Model Embedding lokal
|
||||
print(f"Memuat model embedding lokal '{EMBEDDING_MODEL_NAME}' untuk kueri...")
|
||||
embedding_model = SentenceTransformer(EMBEDDING_MODEL_NAME)
|
||||
print("Model embedding berhasil dimuat.")
|
||||
|
||||
# 3. Konfigurasi koneksi LM Studio native v1 API
|
||||
LM_STUDIO_API_URL = os.getenv("LM_STUDIO_API_URL")
|
||||
if not LM_STUDIO_API_URL:
|
||||
openai_base = os.getenv("OPENAI_BASE_URL", "http://localhost:1234/v1")
|
||||
if openai_base.endswith("/v1"):
|
||||
LM_STUDIO_API_URL = openai_base.replace("/v1", "/api/v1/chat")
|
||||
else:
|
||||
LM_STUDIO_API_URL = f"{openai_base.rstrip('/')}/api/v1/chat"
|
||||
|
||||
print(f"Mengonfigurasi koneksi LLM ke native API: {LM_STUDIO_API_URL} (Model: {LLM_MODEL_NAME})...")
|
||||
|
||||
print("\n" + "="*60)
|
||||
print(" PIPELINE RAG LOKAL - ASISTEN SOP PERUSAHAAN (QWEN 2.5)")
|
||||
print(" Ketik 'keluar' atau 'exit' untuk menyudahi percakapan.")
|
||||
print("="*60 + "\n")
|
||||
|
||||
while True:
|
||||
try:
|
||||
query = input("\nPertanyaan Anda: ").strip()
|
||||
if not query:
|
||||
continue
|
||||
if query.lower() in ["keluar", "exit", "q", "quit"]:
|
||||
print("Sampai jumpa!")
|
||||
break
|
||||
|
||||
print("\n[1/3] Mencari dokumen referensi relevan di database lokal...", end="", flush=True)
|
||||
# Buat embedding kueri
|
||||
query_embedding = embedding_model.encode([query])[0].tolist()
|
||||
|
||||
# Cari kueri di Chroma DB (ambil 6 chunk teratas)
|
||||
results = collection.query(
|
||||
query_embeddings=[query_embedding],
|
||||
n_results=6
|
||||
)
|
||||
print(" Selesai!")
|
||||
|
||||
retrieved_chunks = results['documents'][0]
|
||||
retrieved_metadatas = results['metadatas'][0]
|
||||
|
||||
if not retrieved_chunks or len(retrieved_chunks) == 0:
|
||||
print("⚠️ Tidak ditemukan referensi dokumen yang cocok dengan pertanyaan Anda.")
|
||||
continue
|
||||
|
||||
# Tampilkan referensi yang ditemukan
|
||||
print("\n[Referensi yang Ditemukan]:")
|
||||
for idx, meta in enumerate(retrieved_metadatas):
|
||||
print(f" - [{idx+1}] File: {meta['source']} (Chunk: {meta['chunk_index']})")
|
||||
|
||||
# 4. Susun Prompt dengan Konteks SOP
|
||||
context = "\n\n---\n\n".join(retrieved_chunks)
|
||||
|
||||
system_prompt = (
|
||||
"Anda adalah asisten AI perusahaan yang profesional. Tugas Anda adalah memberikan jawaban "
|
||||
"yang valid, akurat, dan sesuai dengan Standar Operasional Prosedur (SOP) atau dokumen acuan perusahaan "
|
||||
"yang disediakan di bawah ini.\n"
|
||||
"Patuhi aturan berikut:\n"
|
||||
"1. Jawablah HANYA berdasarkan informasi yang ada dalam dokumen acuan di bawah.\n"
|
||||
"2. Jika jawaban tidak dapat ditemukan di dalam dokumen tersebut secara eksplisit atau logis, katakan dengan sopan "
|
||||
"bahwa 'Maaf, informasi tersebut tidak ditemukan dalam dokumen SOP/acuan perusahaan kami.' Jangan mengarang informasi.\n"
|
||||
"3. Sajikan data dengan valid dan rapi."
|
||||
)
|
||||
|
||||
user_prompt = f"""Dokumen SOP / Acuan Perusahaan:
|
||||
=========================================
|
||||
{context}
|
||||
=========================================
|
||||
|
||||
Pertanyaan Pengguna: {query}
|
||||
|
||||
Jawaban berdasarkan Dokumen Acuan:"""
|
||||
|
||||
print(f"\n[2/3] Menghubungi LLM Qwen 2.5 lokal di {LM_STUDIO_API_URL}...", end="", flush=True)
|
||||
|
||||
# Panggil LM Studio native v1 API
|
||||
headers = {
|
||||
"Content-Type": "application/json"
|
||||
}
|
||||
if OPENAI_API_KEY and OPENAI_API_KEY != "lm-studio":
|
||||
headers["Authorization"] = f"Bearer {OPENAI_API_KEY}"
|
||||
|
||||
payload = {
|
||||
"model": LLM_MODEL_NAME,
|
||||
"input": user_prompt,
|
||||
"system_prompt": system_prompt,
|
||||
"temperature": 0.1,
|
||||
"max_output_tokens": 32000
|
||||
}
|
||||
|
||||
response = requests.post(LM_STUDIO_API_URL, json=payload, headers=headers)
|
||||
response.raise_for_status()
|
||||
print(" Selesai!")
|
||||
|
||||
answer = response.json()["response"]
|
||||
|
||||
# Pisahkan proses berpikir (<think>) jika ada (khusus model reasoning seperti Qwen 2.5)
|
||||
import re
|
||||
think_match = re.search(r'<think>(.*?)</think>', answer, re.DOTALL)
|
||||
clean_answer = re.sub(r'<think>.*?</think>', '', answer, flags=re.DOTALL).strip()
|
||||
|
||||
if think_match and think_match.group(1).strip():
|
||||
print(" Selesai!")
|
||||
print("\n[Proses Berpikir Qwen 2.5]:")
|
||||
print("." * 50)
|
||||
print(think_match.group(1).strip())
|
||||
print("." * 50)
|
||||
else:
|
||||
print(" Selesai!")
|
||||
|
||||
print("\n[3/3] Respon Asisten SOP:")
|
||||
print("-"*50)
|
||||
print(clean_answer)
|
||||
print("-"*50)
|
||||
|
||||
except Exception as e:
|
||||
print(f"\n❌ Terjadi kesalahan: {e}")
|
||||
print("Harap pastikan server LLM lokal Anda (LM Studio/vLLM/llama.cpp) sedang berjalan dan dapat diakses.")
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,198 @@
|
||||
"""
|
||||
RAG microservice for the CP 707 knowledge base (ChromaDB + SentenceTransformers).
|
||||
|
||||
Runtime entrypoint used by Django `insight_service.fetch_rag_chunks`
|
||||
via HTTP `RAG_SERVICE_URL` (default compose :5002; NUC AI lab often :5102).
|
||||
Offline HuggingFace mode — do not confuse with `query_rag.py` (CLI/dev only).
|
||||
"""
|
||||
|
||||
import os
|
||||
import sys
|
||||
from pathlib import Path
|
||||
from dotenv import load_dotenv
|
||||
|
||||
# Set offline agar tidak download dari HuggingFace
|
||||
os.environ["HF_HUB_OFFLINE"] = "1"
|
||||
os.environ["TRANSFORMERS_OFFLINE"] = "1"
|
||||
|
||||
# Load .env dari direktori yang sama dengan script ini
|
||||
script_dir = Path(__file__).parent
|
||||
load_dotenv(script_dir / ".env")
|
||||
|
||||
CHROMA_DB_DIR = str(script_dir / os.getenv("CHROMA_DB_DIR", "chroma_db"))
|
||||
EMBEDDING_MODEL_NAME = os.getenv("EMBEDDING_MODEL_NAME", "sentence-transformers/paraphrase-multilingual-MiniLM-L12-v2")
|
||||
RAG_PORT = int(os.getenv("RAG_PORT", "5002"))
|
||||
COLLECTION_NAME = "company_sop"
|
||||
|
||||
print(f"[RAG Service] ChromaDB path: {CHROMA_DB_DIR}")
|
||||
print(f"[RAG Service] Embedding model: {EMBEDDING_MODEL_NAME}")
|
||||
|
||||
# Import setelah env set
|
||||
import chromadb
|
||||
from sentence_transformers import SentenceTransformer
|
||||
from fastapi import FastAPI, HTTPException
|
||||
from fastapi.middleware.cors import CORSMiddleware
|
||||
from pydantic import BaseModel, Field
|
||||
import uvicorn
|
||||
|
||||
# ─── Init ChromaDB & Embedding Model ─────────────────────────────────────────
|
||||
print("[RAG Service] Memuat ChromaDB...")
|
||||
try:
|
||||
chroma_client = chromadb.PersistentClient(path=CHROMA_DB_DIR)
|
||||
collection = chroma_client.get_collection(name=COLLECTION_NAME)
|
||||
total_chunks = collection.count()
|
||||
print(f"[RAG Service] ChromaDB loaded. Total chunks: {total_chunks}")
|
||||
except Exception as e:
|
||||
print(f"[RAG Service] ERROR: Gagal load ChromaDB: {e}")
|
||||
sys.exit(1)
|
||||
|
||||
print(f"[RAG Service] Memuat embedding model '{EMBEDDING_MODEL_NAME}'...")
|
||||
try:
|
||||
embedding_model = SentenceTransformer(EMBEDDING_MODEL_NAME)
|
||||
print("[RAG Service] Embedding model berhasil dimuat.")
|
||||
except Exception as e:
|
||||
print(f"[RAG Service] ERROR: Gagal load embedding model: {e}")
|
||||
print("[RAG Service] Pastikan model sudah didownload. Jalankan populate_db.py terlebih dahulu.")
|
||||
sys.exit(1)
|
||||
|
||||
# ─── FastAPI App ──────────────────────────────────────────────────────────────
|
||||
app = FastAPI(
|
||||
title="CP707 RAG Service",
|
||||
description="Retrieval-Augmented Generation service untuk buku Manajemen Broiler CP 707",
|
||||
version="1.0.0"
|
||||
)
|
||||
|
||||
app.add_middleware(
|
||||
CORSMiddleware,
|
||||
allow_origins=[
|
||||
"http://localhost:3000",
|
||||
"http://localhost:3001",
|
||||
"http://localhost:5001",
|
||||
"http://127.0.0.1:5001",
|
||||
"http://localhost:8000",
|
||||
"http://127.0.0.1:8000",
|
||||
],
|
||||
allow_methods=["GET", "POST"],
|
||||
allow_headers=["*"],
|
||||
)
|
||||
|
||||
# ─── Request/Response Models ─────────────────────────────────────────────────
|
||||
class QueryRequest(BaseModel):
|
||||
query: str = Field(..., min_length=1, description="Query text untuk mencari chunk CP707 relevan")
|
||||
n_results: int = Field(default=4, ge=1, le=10, description="Jumlah chunk yang dikembalikan")
|
||||
topic: str = Field(default="", description="Topic insight: berat_ayam, fcr, iot_panel, dll")
|
||||
# Prefer prose SOP; table rows without headers are dangerous for the LLM.
|
||||
tipe: str = Field(default="prosa", description="Filter metadata tipe: prosa | tabel | any")
|
||||
|
||||
class QueryResponse(BaseModel):
|
||||
success: bool
|
||||
chunks: list
|
||||
sources: list
|
||||
metadatas: list
|
||||
total_found: int
|
||||
query_used: str
|
||||
|
||||
class HealthResponse(BaseModel):
|
||||
status: str
|
||||
total_chunks: int
|
||||
embedding_model: str
|
||||
|
||||
# ─── Topic → Query enhancement mapping ──────────────────────────────────────
|
||||
# Tambahkan keyword relevan per topic agar embedding search lebih tepat sasaran
|
||||
TOPIC_QUERY_HINTS = {
|
||||
"berat_ayam": "berat badan target bobot ADG pertumbuhan standar mingguan ayam broiler",
|
||||
"fcr": "FCR feed conversion ratio konsumsi pakan efisiensi standar broiler",
|
||||
"iot_panel": "suhu kandang kelembapan amonia CO2 ventilasi lingkungan pemeliharaan broiler",
|
||||
"eef": "EEF indeks performa IP efisiensi produksi siklus broiler",
|
||||
"hitung_ayam": "mortalitas deplesi kematian afkir populasi standar toleransi broiler",
|
||||
"hitung_karung": "pakan karung konsumsi harian feed intake standar broiler",
|
||||
}
|
||||
|
||||
# ─── Endpoints ───────────────────────────────────────────────────────────────
|
||||
@app.get("/health", response_model=HealthResponse)
|
||||
def health_check():
|
||||
return HealthResponse(
|
||||
status="ok",
|
||||
total_chunks=collection.count(),
|
||||
embedding_model=EMBEDDING_MODEL_NAME,
|
||||
)
|
||||
|
||||
@app.post("/query", response_model=QueryResponse)
|
||||
def query_cp707(req: QueryRequest):
|
||||
"""
|
||||
Cari chunk CP707 yang relevan berdasarkan query.
|
||||
Jika topic disediakan, tambahkan hint keyword agar hasil lebih relevan.
|
||||
"""
|
||||
enhanced_query = req.query
|
||||
if req.topic and req.topic in TOPIC_QUERY_HINTS:
|
||||
enhanced_query = f"{req.query} {TOPIC_QUERY_HINTS[req.topic]}"
|
||||
|
||||
try:
|
||||
query_embedding = embedding_model.encode([enhanced_query])[0].tolist()
|
||||
|
||||
# Over-fetch then filter by tipe so prosa chunks win when metadata exists.
|
||||
fetch_n = min(max(req.n_results * 3, req.n_results), max(collection.count(), 1))
|
||||
where = None
|
||||
if req.tipe and req.tipe != "any":
|
||||
where = {"tipe": req.tipe}
|
||||
|
||||
try:
|
||||
results = collection.query(
|
||||
query_embeddings=[query_embedding],
|
||||
n_results=fetch_n,
|
||||
where=where,
|
||||
)
|
||||
except Exception:
|
||||
# Older indexes may lack tipe metadata — fall back unfiltered.
|
||||
results = collection.query(
|
||||
query_embeddings=[query_embedding],
|
||||
n_results=fetch_n,
|
||||
)
|
||||
|
||||
chunks = results["documents"][0] if results["documents"] else []
|
||||
metadatas = results["metadatas"][0] if results["metadatas"] else []
|
||||
|
||||
# If unfiltered fallback returned tables, drop them when prosa was requested.
|
||||
if req.tipe == "prosa" and metadatas:
|
||||
paired = [
|
||||
(c, m)
|
||||
for c, m in zip(chunks, metadatas)
|
||||
if (m or {}).get("tipe", "prosa") != "tabel"
|
||||
]
|
||||
if paired:
|
||||
chunks, metadatas = [list(x) for x in zip(*paired)]
|
||||
else:
|
||||
# Keep original if everything was tabel (better something than nothing;
|
||||
# Django stripHeaderlessTables still cleans numeric rows).
|
||||
pass
|
||||
|
||||
chunks = chunks[: req.n_results]
|
||||
metadatas = metadatas[: req.n_results]
|
||||
|
||||
sources = []
|
||||
for m in metadatas:
|
||||
bab = (m or {}).get("bab") or ""
|
||||
src = (m or {}).get("source", "unknown")
|
||||
idx = (m or {}).get("chunk_index", "?")
|
||||
label = f"{src}"
|
||||
if bab:
|
||||
label += f" / {bab}"
|
||||
label += f" (chunk {idx})"
|
||||
sources.append(label)
|
||||
|
||||
return QueryResponse(
|
||||
success=True,
|
||||
chunks=chunks,
|
||||
sources=sources,
|
||||
metadatas=metadatas,
|
||||
total_found=len(chunks),
|
||||
query_used=enhanced_query,
|
||||
)
|
||||
|
||||
except Exception as e:
|
||||
raise HTTPException(status_code=500, detail=f"RAG query error: {str(e)}")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
print(f"[RAG Service] Starting on http://0.0.0.0:{RAG_PORT}")
|
||||
uvicorn.run(app, host="0.0.0.0", port=RAG_PORT, log_level="warning")
|
||||
@@ -0,0 +1,8 @@
|
||||
chromadb
|
||||
sentence-transformers
|
||||
python-docx
|
||||
pypdf
|
||||
python-dotenv
|
||||
openai
|
||||
fastapi
|
||||
uvicorn
|
||||
Reference in new issue
Block a user