181 lines
6.6 KiB
Python
181 lines
6.6 KiB
Python
import os
|
|
import glob
|
|
from pathlib import Path
|
|
from dotenv import load_dotenv
|
|
import chromadb
|
|
from sentence_transformers import SentenceTransformer
|
|
|
|
from chunk_classify import classify_chunk_tipe, detect_bab
|
|
|
|
# Set offline mode agar sentence-transformers tidak mencoba menghubungi Hugging Face di jaringan on-premise
|
|
os.environ["HF_HUB_OFFLINE"] = "1"
|
|
|
|
# Load environment variables
|
|
load_dotenv()
|
|
|
|
SCRIPT_DIR = Path(__file__).parent
|
|
OUTPUT_TXT_DIR = os.getenv("OUTPUT_TXT_DIR", str(SCRIPT_DIR / "extracted_txt"))
|
|
CHROMA_DB_DIR = os.getenv("CHROMA_DB_DIR", str(SCRIPT_DIR / "chroma_db"))
|
|
EMBEDDING_MODEL_NAME = os.getenv(
|
|
"EMBEDDING_MODEL_NAME",
|
|
"sentence-transformers/paraphrase-multilingual-MiniLM-L12-v2",
|
|
)
|
|
|
|
def chunk_text(text, chunk_size=800, chunk_overlap=150):
|
|
"""Memecah teks menjadi chunk berdasarkan paragraf, baris, atau kata."""
|
|
paragraphs = text.split("\n\n")
|
|
chunks = []
|
|
current_chunk = ""
|
|
|
|
for para in paragraphs:
|
|
para = para.strip()
|
|
if not para:
|
|
continue
|
|
|
|
# Jika paragraf itu sendiri lebih besar dari chunk_size, bagi berdasarkan baris
|
|
if len(para) > chunk_size:
|
|
if current_chunk:
|
|
chunks.append(current_chunk)
|
|
current_chunk = ""
|
|
|
|
lines = para.split("\n")
|
|
for line in lines:
|
|
line = line.strip()
|
|
if not line:
|
|
continue
|
|
|
|
if len(line) > chunk_size:
|
|
# Bagi baris panjang berdasarkan kata
|
|
words = line.split(" ")
|
|
temp_chunk = ""
|
|
for word in words:
|
|
if len(temp_chunk) + len(word) + 1 > chunk_size:
|
|
if temp_chunk:
|
|
chunks.append(temp_chunk)
|
|
overlap_start = max(0, len(temp_chunk) - chunk_overlap)
|
|
temp_chunk = temp_chunk[overlap_start:].strip()
|
|
if temp_chunk:
|
|
temp_chunk += " " + word
|
|
else:
|
|
temp_chunk = word
|
|
else:
|
|
if temp_chunk:
|
|
temp_chunk += " " + word
|
|
else:
|
|
temp_chunk = word
|
|
if temp_chunk:
|
|
current_chunk = temp_chunk
|
|
else:
|
|
if len(current_chunk) + len(line) + 1 > chunk_size:
|
|
chunks.append(current_chunk)
|
|
overlap_start = max(0, len(current_chunk) - chunk_overlap)
|
|
current_chunk = current_chunk[overlap_start:].strip()
|
|
if current_chunk:
|
|
current_chunk += "\n" + line
|
|
else:
|
|
current_chunk = line
|
|
else:
|
|
if current_chunk:
|
|
current_chunk += "\n" + line
|
|
else:
|
|
current_chunk = line
|
|
else:
|
|
# Pengelompokan paragraf standar
|
|
if len(current_chunk) + len(para) + 2 > chunk_size:
|
|
chunks.append(current_chunk)
|
|
overlap_start = max(0, len(current_chunk) - chunk_overlap)
|
|
current_chunk = current_chunk[overlap_start:].strip()
|
|
if current_chunk:
|
|
current_chunk += "\n\n" + para
|
|
else:
|
|
current_chunk = para
|
|
else:
|
|
if current_chunk:
|
|
current_chunk += "\n\n" + para
|
|
else:
|
|
current_chunk = para
|
|
|
|
if current_chunk:
|
|
chunks.append(current_chunk)
|
|
|
|
return chunks
|
|
|
|
def main():
|
|
if not os.path.exists(OUTPUT_TXT_DIR):
|
|
print(f"Folder teks terekstrak '{OUTPUT_TXT_DIR}' tidak ditemukan. Jalankan extract_text.py terlebih dahulu.")
|
|
return
|
|
|
|
txt_files = glob.glob(os.path.join(OUTPUT_TXT_DIR, "*.txt"))
|
|
if not txt_files:
|
|
print(f"Tidak ada file .txt ditemukan di '{OUTPUT_TXT_DIR}'. Jalankan extract_text.py terlebih dahulu.")
|
|
return
|
|
|
|
# Inisialisasi Model Embedding lokal
|
|
print(f"Memuat model embedding lokal '{EMBEDDING_MODEL_NAME}'...")
|
|
model = SentenceTransformer(EMBEDDING_MODEL_NAME)
|
|
print("Model embedding berhasil dimuat.")
|
|
|
|
# Inisialisasi Chroma DB client
|
|
print(f"Menginisialisasi Chroma DB di folder '{CHROMA_DB_DIR}'...")
|
|
client = chromadb.PersistentClient(path=CHROMA_DB_DIR)
|
|
|
|
# Hapus koleksi lama jika ada untuk menghindari duplikasi data lama saat indeks ulang
|
|
try:
|
|
client.delete_collection(name="company_sop")
|
|
print("Koleksi lama 'company_sop' berhasil dihapus untuk indeks ulang.")
|
|
except Exception:
|
|
pass
|
|
|
|
collection = client.create_collection(name="company_sop")
|
|
|
|
total_chunks = 0
|
|
|
|
for file_path in txt_files:
|
|
filename = os.path.basename(file_path)
|
|
print(f"\nMemproses chunking & embedding untuk file: {filename}...")
|
|
|
|
with open(file_path, "r", encoding="utf-8") as f:
|
|
text = f.read()
|
|
|
|
chunks = chunk_text(text, chunk_size=800, chunk_overlap=150)
|
|
if not chunks:
|
|
print(f"File {filename} kosong atau tidak menghasilkan chunk.")
|
|
continue
|
|
|
|
print(f"Menghasilkan {len(chunks)} chunks dari {filename}. Membuat embedding...")
|
|
|
|
# Hitung embeddings
|
|
embeddings = model.encode(chunks)
|
|
embeddings_list = [emb.tolist() for emb in embeddings]
|
|
|
|
# Metadata: tipe=prosa|tabel, bab — query path prefers prosa
|
|
metadatas = []
|
|
for i, chunk in enumerate(chunks):
|
|
tipe = classify_chunk_tipe(chunk)
|
|
bab = detect_bab(chunk, filename)
|
|
metadatas.append(
|
|
{
|
|
"source": filename,
|
|
"chunk_index": i,
|
|
"tipe": tipe,
|
|
"bab": bab or "",
|
|
}
|
|
)
|
|
ids = [f"{filename}_chunk_{i}" for i in range(len(chunks))]
|
|
|
|
# Tambahkan ke Chroma DB
|
|
collection.add(
|
|
documents=chunks,
|
|
embeddings=embeddings_list,
|
|
metadatas=metadatas,
|
|
ids=ids
|
|
)
|
|
|
|
total_chunks += len(chunks)
|
|
print(f"Berhasil menyimpan {len(chunks)} chunks ke Chroma DB.")
|
|
|
|
print(f"\nProses pembuatan database selesai! Total {total_chunks} chunks berhasil disimpan di Chroma DB.")
|
|
|
|
if __name__ == "__main__":
|
|
main()
|