commit
32a36cceff
444 files changed
+67186
No files matched your search
@@ -0,0 +1,162 @@
|
||||
import os
|
||||
import glob
|
||||
from dotenv import load_dotenv
|
||||
import chromadb
|
||||
from sentence_transformers import SentenceTransformer
|
||||
|
||||
# Set offline mode agar sentence-transformers tidak mencoba menghubungi Hugging Face di jaringan on-premise
|
||||
os.environ["HF_HUB_OFFLINE"] = "1"
|
||||
|
||||
# Load environment variables
|
||||
load_dotenv()
|
||||
|
||||
OUTPUT_TXT_DIR = os.getenv("OUTPUT_TXT_DIR", "extracted_txt")
|
||||
CHROMA_DB_DIR = os.getenv("CHROMA_DB_DIR", "chroma_db")
|
||||
EMBEDDING_MODEL_NAME = os.getenv("EMBEDDING_MODEL_NAME", "all-MiniLM-L6-v2")
|
||||
|
||||
def chunk_text(text, chunk_size=800, chunk_overlap=150):
|
||||
"""Memecah teks menjadi chunk berdasarkan paragraf, baris, atau kata."""
|
||||
paragraphs = text.split("\n\n")
|
||||
chunks = []
|
||||
current_chunk = ""
|
||||
|
||||
for para in paragraphs:
|
||||
para = para.strip()
|
||||
if not para:
|
||||
continue
|
||||
|
||||
# Jika paragraf itu sendiri lebih besar dari chunk_size, bagi berdasarkan baris
|
||||
if len(para) > chunk_size:
|
||||
if current_chunk:
|
||||
chunks.append(current_chunk)
|
||||
current_chunk = ""
|
||||
|
||||
lines = para.split("\n")
|
||||
for line in lines:
|
||||
line = line.strip()
|
||||
if not line:
|
||||
continue
|
||||
|
||||
if len(line) > chunk_size:
|
||||
# Bagi baris panjang berdasarkan kata
|
||||
words = line.split(" ")
|
||||
temp_chunk = ""
|
||||
for word in words:
|
||||
if len(temp_chunk) + len(word) + 1 > chunk_size:
|
||||
if temp_chunk:
|
||||
chunks.append(temp_chunk)
|
||||
overlap_start = max(0, len(temp_chunk) - chunk_overlap)
|
||||
temp_chunk = temp_chunk[overlap_start:].strip()
|
||||
if temp_chunk:
|
||||
temp_chunk += " " + word
|
||||
else:
|
||||
temp_chunk = word
|
||||
else:
|
||||
if temp_chunk:
|
||||
temp_chunk += " " + word
|
||||
else:
|
||||
temp_chunk = word
|
||||
if temp_chunk:
|
||||
current_chunk = temp_chunk
|
||||
else:
|
||||
if len(current_chunk) + len(line) + 1 > chunk_size:
|
||||
chunks.append(current_chunk)
|
||||
overlap_start = max(0, len(current_chunk) - chunk_overlap)
|
||||
current_chunk = current_chunk[overlap_start:].strip()
|
||||
if current_chunk:
|
||||
current_chunk += "\n" + line
|
||||
else:
|
||||
current_chunk = line
|
||||
else:
|
||||
if current_chunk:
|
||||
current_chunk += "\n" + line
|
||||
else:
|
||||
current_chunk = line
|
||||
else:
|
||||
# Pengelompokan paragraf standar
|
||||
if len(current_chunk) + len(para) + 2 > chunk_size:
|
||||
chunks.append(current_chunk)
|
||||
overlap_start = max(0, len(current_chunk) - chunk_overlap)
|
||||
current_chunk = current_chunk[overlap_start:].strip()
|
||||
if current_chunk:
|
||||
current_chunk += "\n\n" + para
|
||||
else:
|
||||
current_chunk = para
|
||||
else:
|
||||
if current_chunk:
|
||||
current_chunk += "\n\n" + para
|
||||
else:
|
||||
current_chunk = para
|
||||
|
||||
if current_chunk:
|
||||
chunks.append(current_chunk)
|
||||
|
||||
return chunks
|
||||
|
||||
def main():
|
||||
if not os.path.exists(OUTPUT_TXT_DIR):
|
||||
print(f"Folder teks terekstrak '{OUTPUT_TXT_DIR}' tidak ditemukan. Jalankan extract_text.py terlebih dahulu.")
|
||||
return
|
||||
|
||||
txt_files = glob.glob(os.path.join(OUTPUT_TXT_DIR, "*.txt"))
|
||||
if not txt_files:
|
||||
print(f"Tidak ada file .txt ditemukan di '{OUTPUT_TXT_DIR}'. Jalankan extract_text.py terlebih dahulu.")
|
||||
return
|
||||
|
||||
# Inisialisasi Model Embedding lokal
|
||||
print(f"Memuat model embedding lokal '{EMBEDDING_MODEL_NAME}'...")
|
||||
model = SentenceTransformer(EMBEDDING_MODEL_NAME)
|
||||
print("Model embedding berhasil dimuat.")
|
||||
|
||||
# Inisialisasi Chroma DB client
|
||||
print(f"Menginisialisasi Chroma DB di folder '{CHROMA_DB_DIR}'...")
|
||||
client = chromadb.PersistentClient(path=CHROMA_DB_DIR)
|
||||
|
||||
# Hapus koleksi lama jika ada untuk menghindari duplikasi data lama saat indeks ulang
|
||||
try:
|
||||
client.delete_collection(name="company_sop")
|
||||
print("Koleksi lama 'company_sop' berhasil dihapus untuk indeks ulang.")
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
collection = client.create_collection(name="company_sop")
|
||||
|
||||
total_chunks = 0
|
||||
|
||||
for file_path in txt_files:
|
||||
filename = os.path.basename(file_path)
|
||||
print(f"\nMemproses chunking & embedding untuk file: {filename}...")
|
||||
|
||||
with open(file_path, "r", encoding="utf-8") as f:
|
||||
text = f.read()
|
||||
|
||||
chunks = chunk_text(text, chunk_size=800, chunk_overlap=150)
|
||||
if not chunks:
|
||||
print(f"File {filename} kosong atau tidak menghasilkan chunk.")
|
||||
continue
|
||||
|
||||
print(f"Menghasilkan {len(chunks)} chunks dari {filename}. Membuat embedding...")
|
||||
|
||||
# Hitung embeddings
|
||||
embeddings = model.encode(chunks)
|
||||
embeddings_list = [emb.tolist() for emb in embeddings]
|
||||
|
||||
# Siapkan metadata dan ID unik untuk Chroma DB
|
||||
metadatas = [{"source": filename, "chunk_index": i} for i in range(len(chunks))]
|
||||
ids = [f"{filename}_chunk_{i}" for i in range(len(chunks))]
|
||||
|
||||
# Tambahkan ke Chroma DB
|
||||
collection.add(
|
||||
documents=chunks,
|
||||
embeddings=embeddings_list,
|
||||
metadatas=metadatas,
|
||||
ids=ids
|
||||
)
|
||||
|
||||
total_chunks += len(chunks)
|
||||
print(f"Berhasil menyimpan {len(chunks)} chunks ke Chroma DB.")
|
||||
|
||||
print(f"\nProses pembuatan database selesai! Total {total_chunks} chunks berhasil disimpan di Chroma DB.")
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
Reference in new issue
Block a user