121 lines
4.2 KiB
Python
121 lines
4.2 KiB
Python
import os
|
|
import glob
|
|
from dotenv import load_dotenv
|
|
|
|
# Load environment variables
|
|
load_dotenv()
|
|
|
|
INPUT_DIR = os.getenv("INPUT_DIR", "documents")
|
|
OUTPUT_TXT_DIR = os.getenv("OUTPUT_TXT_DIR", "extracted_txt")
|
|
|
|
def extract_docx(file_path):
|
|
"""Mengekstrak teks dari file .docx dengan menjaga urutan asli paragraf dan tabel."""
|
|
import docx
|
|
from docx.oxml import OxmlElement
|
|
from docx.text.paragraph import Paragraph
|
|
from docx.table import Table
|
|
|
|
doc = docx.Document(file_path)
|
|
full_text = []
|
|
|
|
# Iterasi semua elemen anak di dalam body document untuk menjaga urutan
|
|
for element in doc.element.body:
|
|
tag = element.tag
|
|
if tag.endswith('p'):
|
|
para = Paragraph(element, doc)
|
|
if para.text.strip():
|
|
full_text.append(para.text)
|
|
elif tag.endswith('tbl'):
|
|
table = Table(element, doc)
|
|
table_text = []
|
|
for row in table.rows:
|
|
row_text = []
|
|
for cell in row.cells:
|
|
text = cell.text.strip()
|
|
# Hindari duplikasi text sel gabungan (merged cells) secara berturut-turut
|
|
if not row_text or row_text[-1] != text:
|
|
row_text.append(text)
|
|
if row_text:
|
|
table_text.append(" | ".join(row_text))
|
|
if table_text:
|
|
full_text.append("\n".join(table_text))
|
|
|
|
return "\n\n".join(full_text)
|
|
|
|
def extract_pdf(file_path):
|
|
"""Mengekstrak teks dari file .pdf halaman demi halaman."""
|
|
from pypdf import PdfReader
|
|
reader = PdfReader(file_path)
|
|
full_text = []
|
|
for i, page in enumerate(reader.pages):
|
|
text = page.extract_text()
|
|
if text and text.strip():
|
|
full_text.append(text)
|
|
return "\n".join(full_text)
|
|
|
|
def extract_txt(file_path):
|
|
"""Membaca file teks dengan encoding UTF-8."""
|
|
with open(file_path, "r", encoding="utf-8", errors="ignore") as f:
|
|
return f.read()
|
|
|
|
def main():
|
|
# Pastikan folder input ada
|
|
if not os.path.exists(INPUT_DIR):
|
|
print(f"Folder input '{INPUT_DIR}' tidak ditemukan. Membuat folder...")
|
|
os.makedirs(INPUT_DIR)
|
|
print(f"Silakan letakkan file dokumen Anda di folder '{INPUT_DIR}' lalu jalankan kembali script ini.")
|
|
return
|
|
|
|
# Buat folder output jika belum ada
|
|
os.makedirs(OUTPUT_TXT_DIR, exist_ok=True)
|
|
|
|
# Cari semua dokumen pendukung
|
|
supported_extensions = ["*.docx", "*.pdf", "*.txt"]
|
|
files_to_process = []
|
|
for ext in supported_extensions:
|
|
# Cari case-insensitive atau kombinasikan lowercase/uppercase
|
|
files_to_process.extend(glob.glob(os.path.join(INPUT_DIR, ext)))
|
|
files_to_process.extend(glob.glob(os.path.join(INPUT_DIR, ext.upper())))
|
|
|
|
# Hapus duplikasi jika ada (karena pencarian case-sensitive pada OS tertentu)
|
|
files_to_process = list(set(files_to_process))
|
|
|
|
if not files_to_process:
|
|
print(f"Tidak ada file .docx, .pdf, atau .txt yang ditemukan di folder '{INPUT_DIR}'.")
|
|
return
|
|
|
|
print(f"Menemukan {len(files_to_process)} file dokumen untuk diekstrak.")
|
|
|
|
for file_path in files_to_process:
|
|
filename = os.path.basename(file_path)
|
|
base_name, ext = os.path.splitext(filename)
|
|
output_file_path = os.path.join(OUTPUT_TXT_DIR, f"{base_name}.txt")
|
|
|
|
print(f"Mengekstrak: {filename} ... ", end="", flush=True)
|
|
|
|
try:
|
|
ext_lower = ext.lower()
|
|
if ext_lower == ".docx":
|
|
text = extract_docx(file_path)
|
|
elif ext_lower == ".pdf":
|
|
text = extract_pdf(file_path)
|
|
elif ext_lower == ".txt":
|
|
text = extract_txt(file_path)
|
|
else:
|
|
print("Format tidak didukung (dilewati)")
|
|
continue
|
|
|
|
# Simpan hasil teks ke file .txt di folder output
|
|
with open(output_file_path, "w", encoding="utf-8") as f:
|
|
f.write(text)
|
|
|
|
print(f"Selesai! Disimpan ke: {output_file_path}")
|
|
|
|
except Exception as e:
|
|
print(f"GAGAL! Error: {str(e)}")
|
|
|
|
print("\nProses ekstraksi selesai seluruhnya!")
|
|
|
|
if __name__ == "__main__":
|
|
main()
|