commit
32a36cceff
444 files changed
+67186
No files matched your search
@@ -0,0 +1,120 @@
|
||||
import os
|
||||
import glob
|
||||
from dotenv import load_dotenv
|
||||
|
||||
# Load environment variables
|
||||
load_dotenv()
|
||||
|
||||
INPUT_DIR = os.getenv("INPUT_DIR", "documents")
|
||||
OUTPUT_TXT_DIR = os.getenv("OUTPUT_TXT_DIR", "extracted_txt")
|
||||
|
||||
def extract_docx(file_path):
|
||||
"""Mengekstrak teks dari file .docx dengan menjaga urutan asli paragraf dan tabel."""
|
||||
import docx
|
||||
from docx.oxml import OxmlElement
|
||||
from docx.text.paragraph import Paragraph
|
||||
from docx.table import Table
|
||||
|
||||
doc = docx.Document(file_path)
|
||||
full_text = []
|
||||
|
||||
# Iterasi semua elemen anak di dalam body document untuk menjaga urutan
|
||||
for element in doc.element.body:
|
||||
tag = element.tag
|
||||
if tag.endswith('p'):
|
||||
para = Paragraph(element, doc)
|
||||
if para.text.strip():
|
||||
full_text.append(para.text)
|
||||
elif tag.endswith('tbl'):
|
||||
table = Table(element, doc)
|
||||
table_text = []
|
||||
for row in table.rows:
|
||||
row_text = []
|
||||
for cell in row.cells:
|
||||
text = cell.text.strip()
|
||||
# Hindari duplikasi text sel gabungan (merged cells) secara berturut-turut
|
||||
if not row_text or row_text[-1] != text:
|
||||
row_text.append(text)
|
||||
if row_text:
|
||||
table_text.append(" | ".join(row_text))
|
||||
if table_text:
|
||||
full_text.append("\n".join(table_text))
|
||||
|
||||
return "\n\n".join(full_text)
|
||||
|
||||
def extract_pdf(file_path):
|
||||
"""Mengekstrak teks dari file .pdf halaman demi halaman."""
|
||||
from pypdf import PdfReader
|
||||
reader = PdfReader(file_path)
|
||||
full_text = []
|
||||
for i, page in enumerate(reader.pages):
|
||||
text = page.extract_text()
|
||||
if text and text.strip():
|
||||
full_text.append(text)
|
||||
return "\n".join(full_text)
|
||||
|
||||
def extract_txt(file_path):
|
||||
"""Membaca file teks dengan encoding UTF-8."""
|
||||
with open(file_path, "r", encoding="utf-8", errors="ignore") as f:
|
||||
return f.read()
|
||||
|
||||
def main():
|
||||
# Pastikan folder input ada
|
||||
if not os.path.exists(INPUT_DIR):
|
||||
print(f"Folder input '{INPUT_DIR}' tidak ditemukan. Membuat folder...")
|
||||
os.makedirs(INPUT_DIR)
|
||||
print(f"Silakan letakkan file dokumen Anda di folder '{INPUT_DIR}' lalu jalankan kembali script ini.")
|
||||
return
|
||||
|
||||
# Buat folder output jika belum ada
|
||||
os.makedirs(OUTPUT_TXT_DIR, exist_ok=True)
|
||||
|
||||
# Cari semua dokumen pendukung
|
||||
supported_extensions = ["*.docx", "*.pdf", "*.txt"]
|
||||
files_to_process = []
|
||||
for ext in supported_extensions:
|
||||
# Cari case-insensitive atau kombinasikan lowercase/uppercase
|
||||
files_to_process.extend(glob.glob(os.path.join(INPUT_DIR, ext)))
|
||||
files_to_process.extend(glob.glob(os.path.join(INPUT_DIR, ext.upper())))
|
||||
|
||||
# Hapus duplikasi jika ada (karena pencarian case-sensitive pada OS tertentu)
|
||||
files_to_process = list(set(files_to_process))
|
||||
|
||||
if not files_to_process:
|
||||
print(f"Tidak ada file .docx, .pdf, atau .txt yang ditemukan di folder '{INPUT_DIR}'.")
|
||||
return
|
||||
|
||||
print(f"Menemukan {len(files_to_process)} file dokumen untuk diekstrak.")
|
||||
|
||||
for file_path in files_to_process:
|
||||
filename = os.path.basename(file_path)
|
||||
base_name, ext = os.path.splitext(filename)
|
||||
output_file_path = os.path.join(OUTPUT_TXT_DIR, f"{base_name}.txt")
|
||||
|
||||
print(f"Mengekstrak: {filename} ... ", end="", flush=True)
|
||||
|
||||
try:
|
||||
ext_lower = ext.lower()
|
||||
if ext_lower == ".docx":
|
||||
text = extract_docx(file_path)
|
||||
elif ext_lower == ".pdf":
|
||||
text = extract_pdf(file_path)
|
||||
elif ext_lower == ".txt":
|
||||
text = extract_txt(file_path)
|
||||
else:
|
||||
print("Format tidak didukung (dilewati)")
|
||||
continue
|
||||
|
||||
# Simpan hasil teks ke file .txt di folder output
|
||||
with open(output_file_path, "w", encoding="utf-8") as f:
|
||||
f.write(text)
|
||||
|
||||
print(f"Selesai! Disimpan ke: {output_file_path}")
|
||||
|
||||
except Exception as e:
|
||||
print(f"GAGAL! Error: {str(e)}")
|
||||
|
||||
print("\nProses ekstraksi selesai seluruhnya!")
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
Reference in new issue
Block a user