commit df500733db59f13419ac08911905cc3e0549c655 Author: dsutanto Date: Fri Jul 24 10:04:46 2026 +0700 First Commit diff --git a/.gitignore b/.gitignore new file mode 100644 index 0000000..d8bf50a --- /dev/null +++ b/.gitignore @@ -0,0 +1,18 @@ +# Python +__pycache__/ +*.py[cod] +*.so +*.egg-info/ +dist/ +build/ +*.egg +venv/ +.env + +# EPUB files +*.epub +original/ +translated/ + +# State files +translate_state.json diff --git a/requirements.txt b/requirements.txt new file mode 100644 index 0000000..9176ece --- /dev/null +++ b/requirements.txt @@ -0,0 +1,4 @@ +beautifulsoup4>=4.12 +lxml>=5.0 +requests>=2.30 +urllib3>=2.0 diff --git a/translate_epub.py b/translate_epub.py new file mode 100644 index 0000000..5166c7f --- /dev/null +++ b/translate_epub.py @@ -0,0 +1,346 @@ +#!/usr/bin/env python3 +""" +EPUB Translator EN -> ID. Uses proxies with connection pooling. +Verbose progress. Saves to translated/ folder. +""" + +import socket +_orig = socket.getaddrinfo +def _ipv4(h, p, f=0, t=0, pr=0, fl=0): + return _orig(h, p, socket.AF_INET, t, pr, fl) +socket.getaddrinfo = _ipv4 + +import os, sys, time, glob, json, random +import zipfile, tempfile, shutil +from bs4 import BeautifulSoup +from deep_translator import GoogleTranslator + +# ── Proxy configuration ─────────────────────────────────────────── +PROXIES = [ + '103.80.237.30:8888', + '103.80.237.27:8888', + '103.80.237.163:8888', + '103.185.47.51:8888', + '103.185.47.53:8888', + '103.185.47.54:8888', +] +PROXY_USER = 'dspaceme' +PROXY_PASS = 'anakmanis' + +SEP = '[[SPLIT]]' +MAX_CHUNK = 4000 +DELAY = 0.3 + +BASE_DIR = os.path.dirname(os.path.abspath(__file__)) +SOURCE_DIR = os.path.join(BASE_DIR, 'original') +TRANSLATED_DIR = os.path.join(BASE_DIR, 'translated') +STATE_FILE = os.path.join(BASE_DIR, 'translate_state.json') + +def log(msg): + print(msg, flush=True) + + +def translate_text(text): + """Translate using deep_translator.""" + try: + translator = GoogleTranslator(source='en', target='id') + result = translator.translate(text) + return result if result else None + except Exception: + return None + + +def translate_rotated(text): + """Try translation. If 429, retry after delay.""" + for attempt in range(5): + result = translate_text(text) + if result: + return result + time.sleep(2 ** attempt) + return None + + +def translate_batch(texts): + if not texts: + return texts + non_empty = [t for t in texts if t.strip()] + if not non_empty: + return texts + + combined = SEP.join(non_empty) + result = translate_rotated(combined) + if result: + parts = result.split(SEP) + if len(parts) == len(non_empty): + pi = 0 + out = [] + for t in texts: + if t.strip(): + out.append(parts[pi].strip()) + pi += 1 + else: + out.append('') + return out + + # Fallback: individual + out = [] + for t in texts: + if not t.strip(): + out.append('') + continue + r = translate_rotated(t) + out.append(r.strip() if r else t) + time.sleep(0.1) + return out + + +def epub_output_name(epub_path): + name = os.path.basename(epub_path) + name = name.replace('.epub', '_-_Bahasa_Indonesia.epub') + return name + + +def process_epub(epub_path, output_path, state_info=None): + epub_name = os.path.basename(epub_path) + tmpdir = tempfile.mkdtemp(prefix='epub_trans_') + t_start = time.time() + + try: + log(f' Extracting...') + with zipfile.ZipFile(epub_path, 'r') as z: + z.extractall(tmpdir) + + html_files = [] + for root, dirs, files in os.walk(tmpdir): + for fn in files: + if fn.endswith(('.html', '.xhtml', '.htm')): + html_files.append(os.path.join(root, fn)) + + total_html = len(html_files) + log(f' {total_html} HTML files') + translated_count = 0 + batch_total = 0 + + for i, fp in enumerate(html_files): + rel = os.path.relpath(fp, tmpdir) + try: + with open(fp, 'r', encoding='utf-8') as f: + html = f.read() + soup = BeautifulSoup(html, 'xml') + body = soup.find('body') + if not body: + continue + + elems = [] + for tag in ('p', 'a', 'span', 'h1', 'h2', 'h3', 'h4', 'h5', 'h6'): + for e in body.find_all(tag): + t = e.get_text().strip() + if not t or len(t) <= 2: + continue + kids = set(getattr(c, 'name', None) for c in e.children if hasattr(c, 'name')) + if kids - {None, 'span', 'a', 'b', 'i', 'em', 'strong'}: + continue + elems.append(e) + + if not elems: + continue + + all_texts = [e.get_text().strip() for e in elems] + + batches = [] + cur, curlen = [], 0 + for t in all_texts: + t2 = t.strip() + if not t2: + cur.append('') + continue + if curlen + len(t2) + len(SEP) > MAX_CHUNK and cur: + batches.append(cur) + cur, curlen = [], 0 + cur.append(t2) + curlen += len(t2) + len(SEP) + if cur: + batches.append(cur) + + results = [None] * len(all_texts) + idx = 0 + for batch in batches: + translated = translate_batch(batch) + for j, trans in enumerate(translated): + results[idx + j] = trans + idx += len(batch) + batch_total += 1 + time.sleep(DELAY) + + changed = False + for e, orig, trans in zip(elems, all_texts, results): + trans = trans.strip() if trans else '' + if trans and trans != orig: + e.clear() + e.string = trans + changed = True + + if changed: + with open(fp, 'w', encoding='utf-8') as f: + f.write(str(soup)) + translated_count += 1 + + pct = int((i + 1) * 100 / total_html) + filled = int(30 * (i + 1) / total_html) + bar = '#' * filled + '-' * (30 - filled) + elapsed = time.time() - t_start + rate = (i + 1) / elapsed if elapsed > 0 else 0 + eta = (total_html - i - 1) / rate if rate > 0 else 0 + line = f'\r [{bar}] {i+1}/{total_html} ({pct}%) | tr:{translated_count} | {elapsed:.0f}s | ETA:{eta:.0f}s' + sys.stdout.write(line) + sys.stdout.flush() + + except Exception as e: + sys.stdout.write('\n') + sys.stdout.flush() + log(f' WARN {rel}: {e}') + + sys.stdout.write('\n') + sys.stdout.flush() + + log(f' Metadata...') + opf_files = glob.glob(os.path.join(tmpdir, '**', '*.opf'), recursive=True) + for opf in opf_files: + try: + with open(opf, 'r', encoding='utf-8') as f: + content = f.read() + soup = BeautifulSoup(content, 'xml') + for tag_name, attr in [('dc:language', None), ('package', 'xml:lang'), ('package', 'lang')]: + elem = soup.find(tag_name) if attr is None else soup.find(tag_name.split(':')[0]) + if elem: + if attr: + elem[attr] = 'id' + else: + elem.string = 'id' + dc_title = soup.find('dc:title') + if dc_title and '(Bahasa Indonesia)' not in dc_title.text: + dc_title.string = dc_title.text.strip() + ' (Bahasa Indonesia)' + with open(opf, 'w', encoding='utf-8') as f: + f.write(str(soup)) + except Exception: + pass + + log(f' Building EPUB...') + os.makedirs(os.path.dirname(output_path), exist_ok=True) + with zipfile.ZipFile(output_path, 'w', zipfile.ZIP_DEFLATED) as z: + z.writestr('mimetype', 'application/epub+zip', zipfile.ZIP_STORED) + for root, dirs, files in os.walk(tmpdir): + for fn in files: + full = os.path.join(root, fn) + arcname = os.path.relpath(full, tmpdir) + if arcname == 'mimetype': + continue + z.write(full, arcname, zipfile.ZIP_DEFLATED) + + return translated_count, total_html, batch_total + + finally: + shutil.rmtree(tmpdir, ignore_errors=True) + + +def load_state(): + if os.path.exists(STATE_FILE): + with open(STATE_FILE) as f: + return json.load(f) + return {} + + +def save_state(state): + with open(STATE_FILE, 'w') as f: + json.dump(state, f) + + +def main(): + import argparse + parser = argparse.ArgumentParser(description='Translate EPUBs EN->ID') + parser.add_argument('--force', action='store_true') + parser.add_argument('--dry-run', action='store_true') + parser.add_argument('--proxy', action='append') + parser.add_argument('--proxy-user') + parser.add_argument('--proxy-pass') + args = parser.parse_args() + + global PROXIES, PROXY_USER, PROXY_PASS + if args.proxy: + PROXIES = args.proxy + PROXY_USER = args.proxy_user or '' + PROXY_PASS = args.proxy_pass or '' + + state = load_state() + os.makedirs(TRANSLATED_DIR, exist_ok=True) + + if not os.path.isdir(SOURCE_DIR): + log(f"Source directory not found: {SOURCE_DIR}") + return + + all_epubs = sorted(glob.glob(os.path.join(SOURCE_DIR, '*.epub'))) + existing = set(os.path.basename(tf) for tf in glob.glob(os.path.join(TRANSLATED_DIR, '*.epub'))) + + todo = [] + for epub in all_epubs: + bn = os.path.basename(epub) + if bn.endswith('_-_Bahasa_Indonesia.epub'): + continue + if epub_output_name(epub) in existing and not args.force: + continue + if bn in state.get('done', {}) and not args.force: + continue + todo.append(epub) + + if args.dry_run: + log(f"Would translate {len(todo)} EPUB(s):") + for e in todo: + log(f" {os.path.basename(e)} -> {epub_output_name(e)}") + return + + if not todo: + log("No new EPUBs!") + return + + total = len(todo) + t0 = time.time() + log(f"To translate: {total} EPUB(s)") + log("=" * 60) + + for i, epub_path in enumerate(todo): + epub_name = os.path.basename(epub_path) + output_name = epub_output_name(epub_path) + output_path = os.path.join(TRANSLATED_DIR, output_name) + + elapsed = time.time() - t0 + rate = elapsed / max(i, 1) if i else 0 + eta = rate * (total - i) + fs = os.path.getsize(epub_path) / 1024 + + log(f'[{i+1}/{total}] {epub_name} ({fs:.0f}KB)') + log(f' -> {output_name}') + + try: + t1 = time.time() + tc, th, bt = process_epub(epub_path, output_path) + fe = time.time() - t1 + log(f' DONE: {tc}/{th} files, {bt} batches, {fe:.0f}s ({os.path.getsize(output_path)/1024:.0f}KB)') + state.setdefault('done', {})[epub_name] = { + 'output': output_name, 'time': round(fe), + 'files': tc, 'batches': bt + } + save_state(state) + except Exception as e: + log(f' ERROR: {e}') + import traceback + traceback.print_exc() + + log(f' [{i+1}/{total}] ETA: {eta/60:.0f}m\n') + + log("=" * 60) + log(f"Done: {len(state.get('done',{}))} EPUBs in {(time.time()-t0)/60:.1f} min") + log(f"Output: {TRANSLATED_DIR}/") + + +if __name__ == '__main__': + main()