#!/usr/bin/env python3 """ EPUB Translator EN -> ID. Uses proxies with connection pooling. Verbose progress. Saves to translated/ folder. """ import socket _orig = socket.getaddrinfo def _ipv4(h, p, f=0, t=0, pr=0, fl=0): return _orig(h, p, socket.AF_INET, t, pr, fl) socket.getaddrinfo = _ipv4 import os, sys, time, glob, json, random, re import zipfile, tempfile, shutil from bs4 import BeautifulSoup from deep_translator import GoogleTranslator # ── Proxy configuration ─────────────────────────────────────────── PROXIES = [ '103.80.237.30:8888', '103.80.237.27:8888', '103.80.237.163:8888', '103.185.47.51:8888', '103.185.47.53:8888', '103.185.47.54:8888', ] PROXY_USER = 'dspaceme' PROXY_PASS = 'anakmanis' SEP = '[[SPLIT]]' MAX_CHUNK = 4000 DELAY = 0.3 SENTENCE_RE = re.compile(r'(?<=[.!?])\s+') # Patterns to protect from translation ROMAN_NUMERAL = re.compile(r'\b([IVXLCDM]+)\b', re.IGNORECASE) CHAPTER_PREFIX = re.compile(r'(Chapter\s+)([IVXLCDM]+|\d+)', re.IGNORECASE) ROMAN_STANDALONE = re.compile(r'^\s*([IVXLCDM]+)[.\s]*$', re.IGNORECASE) PLACEHOLDER_RE = re.compile(r'xxKEEP(\d+)xx') BASE_DIR = os.path.dirname(os.path.abspath(__file__)) SOURCE_DIR = os.path.join(BASE_DIR, 'original') TRANSLATED_DIR = os.path.join(BASE_DIR, 'translated') STATE_FILE = os.path.join(BASE_DIR, 'translate_state.json') def log(msg): print(msg, flush=True) def protect_numerals(text): """Replace chapter numbers / Roman numerals with placeholders.""" protected = [] def _store(val): protected.append(val) return f'xxKEEP{len(protected)-1}xx' # Protect "Chapter X" - only replace the numeral, keep "Chapter" for translation text = CHAPTER_PREFIX.sub(lambda m: m.group(1) + _store(m.group(2)), text) # Protect standalone Roman numerals (chapter headings) text = ROMAN_STANDALONE.sub(lambda m: _store(m.group(1)), text) # Protect remaining Roman numerals in text text = ROMAN_NUMERAL.sub(lambda m: _store(m.group(1)), text) return text, protected def restore_numerals(text, protected): """Restore original numerals from placeholders.""" def _restore(m): idx = int(m.group(1)) return protected[idx] if idx < len(protected) else m.group(0) return PLACEHOLDER_RE.sub(_restore, text) def translate_text(text): """Translate using deep_translator, preserving chapter numbers.""" clean_text, protected = protect_numerals(text) try: translator = GoogleTranslator(source='en', target='id') result = translator.translate(clean_text) if result and protected: result = restore_numerals(result, protected) return result if result else None except Exception: return None def translate_rotated(text): """Try translation. If 429, retry after delay.""" for attempt in range(5): result = translate_text(text) if result: return result time.sleep(2 ** attempt) return None def translate_batch(texts): if not texts: return texts non_empty = [t for t in texts if t.strip()] if not non_empty: return texts combined = SEP.join(non_empty) result = translate_rotated(combined) if result: parts = result.split(SEP) if len(parts) == len(non_empty): pi = 0 out = [] for t in texts: if t.strip(): out.append(parts[pi].strip()) pi += 1 else: out.append('') return out # Fallback: individual out = [] for t in texts: if not t.strip(): out.append('') continue r = translate_rotated(t) out.append(r.strip() if r else t) time.sleep(0.1) return out def epub_output_name(epub_path): name = os.path.basename(epub_path) name = name.replace('.epub', '_-_Bahasa_Indonesia.epub') return name def process_epub(epub_path, output_path, state_info=None): epub_name = os.path.basename(epub_path) tmpdir = tempfile.mkdtemp(prefix='epub_trans_') t_start = time.time() try: log(f' Extracting...') with zipfile.ZipFile(epub_path, 'r') as z: z.extractall(tmpdir) html_files = [] for root, dirs, files in os.walk(tmpdir): for fn in files: if fn.endswith(('.html', '.xhtml', '.htm')): html_files.append(os.path.join(root, fn)) total_html = len(html_files) log(f' {total_html} HTML files') translated_count = 0 batch_total = 0 for i, fp in enumerate(html_files): rel = os.path.relpath(fp, tmpdir) try: with open(fp, 'r', encoding='utf-8') as f: html = f.read() soup = BeautifulSoup(html, 'xml') body = soup.find('body') if not body: continue elems = [] for tag in ('p', 'a', 'span', 'h1', 'h2', 'h3', 'h4', 'h5', 'h6'): for e in body.find_all(tag): t = e.get_text().strip() if not t or len(t) <= 2: continue kids = set(getattr(c, 'name', None) for c in e.children if hasattr(c, 'name')) if kids - {None, 'span', 'a', 'b', 'i', 'em', 'strong'}: continue elems.append(e) if not elems: continue all_texts = [e.get_text().strip() for e in elems] # Split each element's text into sentences elem_sentences = [] # list of (elem_idx, [sentence1, sentence2, ...]) for ei, t in enumerate(all_texts): if not t: elem_sentences.append((ei, [])) elif len(t) <= MAX_CHUNK: elem_sentences.append((ei, [t])) else: parts = SENTENCE_RE.split(t) elem_sentences.append((ei, parts)) # Flatten sentences into a list, tracking which element each belongs to flat = [] # (elem_idx, sentence_idx_in_elem, text) for ei, sentences in elem_sentences: for si, s in enumerate(sentences): flat.append((ei, si, s.strip())) # Build batches of sentences up to MAX_CHUNK batches = [] cur, curlen = [], 0 for ei, si, s in flat: if not s: continue if curlen + len(s) + len(SEP) > MAX_CHUNK and cur: batches.append(cur) cur, curlen = [], 0 cur.append((ei, si, s)) curlen += len(s) + len(SEP) if cur: batches.append(cur) # Translate sentence batches and reassemble per element elem_translated_sentences = {} for batch in batches: texts = [s for _, _, s in batch] translated = translate_batch(texts) for (ei, si, _), trans in zip(batch, translated): elem_translated_sentences.setdefault(ei, {})[si] = trans batch_total += 1 time.sleep(DELAY) # Reconstruct per-element text results = [] for ei, (_, sentences) in enumerate(elem_sentences): if ei in elem_translated_sentences and sentences: trans_dict = elem_translated_sentences[ei] parts = [trans_dict.get(si, s) for si, s in enumerate(sentences)] results.append(' '.join(parts)) else: results.append(all_texts[ei]) changed = False for e, orig, trans in zip(elems, all_texts, results): trans = trans.strip() if trans else '' if trans and trans != orig: e.clear() e.string = trans changed = True if changed: with open(fp, 'w', encoding='utf-8') as f: f.write(str(soup)) translated_count += 1 pct = int((i + 1) * 100 / total_html) filled = int(30 * (i + 1) / total_html) bar = '#' * filled + '-' * (30 - filled) elapsed = time.time() - t_start rate = (i + 1) / elapsed if elapsed > 0 else 0 eta = (total_html - i - 1) / rate if rate > 0 else 0 line = f'\r [{bar}] {i+1}/{total_html} ({pct}%) | tr:{translated_count} | {elapsed:.0f}s | ETA:{eta:.0f}s' sys.stdout.write(line) sys.stdout.flush() except Exception as e: sys.stdout.write('\n') sys.stdout.flush() log(f' WARN {rel}: {e}') sys.stdout.write('\n') sys.stdout.flush() log(f' Metadata...') opf_files = glob.glob(os.path.join(tmpdir, '**', '*.opf'), recursive=True) for opf in opf_files: try: with open(opf, 'r', encoding='utf-8') as f: content = f.read() soup = BeautifulSoup(content, 'xml') for tag_name, attr in [('dc:language', None), ('package', 'xml:lang'), ('package', 'lang')]: elem = soup.find(tag_name) if attr is None else soup.find(tag_name.split(':')[0]) if elem: if attr: elem[attr] = 'id' else: elem.string = 'id' dc_title = soup.find('dc:title') if dc_title and '(Bahasa Indonesia)' not in dc_title.text: dc_title.string = dc_title.text.strip() + ' (Bahasa Indonesia)' with open(opf, 'w', encoding='utf-8') as f: f.write(str(soup)) except Exception: pass log(f' Building EPUB...') os.makedirs(os.path.dirname(output_path), exist_ok=True) with zipfile.ZipFile(output_path, 'w', zipfile.ZIP_DEFLATED) as z: z.writestr('mimetype', 'application/epub+zip', zipfile.ZIP_STORED) for root, dirs, files in os.walk(tmpdir): for fn in files: full = os.path.join(root, fn) arcname = os.path.relpath(full, tmpdir) if arcname == 'mimetype': continue z.write(full, arcname, zipfile.ZIP_DEFLATED) return translated_count, total_html, batch_total finally: shutil.rmtree(tmpdir, ignore_errors=True) def load_state(): if os.path.exists(STATE_FILE): with open(STATE_FILE) as f: return json.load(f) return {} def save_state(state): with open(STATE_FILE, 'w') as f: json.dump(state, f) def main(): import argparse parser = argparse.ArgumentParser(description='Translate EPUBs EN->ID') parser.add_argument('--force', action='store_true') parser.add_argument('--dry-run', action='store_true') parser.add_argument('--proxy', action='append') parser.add_argument('--proxy-user') parser.add_argument('--proxy-pass') args = parser.parse_args() global PROXIES, PROXY_USER, PROXY_PASS if args.proxy: PROXIES = args.proxy PROXY_USER = args.proxy_user or '' PROXY_PASS = args.proxy_pass or '' state = load_state() os.makedirs(TRANSLATED_DIR, exist_ok=True) if not os.path.isdir(SOURCE_DIR): log(f"Source directory not found: {SOURCE_DIR}") return all_epubs = sorted(glob.glob(os.path.join(SOURCE_DIR, '*.epub'))) existing = set(os.path.basename(tf) for tf in glob.glob(os.path.join(TRANSLATED_DIR, '*.epub'))) todo = [] for epub in all_epubs: bn = os.path.basename(epub) if bn.endswith('_-_Bahasa_Indonesia.epub'): continue if epub_output_name(epub) in existing and not args.force: continue if bn in state.get('done', {}) and not args.force: continue todo.append(epub) if args.dry_run: log(f"Would translate {len(todo)} EPUB(s):") for e in todo: log(f" {os.path.basename(e)} -> {epub_output_name(e)}") return if not todo: log("No new EPUBs!") return total = len(todo) t0 = time.time() log(f"To translate: {total} EPUB(s)") log("=" * 60) for i, epub_path in enumerate(todo): epub_name = os.path.basename(epub_path) output_name = epub_output_name(epub_path) output_path = os.path.join(TRANSLATED_DIR, output_name) elapsed = time.time() - t0 rate = elapsed / max(i, 1) if i else 0 eta = rate * (total - i) fs = os.path.getsize(epub_path) / 1024 log(f'[{i+1}/{total}] {epub_name} ({fs:.0f}KB)') log(f' -> {output_name}') try: t1 = time.time() tc, th, bt = process_epub(epub_path, output_path) fe = time.time() - t1 log(f' DONE: {tc}/{th} files, {bt} batches, {fe:.0f}s ({os.path.getsize(output_path)/1024:.0f}KB)') state.setdefault('done', {})[epub_name] = { 'output': output_name, 'time': round(fe), 'files': tc, 'batches': bt } save_state(state) except Exception as e: log(f' ERROR: {e}') import traceback traceback.print_exc() log(f' [{i+1}/{total}] ETA: {eta/60:.0f}m\n') log("=" * 60) log(f"Done: {len(state.get('done',{}))} EPUBs in {(time.time()-t0)/60:.1f} min") log(f"Output: {TRANSLATED_DIR}/") if __name__ == '__main__': main()