374 lines
13 KiB
Python
374 lines
13 KiB
Python
#!/usr/bin/env python3
|
|
"""
|
|
EPUB Translator EN -> ID. Uses proxies with connection pooling.
|
|
Verbose progress. Saves to translated/ folder.
|
|
"""
|
|
|
|
import socket
|
|
_orig = socket.getaddrinfo
|
|
def _ipv4(h, p, f=0, t=0, pr=0, fl=0):
|
|
return _orig(h, p, socket.AF_INET, t, pr, fl)
|
|
socket.getaddrinfo = _ipv4
|
|
|
|
import os, sys, time, glob, json, random, re
|
|
import zipfile, tempfile, shutil
|
|
from bs4 import BeautifulSoup
|
|
from deep_translator import GoogleTranslator
|
|
|
|
# ── Proxy configuration ───────────────────────────────────────────
|
|
PROXIES = [
|
|
'103.80.237.30:8888',
|
|
'103.80.237.27:8888',
|
|
'103.80.237.163:8888',
|
|
'103.185.47.51:8888',
|
|
'103.185.47.53:8888',
|
|
'103.185.47.54:8888',
|
|
]
|
|
PROXY_USER = 'dspaceme'
|
|
PROXY_PASS = 'anakmanis'
|
|
|
|
SEP = '[[SPLIT]]'
|
|
MAX_CHUNK = 4000
|
|
DELAY = 0.3
|
|
SENTENCE_RE = re.compile(r'(?<=[.!?])\s+')
|
|
|
|
BASE_DIR = os.path.dirname(os.path.abspath(__file__))
|
|
SOURCE_DIR = os.path.join(BASE_DIR, 'original')
|
|
TRANSLATED_DIR = os.path.join(BASE_DIR, 'translated')
|
|
STATE_FILE = os.path.join(BASE_DIR, 'translate_state.json')
|
|
|
|
def log(msg):
|
|
print(msg, flush=True)
|
|
|
|
|
|
def translate_text(text):
|
|
"""Translate using deep_translator."""
|
|
try:
|
|
translator = GoogleTranslator(source='en', target='id')
|
|
result = translator.translate(text)
|
|
return result if result else None
|
|
except Exception:
|
|
return None
|
|
|
|
|
|
def translate_rotated(text):
|
|
"""Try translation. If 429, retry after delay."""
|
|
for attempt in range(5):
|
|
result = translate_text(text)
|
|
if result:
|
|
return result
|
|
time.sleep(2 ** attempt)
|
|
return None
|
|
|
|
|
|
def translate_batch(texts):
|
|
if not texts:
|
|
return texts
|
|
non_empty = [t for t in texts if t.strip()]
|
|
if not non_empty:
|
|
return texts
|
|
|
|
combined = SEP.join(non_empty)
|
|
result = translate_rotated(combined)
|
|
if result:
|
|
parts = result.split(SEP)
|
|
if len(parts) == len(non_empty):
|
|
pi = 0
|
|
out = []
|
|
for t in texts:
|
|
if t.strip():
|
|
out.append(parts[pi].strip())
|
|
pi += 1
|
|
else:
|
|
out.append('')
|
|
return out
|
|
|
|
# Fallback: individual
|
|
out = []
|
|
for t in texts:
|
|
if not t.strip():
|
|
out.append('')
|
|
continue
|
|
r = translate_rotated(t)
|
|
out.append(r.strip() if r else t)
|
|
time.sleep(0.1)
|
|
return out
|
|
|
|
|
|
def epub_output_name(epub_path):
|
|
name = os.path.basename(epub_path)
|
|
name = name.replace('.epub', '_-_Bahasa_Indonesia.epub')
|
|
return name
|
|
|
|
|
|
def process_epub(epub_path, output_path, state_info=None):
|
|
epub_name = os.path.basename(epub_path)
|
|
tmpdir = tempfile.mkdtemp(prefix='epub_trans_')
|
|
t_start = time.time()
|
|
|
|
try:
|
|
log(f' Extracting...')
|
|
with zipfile.ZipFile(epub_path, 'r') as z:
|
|
z.extractall(tmpdir)
|
|
|
|
html_files = []
|
|
for root, dirs, files in os.walk(tmpdir):
|
|
for fn in files:
|
|
if fn.endswith(('.html', '.xhtml', '.htm')):
|
|
html_files.append(os.path.join(root, fn))
|
|
|
|
total_html = len(html_files)
|
|
log(f' {total_html} HTML files')
|
|
translated_count = 0
|
|
batch_total = 0
|
|
|
|
for i, fp in enumerate(html_files):
|
|
rel = os.path.relpath(fp, tmpdir)
|
|
try:
|
|
with open(fp, 'r', encoding='utf-8') as f:
|
|
html = f.read()
|
|
soup = BeautifulSoup(html, 'xml')
|
|
body = soup.find('body')
|
|
if not body:
|
|
continue
|
|
|
|
elems = []
|
|
for tag in ('p', 'a', 'span', 'h1', 'h2', 'h3', 'h4', 'h5', 'h6'):
|
|
for e in body.find_all(tag):
|
|
t = e.get_text().strip()
|
|
if not t or len(t) <= 2:
|
|
continue
|
|
kids = set(getattr(c, 'name', None) for c in e.children if hasattr(c, 'name'))
|
|
if kids - {None, 'span', 'a', 'b', 'i', 'em', 'strong'}:
|
|
continue
|
|
elems.append(e)
|
|
|
|
if not elems:
|
|
continue
|
|
|
|
all_texts = [e.get_text().strip() for e in elems]
|
|
|
|
# Split each element's text into sentences
|
|
elem_sentences = [] # list of (elem_idx, [sentence1, sentence2, ...])
|
|
for ei, t in enumerate(all_texts):
|
|
if not t:
|
|
elem_sentences.append((ei, []))
|
|
elif len(t) <= MAX_CHUNK:
|
|
elem_sentences.append((ei, [t]))
|
|
else:
|
|
parts = SENTENCE_RE.split(t)
|
|
elem_sentences.append((ei, parts))
|
|
|
|
# Flatten sentences into a list, tracking which element each belongs to
|
|
flat = [] # (elem_idx, sentence_idx_in_elem, text)
|
|
for ei, sentences in elem_sentences:
|
|
for si, s in enumerate(sentences):
|
|
flat.append((ei, si, s.strip()))
|
|
|
|
# Build batches of sentences up to MAX_CHUNK
|
|
batches = []
|
|
cur, curlen = [], 0
|
|
for ei, si, s in flat:
|
|
if not s:
|
|
continue
|
|
if curlen + len(s) + len(SEP) > MAX_CHUNK and cur:
|
|
batches.append(cur)
|
|
cur, curlen = [], 0
|
|
cur.append((ei, si, s))
|
|
curlen += len(s) + len(SEP)
|
|
if cur:
|
|
batches.append(cur)
|
|
|
|
# Translate sentence batches and reassemble per element
|
|
elem_translated_sentences = {}
|
|
for batch in batches:
|
|
texts = [s for _, _, s in batch]
|
|
translated = translate_batch(texts)
|
|
for (ei, si, _), trans in zip(batch, translated):
|
|
elem_translated_sentences.setdefault(ei, {})[si] = trans
|
|
batch_total += 1
|
|
time.sleep(DELAY)
|
|
|
|
# Reconstruct per-element text
|
|
results = []
|
|
for ei, (_, sentences) in enumerate(elem_sentences):
|
|
if ei in elem_translated_sentences and sentences:
|
|
trans_dict = elem_translated_sentences[ei]
|
|
parts = [trans_dict.get(si, s) for si, s in enumerate(sentences)]
|
|
results.append(' '.join(parts))
|
|
else:
|
|
results.append(all_texts[ei])
|
|
|
|
changed = False
|
|
for e, orig, trans in zip(elems, all_texts, results):
|
|
trans = trans.strip() if trans else ''
|
|
if trans and trans != orig:
|
|
e.clear()
|
|
e.string = trans
|
|
changed = True
|
|
|
|
if changed:
|
|
with open(fp, 'w', encoding='utf-8') as f:
|
|
f.write(str(soup))
|
|
translated_count += 1
|
|
|
|
pct = int((i + 1) * 100 / total_html)
|
|
filled = int(30 * (i + 1) / total_html)
|
|
bar = '#' * filled + '-' * (30 - filled)
|
|
elapsed = time.time() - t_start
|
|
rate = (i + 1) / elapsed if elapsed > 0 else 0
|
|
eta = (total_html - i - 1) / rate if rate > 0 else 0
|
|
line = f'\r [{bar}] {i+1}/{total_html} ({pct}%) | tr:{translated_count} | {elapsed:.0f}s | ETA:{eta:.0f}s'
|
|
sys.stdout.write(line)
|
|
sys.stdout.flush()
|
|
|
|
except Exception as e:
|
|
sys.stdout.write('\n')
|
|
sys.stdout.flush()
|
|
log(f' WARN {rel}: {e}')
|
|
|
|
sys.stdout.write('\n')
|
|
sys.stdout.flush()
|
|
|
|
log(f' Metadata...')
|
|
opf_files = glob.glob(os.path.join(tmpdir, '**', '*.opf'), recursive=True)
|
|
for opf in opf_files:
|
|
try:
|
|
with open(opf, 'r', encoding='utf-8') as f:
|
|
content = f.read()
|
|
soup = BeautifulSoup(content, 'xml')
|
|
for tag_name, attr in [('dc:language', None), ('package', 'xml:lang'), ('package', 'lang')]:
|
|
elem = soup.find(tag_name) if attr is None else soup.find(tag_name.split(':')[0])
|
|
if elem:
|
|
if attr:
|
|
elem[attr] = 'id'
|
|
else:
|
|
elem.string = 'id'
|
|
dc_title = soup.find('dc:title')
|
|
if dc_title and '(Bahasa Indonesia)' not in dc_title.text:
|
|
dc_title.string = dc_title.text.strip() + ' (Bahasa Indonesia)'
|
|
with open(opf, 'w', encoding='utf-8') as f:
|
|
f.write(str(soup))
|
|
except Exception:
|
|
pass
|
|
|
|
log(f' Building EPUB...')
|
|
os.makedirs(os.path.dirname(output_path), exist_ok=True)
|
|
with zipfile.ZipFile(output_path, 'w', zipfile.ZIP_DEFLATED) as z:
|
|
z.writestr('mimetype', 'application/epub+zip', zipfile.ZIP_STORED)
|
|
for root, dirs, files in os.walk(tmpdir):
|
|
for fn in files:
|
|
full = os.path.join(root, fn)
|
|
arcname = os.path.relpath(full, tmpdir)
|
|
if arcname == 'mimetype':
|
|
continue
|
|
z.write(full, arcname, zipfile.ZIP_DEFLATED)
|
|
|
|
return translated_count, total_html, batch_total
|
|
|
|
finally:
|
|
shutil.rmtree(tmpdir, ignore_errors=True)
|
|
|
|
|
|
def load_state():
|
|
if os.path.exists(STATE_FILE):
|
|
with open(STATE_FILE) as f:
|
|
return json.load(f)
|
|
return {}
|
|
|
|
|
|
def save_state(state):
|
|
with open(STATE_FILE, 'w') as f:
|
|
json.dump(state, f)
|
|
|
|
|
|
def main():
|
|
import argparse
|
|
parser = argparse.ArgumentParser(description='Translate EPUBs EN->ID')
|
|
parser.add_argument('--force', action='store_true')
|
|
parser.add_argument('--dry-run', action='store_true')
|
|
parser.add_argument('--proxy', action='append')
|
|
parser.add_argument('--proxy-user')
|
|
parser.add_argument('--proxy-pass')
|
|
args = parser.parse_args()
|
|
|
|
global PROXIES, PROXY_USER, PROXY_PASS
|
|
if args.proxy:
|
|
PROXIES = args.proxy
|
|
PROXY_USER = args.proxy_user or ''
|
|
PROXY_PASS = args.proxy_pass or ''
|
|
|
|
state = load_state()
|
|
os.makedirs(TRANSLATED_DIR, exist_ok=True)
|
|
|
|
if not os.path.isdir(SOURCE_DIR):
|
|
log(f"Source directory not found: {SOURCE_DIR}")
|
|
return
|
|
|
|
all_epubs = sorted(glob.glob(os.path.join(SOURCE_DIR, '*.epub')))
|
|
existing = set(os.path.basename(tf) for tf in glob.glob(os.path.join(TRANSLATED_DIR, '*.epub')))
|
|
|
|
todo = []
|
|
for epub in all_epubs:
|
|
bn = os.path.basename(epub)
|
|
if bn.endswith('_-_Bahasa_Indonesia.epub'):
|
|
continue
|
|
if epub_output_name(epub) in existing and not args.force:
|
|
continue
|
|
if bn in state.get('done', {}) and not args.force:
|
|
continue
|
|
todo.append(epub)
|
|
|
|
if args.dry_run:
|
|
log(f"Would translate {len(todo)} EPUB(s):")
|
|
for e in todo:
|
|
log(f" {os.path.basename(e)} -> {epub_output_name(e)}")
|
|
return
|
|
|
|
if not todo:
|
|
log("No new EPUBs!")
|
|
return
|
|
|
|
total = len(todo)
|
|
t0 = time.time()
|
|
log(f"To translate: {total} EPUB(s)")
|
|
log("=" * 60)
|
|
|
|
for i, epub_path in enumerate(todo):
|
|
epub_name = os.path.basename(epub_path)
|
|
output_name = epub_output_name(epub_path)
|
|
output_path = os.path.join(TRANSLATED_DIR, output_name)
|
|
|
|
elapsed = time.time() - t0
|
|
rate = elapsed / max(i, 1) if i else 0
|
|
eta = rate * (total - i)
|
|
fs = os.path.getsize(epub_path) / 1024
|
|
|
|
log(f'[{i+1}/{total}] {epub_name} ({fs:.0f}KB)')
|
|
log(f' -> {output_name}')
|
|
|
|
try:
|
|
t1 = time.time()
|
|
tc, th, bt = process_epub(epub_path, output_path)
|
|
fe = time.time() - t1
|
|
log(f' DONE: {tc}/{th} files, {bt} batches, {fe:.0f}s ({os.path.getsize(output_path)/1024:.0f}KB)')
|
|
state.setdefault('done', {})[epub_name] = {
|
|
'output': output_name, 'time': round(fe),
|
|
'files': tc, 'batches': bt
|
|
}
|
|
save_state(state)
|
|
except Exception as e:
|
|
log(f' ERROR: {e}')
|
|
import traceback
|
|
traceback.print_exc()
|
|
|
|
log(f' [{i+1}/{total}] ETA: {eta/60:.0f}m\n')
|
|
|
|
log("=" * 60)
|
|
log(f"Done: {len(state.get('done',{}))} EPUBs in {(time.time()-t0)/60:.1f} min")
|
|
log(f"Output: {TRANSLATED_DIR}/")
|
|
|
|
|
|
if __name__ == '__main__':
|
|
main()
|