First Commit
This commit is contained in:
commit
df500733db
3 files changed
+368
No files matched your search
+18
@@ -0,0 +1,18 @@
|
||||
# Python
|
||||
__pycache__/
|
||||
*.py[cod]
|
||||
*.so
|
||||
*.egg-info/
|
||||
dist/
|
||||
build/
|
||||
*.egg
|
||||
venv/
|
||||
.env
|
||||
|
||||
# EPUB files
|
||||
*.epub
|
||||
original/
|
||||
translated/
|
||||
|
||||
# State files
|
||||
translate_state.json
|
||||
@@ -0,0 +1,4 @@
|
||||
beautifulsoup4>=4.12
|
||||
lxml>=5.0
|
||||
requests>=2.30
|
||||
urllib3>=2.0
|
||||
@@ -0,0 +1,346 @@
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
EPUB Translator EN -> ID. Uses proxies with connection pooling.
|
||||
Verbose progress. Saves to translated/ folder.
|
||||
"""
|
||||
|
||||
import socket
|
||||
_orig = socket.getaddrinfo
|
||||
def _ipv4(h, p, f=0, t=0, pr=0, fl=0):
|
||||
return _orig(h, p, socket.AF_INET, t, pr, fl)
|
||||
socket.getaddrinfo = _ipv4
|
||||
|
||||
import os, sys, time, glob, json, random
|
||||
import zipfile, tempfile, shutil
|
||||
from bs4 import BeautifulSoup
|
||||
from deep_translator import GoogleTranslator
|
||||
|
||||
# ── Proxy configuration ───────────────────────────────────────────
|
||||
PROXIES = [
|
||||
'103.80.237.30:8888',
|
||||
'103.80.237.27:8888',
|
||||
'103.80.237.163:8888',
|
||||
'103.185.47.51:8888',
|
||||
'103.185.47.53:8888',
|
||||
'103.185.47.54:8888',
|
||||
]
|
||||
PROXY_USER = 'dspaceme'
|
||||
PROXY_PASS = 'anakmanis'
|
||||
|
||||
SEP = '[[SPLIT]]'
|
||||
MAX_CHUNK = 4000
|
||||
DELAY = 0.3
|
||||
|
||||
BASE_DIR = os.path.dirname(os.path.abspath(__file__))
|
||||
SOURCE_DIR = os.path.join(BASE_DIR, 'original')
|
||||
TRANSLATED_DIR = os.path.join(BASE_DIR, 'translated')
|
||||
STATE_FILE = os.path.join(BASE_DIR, 'translate_state.json')
|
||||
|
||||
def log(msg):
|
||||
print(msg, flush=True)
|
||||
|
||||
|
||||
def translate_text(text):
|
||||
"""Translate using deep_translator."""
|
||||
try:
|
||||
translator = GoogleTranslator(source='en', target='id')
|
||||
result = translator.translate(text)
|
||||
return result if result else None
|
||||
except Exception:
|
||||
return None
|
||||
|
||||
|
||||
def translate_rotated(text):
|
||||
"""Try translation. If 429, retry after delay."""
|
||||
for attempt in range(5):
|
||||
result = translate_text(text)
|
||||
if result:
|
||||
return result
|
||||
time.sleep(2 ** attempt)
|
||||
return None
|
||||
|
||||
|
||||
def translate_batch(texts):
|
||||
if not texts:
|
||||
return texts
|
||||
non_empty = [t for t in texts if t.strip()]
|
||||
if not non_empty:
|
||||
return texts
|
||||
|
||||
combined = SEP.join(non_empty)
|
||||
result = translate_rotated(combined)
|
||||
if result:
|
||||
parts = result.split(SEP)
|
||||
if len(parts) == len(non_empty):
|
||||
pi = 0
|
||||
out = []
|
||||
for t in texts:
|
||||
if t.strip():
|
||||
out.append(parts[pi].strip())
|
||||
pi += 1
|
||||
else:
|
||||
out.append('')
|
||||
return out
|
||||
|
||||
# Fallback: individual
|
||||
out = []
|
||||
for t in texts:
|
||||
if not t.strip():
|
||||
out.append('')
|
||||
continue
|
||||
r = translate_rotated(t)
|
||||
out.append(r.strip() if r else t)
|
||||
time.sleep(0.1)
|
||||
return out
|
||||
|
||||
|
||||
def epub_output_name(epub_path):
|
||||
name = os.path.basename(epub_path)
|
||||
name = name.replace('.epub', '_-_Bahasa_Indonesia.epub')
|
||||
return name
|
||||
|
||||
|
||||
def process_epub(epub_path, output_path, state_info=None):
|
||||
epub_name = os.path.basename(epub_path)
|
||||
tmpdir = tempfile.mkdtemp(prefix='epub_trans_')
|
||||
t_start = time.time()
|
||||
|
||||
try:
|
||||
log(f' Extracting...')
|
||||
with zipfile.ZipFile(epub_path, 'r') as z:
|
||||
z.extractall(tmpdir)
|
||||
|
||||
html_files = []
|
||||
for root, dirs, files in os.walk(tmpdir):
|
||||
for fn in files:
|
||||
if fn.endswith(('.html', '.xhtml', '.htm')):
|
||||
html_files.append(os.path.join(root, fn))
|
||||
|
||||
total_html = len(html_files)
|
||||
log(f' {total_html} HTML files')
|
||||
translated_count = 0
|
||||
batch_total = 0
|
||||
|
||||
for i, fp in enumerate(html_files):
|
||||
rel = os.path.relpath(fp, tmpdir)
|
||||
try:
|
||||
with open(fp, 'r', encoding='utf-8') as f:
|
||||
html = f.read()
|
||||
soup = BeautifulSoup(html, 'xml')
|
||||
body = soup.find('body')
|
||||
if not body:
|
||||
continue
|
||||
|
||||
elems = []
|
||||
for tag in ('p', 'a', 'span', 'h1', 'h2', 'h3', 'h4', 'h5', 'h6'):
|
||||
for e in body.find_all(tag):
|
||||
t = e.get_text().strip()
|
||||
if not t or len(t) <= 2:
|
||||
continue
|
||||
kids = set(getattr(c, 'name', None) for c in e.children if hasattr(c, 'name'))
|
||||
if kids - {None, 'span', 'a', 'b', 'i', 'em', 'strong'}:
|
||||
continue
|
||||
elems.append(e)
|
||||
|
||||
if not elems:
|
||||
continue
|
||||
|
||||
all_texts = [e.get_text().strip() for e in elems]
|
||||
|
||||
batches = []
|
||||
cur, curlen = [], 0
|
||||
for t in all_texts:
|
||||
t2 = t.strip()
|
||||
if not t2:
|
||||
cur.append('')
|
||||
continue
|
||||
if curlen + len(t2) + len(SEP) > MAX_CHUNK and cur:
|
||||
batches.append(cur)
|
||||
cur, curlen = [], 0
|
||||
cur.append(t2)
|
||||
curlen += len(t2) + len(SEP)
|
||||
if cur:
|
||||
batches.append(cur)
|
||||
|
||||
results = [None] * len(all_texts)
|
||||
idx = 0
|
||||
for batch in batches:
|
||||
translated = translate_batch(batch)
|
||||
for j, trans in enumerate(translated):
|
||||
results[idx + j] = trans
|
||||
idx += len(batch)
|
||||
batch_total += 1
|
||||
time.sleep(DELAY)
|
||||
|
||||
changed = False
|
||||
for e, orig, trans in zip(elems, all_texts, results):
|
||||
trans = trans.strip() if trans else ''
|
||||
if trans and trans != orig:
|
||||
e.clear()
|
||||
e.string = trans
|
||||
changed = True
|
||||
|
||||
if changed:
|
||||
with open(fp, 'w', encoding='utf-8') as f:
|
||||
f.write(str(soup))
|
||||
translated_count += 1
|
||||
|
||||
pct = int((i + 1) * 100 / total_html)
|
||||
filled = int(30 * (i + 1) / total_html)
|
||||
bar = '#' * filled + '-' * (30 - filled)
|
||||
elapsed = time.time() - t_start
|
||||
rate = (i + 1) / elapsed if elapsed > 0 else 0
|
||||
eta = (total_html - i - 1) / rate if rate > 0 else 0
|
||||
line = f'\r [{bar}] {i+1}/{total_html} ({pct}%) | tr:{translated_count} | {elapsed:.0f}s | ETA:{eta:.0f}s'
|
||||
sys.stdout.write(line)
|
||||
sys.stdout.flush()
|
||||
|
||||
except Exception as e:
|
||||
sys.stdout.write('\n')
|
||||
sys.stdout.flush()
|
||||
log(f' WARN {rel}: {e}')
|
||||
|
||||
sys.stdout.write('\n')
|
||||
sys.stdout.flush()
|
||||
|
||||
log(f' Metadata...')
|
||||
opf_files = glob.glob(os.path.join(tmpdir, '**', '*.opf'), recursive=True)
|
||||
for opf in opf_files:
|
||||
try:
|
||||
with open(opf, 'r', encoding='utf-8') as f:
|
||||
content = f.read()
|
||||
soup = BeautifulSoup(content, 'xml')
|
||||
for tag_name, attr in [('dc:language', None), ('package', 'xml:lang'), ('package', 'lang')]:
|
||||
elem = soup.find(tag_name) if attr is None else soup.find(tag_name.split(':')[0])
|
||||
if elem:
|
||||
if attr:
|
||||
elem[attr] = 'id'
|
||||
else:
|
||||
elem.string = 'id'
|
||||
dc_title = soup.find('dc:title')
|
||||
if dc_title and '(Bahasa Indonesia)' not in dc_title.text:
|
||||
dc_title.string = dc_title.text.strip() + ' (Bahasa Indonesia)'
|
||||
with open(opf, 'w', encoding='utf-8') as f:
|
||||
f.write(str(soup))
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
log(f' Building EPUB...')
|
||||
os.makedirs(os.path.dirname(output_path), exist_ok=True)
|
||||
with zipfile.ZipFile(output_path, 'w', zipfile.ZIP_DEFLATED) as z:
|
||||
z.writestr('mimetype', 'application/epub+zip', zipfile.ZIP_STORED)
|
||||
for root, dirs, files in os.walk(tmpdir):
|
||||
for fn in files:
|
||||
full = os.path.join(root, fn)
|
||||
arcname = os.path.relpath(full, tmpdir)
|
||||
if arcname == 'mimetype':
|
||||
continue
|
||||
z.write(full, arcname, zipfile.ZIP_DEFLATED)
|
||||
|
||||
return translated_count, total_html, batch_total
|
||||
|
||||
finally:
|
||||
shutil.rmtree(tmpdir, ignore_errors=True)
|
||||
|
||||
|
||||
def load_state():
|
||||
if os.path.exists(STATE_FILE):
|
||||
with open(STATE_FILE) as f:
|
||||
return json.load(f)
|
||||
return {}
|
||||
|
||||
|
||||
def save_state(state):
|
||||
with open(STATE_FILE, 'w') as f:
|
||||
json.dump(state, f)
|
||||
|
||||
|
||||
def main():
|
||||
import argparse
|
||||
parser = argparse.ArgumentParser(description='Translate EPUBs EN->ID')
|
||||
parser.add_argument('--force', action='store_true')
|
||||
parser.add_argument('--dry-run', action='store_true')
|
||||
parser.add_argument('--proxy', action='append')
|
||||
parser.add_argument('--proxy-user')
|
||||
parser.add_argument('--proxy-pass')
|
||||
args = parser.parse_args()
|
||||
|
||||
global PROXIES, PROXY_USER, PROXY_PASS
|
||||
if args.proxy:
|
||||
PROXIES = args.proxy
|
||||
PROXY_USER = args.proxy_user or ''
|
||||
PROXY_PASS = args.proxy_pass or ''
|
||||
|
||||
state = load_state()
|
||||
os.makedirs(TRANSLATED_DIR, exist_ok=True)
|
||||
|
||||
if not os.path.isdir(SOURCE_DIR):
|
||||
log(f"Source directory not found: {SOURCE_DIR}")
|
||||
return
|
||||
|
||||
all_epubs = sorted(glob.glob(os.path.join(SOURCE_DIR, '*.epub')))
|
||||
existing = set(os.path.basename(tf) for tf in glob.glob(os.path.join(TRANSLATED_DIR, '*.epub')))
|
||||
|
||||
todo = []
|
||||
for epub in all_epubs:
|
||||
bn = os.path.basename(epub)
|
||||
if bn.endswith('_-_Bahasa_Indonesia.epub'):
|
||||
continue
|
||||
if epub_output_name(epub) in existing and not args.force:
|
||||
continue
|
||||
if bn in state.get('done', {}) and not args.force:
|
||||
continue
|
||||
todo.append(epub)
|
||||
|
||||
if args.dry_run:
|
||||
log(f"Would translate {len(todo)} EPUB(s):")
|
||||
for e in todo:
|
||||
log(f" {os.path.basename(e)} -> {epub_output_name(e)}")
|
||||
return
|
||||
|
||||
if not todo:
|
||||
log("No new EPUBs!")
|
||||
return
|
||||
|
||||
total = len(todo)
|
||||
t0 = time.time()
|
||||
log(f"To translate: {total} EPUB(s)")
|
||||
log("=" * 60)
|
||||
|
||||
for i, epub_path in enumerate(todo):
|
||||
epub_name = os.path.basename(epub_path)
|
||||
output_name = epub_output_name(epub_path)
|
||||
output_path = os.path.join(TRANSLATED_DIR, output_name)
|
||||
|
||||
elapsed = time.time() - t0
|
||||
rate = elapsed / max(i, 1) if i else 0
|
||||
eta = rate * (total - i)
|
||||
fs = os.path.getsize(epub_path) / 1024
|
||||
|
||||
log(f'[{i+1}/{total}] {epub_name} ({fs:.0f}KB)')
|
||||
log(f' -> {output_name}')
|
||||
|
||||
try:
|
||||
t1 = time.time()
|
||||
tc, th, bt = process_epub(epub_path, output_path)
|
||||
fe = time.time() - t1
|
||||
log(f' DONE: {tc}/{th} files, {bt} batches, {fe:.0f}s ({os.path.getsize(output_path)/1024:.0f}KB)')
|
||||
state.setdefault('done', {})[epub_name] = {
|
||||
'output': output_name, 'time': round(fe),
|
||||
'files': tc, 'batches': bt
|
||||
}
|
||||
save_state(state)
|
||||
except Exception as e:
|
||||
log(f' ERROR: {e}')
|
||||
import traceback
|
||||
traceback.print_exc()
|
||||
|
||||
log(f' [{i+1}/{total}] ETA: {eta/60:.0f}m\n')
|
||||
|
||||
log("=" * 60)
|
||||
log(f"Done: {len(state.get('done',{}))} EPUBs in {(time.time()-t0)/60:.1f} min")
|
||||
log(f"Output: {TRANSLATED_DIR}/")
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
main()
|
||||
Reference in new issue
Block a user