Update Commit
This commit is contained in:
1 parent
36e16ba9b0
commit
20e9be32db
3 files changed
+2630
-22
No files matched your search
+2588
-8
File diff suppressed because it is too large.
Load diff
@@ -1,4 +1,5 @@
|
||||
beautifulsoup4>=4.12
|
||||
deep-translator>=1.11
|
||||
lxml>=5.0
|
||||
requests>=2.30
|
||||
urllib3>=2.0
|
||||
+41
-14
@@ -10,7 +10,7 @@ def _ipv4(h, p, f=0, t=0, pr=0, fl=0):
|
||||
return _orig(h, p, socket.AF_INET, t, pr, fl)
|
||||
socket.getaddrinfo = _ipv4
|
||||
|
||||
import os, sys, time, glob, json, random
|
||||
import os, sys, time, glob, json, random, re
|
||||
import zipfile, tempfile, shutil
|
||||
from bs4 import BeautifulSoup
|
||||
from deep_translator import GoogleTranslator
|
||||
@@ -30,6 +30,7 @@ PROXY_PASS = 'anakmanis'
|
||||
SEP = '[[SPLIT]]'
|
||||
MAX_CHUNK = 4000
|
||||
DELAY = 0.3
|
||||
SENTENCE_RE = re.compile(r'(?<=[.!?])\s+')
|
||||
|
||||
BASE_DIR = os.path.dirname(os.path.abspath(__file__))
|
||||
SOURCE_DIR = os.path.join(BASE_DIR, 'original')
|
||||
@@ -147,31 +148,57 @@ def process_epub(epub_path, output_path, state_info=None):
|
||||
|
||||
all_texts = [e.get_text().strip() for e in elems]
|
||||
|
||||
# Split each element's text into sentences
|
||||
elem_sentences = [] # list of (elem_idx, [sentence1, sentence2, ...])
|
||||
for ei, t in enumerate(all_texts):
|
||||
if not t:
|
||||
elem_sentences.append((ei, []))
|
||||
elif len(t) <= MAX_CHUNK:
|
||||
elem_sentences.append((ei, [t]))
|
||||
else:
|
||||
parts = SENTENCE_RE.split(t)
|
||||
elem_sentences.append((ei, parts))
|
||||
|
||||
# Flatten sentences into a list, tracking which element each belongs to
|
||||
flat = [] # (elem_idx, sentence_idx_in_elem, text)
|
||||
for ei, sentences in elem_sentences:
|
||||
for si, s in enumerate(sentences):
|
||||
flat.append((ei, si, s.strip()))
|
||||
|
||||
# Build batches of sentences up to MAX_CHUNK
|
||||
batches = []
|
||||
cur, curlen = [], 0
|
||||
for t in all_texts:
|
||||
t2 = t.strip()
|
||||
if not t2:
|
||||
cur.append('')
|
||||
for ei, si, s in flat:
|
||||
if not s:
|
||||
continue
|
||||
if curlen + len(t2) + len(SEP) > MAX_CHUNK and cur:
|
||||
if curlen + len(s) + len(SEP) > MAX_CHUNK and cur:
|
||||
batches.append(cur)
|
||||
cur, curlen = [], 0
|
||||
cur.append(t2)
|
||||
curlen += len(t2) + len(SEP)
|
||||
cur.append((ei, si, s))
|
||||
curlen += len(s) + len(SEP)
|
||||
if cur:
|
||||
batches.append(cur)
|
||||
|
||||
results = [None] * len(all_texts)
|
||||
idx = 0
|
||||
# Translate sentence batches and reassemble per element
|
||||
elem_translated_sentences = {}
|
||||
for batch in batches:
|
||||
translated = translate_batch(batch)
|
||||
for j, trans in enumerate(translated):
|
||||
results[idx + j] = trans
|
||||
idx += len(batch)
|
||||
texts = [s for _, _, s in batch]
|
||||
translated = translate_batch(texts)
|
||||
for (ei, si, _), trans in zip(batch, translated):
|
||||
elem_translated_sentences.setdefault(ei, {})[si] = trans
|
||||
batch_total += 1
|
||||
time.sleep(DELAY)
|
||||
|
||||
# Reconstruct per-element text
|
||||
results = []
|
||||
for ei, (_, sentences) in enumerate(elem_sentences):
|
||||
if ei in elem_translated_sentences and sentences:
|
||||
trans_dict = elem_translated_sentences[ei]
|
||||
parts = [trans_dict.get(si, s) for si, s in enumerate(sentences)]
|
||||
results.append(' '.join(parts))
|
||||
else:
|
||||
results.append(all_texts[ei])
|
||||
|
||||
changed = False
|
||||
for e, orig, trans in zip(elems, all_texts, results):
|
||||
trans = trans.strip() if trans else ''
|
||||
|
||||
Reference in new issue
Block a user