Update Commit

This commit is contained in:
proitlab committed 2026-07-24 16:02:10 +07:00
1 parent 36e16ba9b0
commit 20e9be32db
3 files changed
+2630 -22

No files matched your search

+2588 -8
View File
File diff suppressed because it is too large. Load diff
+1
View File
@@ -1,4 +1,5 @@
beautifulsoup4>=4.12
deep-translator>=1.11
lxml>=5.0
requests>=2.30
urllib3>=2.0
+41 -14
View File
@@ -10,7 +10,7 @@ def _ipv4(h, p, f=0, t=0, pr=0, fl=0):
return _orig(h, p, socket.AF_INET, t, pr, fl)
socket.getaddrinfo = _ipv4
import os, sys, time, glob, json, random
import os, sys, time, glob, json, random, re
import zipfile, tempfile, shutil
from bs4 import BeautifulSoup
from deep_translator import GoogleTranslator
@@ -30,6 +30,7 @@ PROXY_PASS = 'anakmanis'
SEP = '[[SPLIT]]'
MAX_CHUNK = 4000
DELAY = 0.3
SENTENCE_RE = re.compile(r'(?<=[.!?])\s+')
BASE_DIR = os.path.dirname(os.path.abspath(__file__))
SOURCE_DIR = os.path.join(BASE_DIR, 'original')
@@ -147,31 +148,57 @@ def process_epub(epub_path, output_path, state_info=None):
all_texts = [e.get_text().strip() for e in elems]
# Split each element's text into sentences
elem_sentences = [] # list of (elem_idx, [sentence1, sentence2, ...])
for ei, t in enumerate(all_texts):
if not t:
elem_sentences.append((ei, []))
elif len(t) <= MAX_CHUNK:
elem_sentences.append((ei, [t]))
else:
parts = SENTENCE_RE.split(t)
elem_sentences.append((ei, parts))
# Flatten sentences into a list, tracking which element each belongs to
flat = [] # (elem_idx, sentence_idx_in_elem, text)
for ei, sentences in elem_sentences:
for si, s in enumerate(sentences):
flat.append((ei, si, s.strip()))
# Build batches of sentences up to MAX_CHUNK
batches = []
cur, curlen = [], 0
for t in all_texts:
t2 = t.strip()
if not t2:
cur.append('')
for ei, si, s in flat:
if not s:
continue
if curlen + len(t2) + len(SEP) > MAX_CHUNK and cur:
if curlen + len(s) + len(SEP) > MAX_CHUNK and cur:
batches.append(cur)
cur, curlen = [], 0
cur.append(t2)
curlen += len(t2) + len(SEP)
cur.append((ei, si, s))
curlen += len(s) + len(SEP)
if cur:
batches.append(cur)
results = [None] * len(all_texts)
idx = 0
# Translate sentence batches and reassemble per element
elem_translated_sentences = {}
for batch in batches:
translated = translate_batch(batch)
for j, trans in enumerate(translated):
results[idx + j] = trans
idx += len(batch)
texts = [s for _, _, s in batch]
translated = translate_batch(texts)
for (ei, si, _), trans in zip(batch, translated):
elem_translated_sentences.setdefault(ei, {})[si] = trans
batch_total += 1
time.sleep(DELAY)
# Reconstruct per-element text
results = []
for ei, (_, sentences) in enumerate(elem_sentences):
if ei in elem_translated_sentences and sentences:
trans_dict = elem_translated_sentences[ei]
parts = [trans_dict.get(si, s) for si, s in enumerate(sentences)]
results.append(' '.join(parts))
else:
results.append(all_texts[ei])
changed = False
for e, orig, trans in zip(elems, all_texts, results):
trans = trans.strip() if trans else ''