mirror of
https://git.victorphan.net/basketballcantho/book_translator.git
synced 2026-08-05 06:33:11 +07:00
665 lines
24 KiB
Python
665 lines
24 KiB
Python
#!/usr/bin/env python3
|
|
"""
|
|
Bilingual EPUB Translator (English → English/Vietnamese)
|
|
=========================================================
|
|
Translates an English .epub into a bilingual English-Vietnamese .epub
|
|
by loading the LLM directly in Python via HuggingFace Transformers.
|
|
|
|
NO external server (Ollama, vLLM, etc.) is needed — the model runs
|
|
inside this process. It is downloaded once from HuggingFace Hub and
|
|
cached at ~/.cache/huggingface for all future runs.
|
|
|
|
Features:
|
|
- Direct model loading (no server dependency)
|
|
- Chunk-based translation (configurable paragraphs per chunk)
|
|
- Checkpoint/resume so you can restart without losing progress
|
|
- tqdm progress bars for chapter and chunk tracking
|
|
- CSS-styled bilingual output with vocabulary boxes
|
|
|
|
Usage:
|
|
python translate_epub.py input.epub
|
|
python translate_epub.py input.epub -o output.epub --chunk-size 3
|
|
"""
|
|
|
|
import argparse
|
|
import copy
|
|
import hashlib
|
|
import os
|
|
import re
|
|
import sys
|
|
import time
|
|
from pathlib import Path
|
|
|
|
import ebooklib
|
|
from ebooklib import epub
|
|
from bs4 import BeautifulSoup, NavigableString, Tag
|
|
from tqdm import tqdm
|
|
|
|
# Lazy-imported at model load time:
|
|
# torch, transformers
|
|
|
|
# ──────────────────────────────────────────────────────────────
|
|
# Configuration Defaults
|
|
# ──────────────────────────────────────────────────────────────
|
|
import os
|
|
os.environ["HF_TOKEN"] = "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"
|
|
|
|
DEFAULT_MODEL_ID = "Qwen/Qwen2.5-72B-Instruct"
|
|
CHUNK_SIZE = 5 # paragraphs per inference call
|
|
MAX_NEW_TOKENS = 8192 # max tokens the model generates per chunk
|
|
|
|
# ──────────────────────────────────────────────────────────────
|
|
# Translation Prompt
|
|
# ──────────────────────────────────────────────────────────────
|
|
SYSTEM_PROMPT = (
|
|
"You are creating a bilingual English-Vietnamese book for a student "
|
|
"learning English. I will provide a section of the book. Format the "
|
|
"output strictly using HTML tags. For EVERY paragraph, output the "
|
|
"English first, then the Vietnamese, like this: "
|
|
"<p class='english'>[Original English Paragraph]</p> "
|
|
"<p class='vietnamese'>[Vietnamese Translation]</p>. "
|
|
"Do not miss any paragraphs. At the end of the section, add a "
|
|
"vocabulary breakdown highlighting 3-5 difficult English words/idioms "
|
|
"with their Vietnamese explanations wrapped in <div class='vocab'> tags."
|
|
)
|
|
|
|
# ──────────────────────────────────────────────────────────────
|
|
# CSS Stylesheet for Bilingual Layout
|
|
# ──────────────────────────────────────────────────────────────
|
|
BILINGUAL_CSS = """\
|
|
/* Bilingual EN-VI Stylesheet */
|
|
body {
|
|
font-family: Georgia, "Times New Roman", serif;
|
|
line-height: 1.8;
|
|
margin: 1.2em;
|
|
color: #2c2c2c;
|
|
background-color: #fefefe;
|
|
}
|
|
|
|
h1, h2, h3, h4, h5, h6 {
|
|
color: #1a1a2e;
|
|
margin-top: 1.5em;
|
|
margin-bottom: 0.5em;
|
|
line-height: 1.3;
|
|
}
|
|
|
|
p.english {
|
|
font-weight: bold;
|
|
color: #1a1a2e;
|
|
margin-bottom: 0.15em;
|
|
margin-top: 0.8em;
|
|
font-size: 1em;
|
|
line-height: 1.7;
|
|
}
|
|
|
|
p.vietnamese {
|
|
font-style: italic;
|
|
color: #34495e;
|
|
margin-top: 0;
|
|
margin-bottom: 1em;
|
|
padding-left: 0.8em;
|
|
border-left: 3px solid #e74c3c;
|
|
font-size: 0.95em;
|
|
line-height: 1.6;
|
|
}
|
|
|
|
div.vocab {
|
|
background-color: #f9f3e3;
|
|
border: 1px solid #ddd;
|
|
border-left: 5px solid #e74c3c;
|
|
border-radius: 6px;
|
|
padding: 1em 1.2em;
|
|
margin: 2em 0;
|
|
font-size: 0.9em;
|
|
line-height: 1.6;
|
|
page-break-inside: avoid;
|
|
}
|
|
|
|
div.vocab h4,
|
|
div.vocab strong {
|
|
color: #c0392b;
|
|
margin-top: 0;
|
|
}
|
|
|
|
div.vocab ul {
|
|
padding-left: 1.2em;
|
|
}
|
|
|
|
div.vocab li {
|
|
margin-bottom: 0.4em;
|
|
}
|
|
|
|
hr {
|
|
border: none;
|
|
border-top: 1px solid #e0e0e0;
|
|
margin: 2em 0;
|
|
}
|
|
"""
|
|
|
|
|
|
# ──────────────────────────────────────────────────────────────
|
|
# Checkpoint System (for resuming multi-hour translations)
|
|
# ──────────────────────────────────────────────────────────────
|
|
def _checkpoint_dir(input_path: str) -> str:
|
|
"""Return a checkpoint directory path based on the input file."""
|
|
parent = os.path.dirname(os.path.abspath(input_path))
|
|
book_hash = hashlib.sha256(
|
|
os.path.basename(input_path).encode()
|
|
).hexdigest()[:12]
|
|
return os.path.join(parent, f".bilingual_cache_{book_hash}")
|
|
|
|
|
|
def _checkpoint_path(input_path: str, item_name: str) -> str:
|
|
safe = re.sub(r'[^\w\-.]', '_', item_name)
|
|
return os.path.join(_checkpoint_dir(input_path), f"{safe}.html")
|
|
|
|
|
|
def load_checkpoint(input_path: str, item_name: str) -> str | None:
|
|
"""Load previously translated content, or None if not cached."""
|
|
path = _checkpoint_path(input_path, item_name)
|
|
if os.path.exists(path):
|
|
with open(path, "r", encoding="utf-8") as f:
|
|
return f.read()
|
|
return None
|
|
|
|
|
|
def save_checkpoint(input_path: str, item_name: str, content: str) -> None:
|
|
"""Persist translated chapter HTML for resume capability."""
|
|
cdir = _checkpoint_dir(input_path)
|
|
os.makedirs(cdir, exist_ok=True)
|
|
path = _checkpoint_path(input_path, item_name)
|
|
with open(path, "w", encoding="utf-8") as f:
|
|
f.write(content)
|
|
|
|
|
|
def clear_checkpoints(input_path: str) -> None:
|
|
"""Remove all checkpoint files for a given input."""
|
|
cdir = _checkpoint_dir(input_path)
|
|
if os.path.isdir(cdir):
|
|
import shutil
|
|
shutil.rmtree(cdir)
|
|
print(f" 🗑 Cleared checkpoint cache: {cdir}")
|
|
|
|
|
|
# ──────────────────────────────────────────────────────────────
|
|
# Model Loading (runs once, stays in memory)
|
|
# ──────────────────────────────────────────────────────────────
|
|
_model = None
|
|
_tokenizer = None
|
|
|
|
|
|
def load_model(model_id: str, load_in_8bit: bool = True) -> None:
|
|
"""
|
|
Load the model + tokenizer into memory. Called once at startup.
|
|
|
|
The model is downloaded from HuggingFace Hub on the first run and
|
|
cached at ~/.cache/huggingface — subsequent runs load from cache
|
|
with no re-download.
|
|
|
|
Args:
|
|
model_id: HuggingFace model identifier
|
|
(e.g. "Qwen/Qwen2.5-72B-Instruct")
|
|
load_in_8bit: If True, quantize to INT8 (~72 GB for 72B).
|
|
If False, load in FP32 (~288 GB) or FP16/BF16.
|
|
"""
|
|
global _model, _tokenizer
|
|
|
|
import torch
|
|
from transformers import AutoModelForCausalLM, AutoTokenizer
|
|
|
|
print(f"\n🔄 Loading model: {model_id}")
|
|
print(f" Quantization : {'INT8 (bitsandbytes)' if load_in_8bit else 'auto (float16/bfloat16)'}")
|
|
print(f" Cache dir : {os.environ.get('HF_HOME', '~/.cache/huggingface')}")
|
|
print(" This may take several minutes on first run (downloading weights)…\n")
|
|
|
|
t0 = time.time()
|
|
|
|
_tokenizer = AutoTokenizer.from_pretrained(
|
|
model_id,
|
|
trust_remote_code=True,
|
|
)
|
|
|
|
if load_in_8bit:
|
|
# INT8 quantization via bitsandbytes — fits ~72 GB for a 72B model
|
|
_model = AutoModelForCausalLM.from_pretrained(
|
|
model_id,
|
|
device_map="auto",
|
|
load_in_8bit=True,
|
|
trust_remote_code=True,
|
|
torch_dtype=torch.float16,
|
|
)
|
|
else:
|
|
# Auto dtype (usually bfloat16 on modern CPUs)
|
|
_model = AutoModelForCausalLM.from_pretrained(
|
|
model_id,
|
|
device_map="auto",
|
|
trust_remote_code=True,
|
|
torch_dtype="auto",
|
|
)
|
|
|
|
elapsed = time.time() - t0
|
|
print(f"✅ Model loaded in {elapsed:.0f}s and pinned in memory.\n")
|
|
|
|
|
|
# ──────────────────────────────────────────────────────────────
|
|
# Inference (replaces Ollama API calls)
|
|
# ──────────────────────────────────────────────────────────────
|
|
def clean_llm_response(text: str) -> str:
|
|
"""Strip markdown code fences and stray wrappers from LLM output."""
|
|
text = re.sub(r"^```(?:html|HTML)?\s*\n", "", text.strip())
|
|
text = re.sub(r"\n?```\s*$", "", text.strip())
|
|
return text.strip()
|
|
|
|
|
|
def generate_translation(
|
|
chunk_html: str,
|
|
chunk_idx: int,
|
|
total_chunks: int,
|
|
*,
|
|
max_new_tokens: int,
|
|
) -> str:
|
|
"""
|
|
Run the loaded model on a chunk of HTML and return bilingual HTML.
|
|
|
|
Uses the chat template so the model sees system + user messages.
|
|
"""
|
|
import torch
|
|
|
|
if _model is None or _tokenizer is None:
|
|
raise RuntimeError("Model not loaded. Call load_model() first.")
|
|
|
|
messages = [
|
|
{"role": "system", "content": SYSTEM_PROMPT},
|
|
{
|
|
"role": "user",
|
|
"content": f"Here is the section to translate:\n\n{chunk_html}",
|
|
},
|
|
]
|
|
|
|
try:
|
|
# Apply the model's chat template
|
|
input_text = _tokenizer.apply_chat_template(
|
|
messages,
|
|
tokenize=False,
|
|
add_generation_prompt=True,
|
|
)
|
|
inputs = _tokenizer(input_text, return_tensors="pt").to(_model.device)
|
|
|
|
with torch.no_grad():
|
|
output_ids = _model.generate(
|
|
**inputs,
|
|
max_new_tokens=max_new_tokens,
|
|
temperature=0.3,
|
|
do_sample=True,
|
|
top_p=0.9,
|
|
repetition_penalty=1.05,
|
|
)
|
|
|
|
# Decode only the newly generated tokens
|
|
generated_ids = output_ids[0][inputs["input_ids"].shape[1]:]
|
|
raw_output = _tokenizer.decode(generated_ids, skip_special_tokens=True)
|
|
return clean_llm_response(raw_output)
|
|
|
|
except Exception as exc:
|
|
tqdm.write(
|
|
f" ✗ Generation error on chunk {chunk_idx}/{total_chunks}: {exc}"
|
|
)
|
|
return f"<!-- TRANSLATION FAILED FOR THIS CHUNK -->\n{chunk_html}"
|
|
|
|
|
|
# ──────────────────────────────────────────────────────────────
|
|
# Chapter Translation Logic
|
|
# ──────────────────────────────────────────────────────────────
|
|
def chunk_list(lst: list, n: int):
|
|
"""Yield successive n-sized chunks from a list."""
|
|
for i in range(0, len(lst), n):
|
|
yield lst[i : i + n]
|
|
|
|
|
|
def translate_chapter(
|
|
html_content: str,
|
|
chapter_label: str,
|
|
*,
|
|
chunk_size: int,
|
|
max_new_tokens: int,
|
|
) -> str:
|
|
"""
|
|
Translate a chapter's HTML into bilingual (EN + VI) format.
|
|
|
|
Strategy:
|
|
1. Parse the chapter and find all <p> tags.
|
|
2. Group paragraphs into chunks of `chunk_size`.
|
|
3. For each chunk, run the model to get bilingual HTML.
|
|
4. Replace original paragraphs in-place with translated content,
|
|
preserving headings, images, and other non-paragraph elements
|
|
in their original positions.
|
|
"""
|
|
soup = BeautifulSoup(html_content, "html.parser")
|
|
paragraphs = soup.find_all("p")
|
|
|
|
if not paragraphs:
|
|
return html_content # nothing to translate
|
|
|
|
chunks = list(chunk_list(paragraphs, chunk_size))
|
|
total_chunks = len(chunks)
|
|
|
|
for i, para_chunk in enumerate(
|
|
tqdm(chunks, desc=" Chunks", unit="chunk", leave=False)
|
|
):
|
|
# Build the raw HTML for this chunk
|
|
chunk_html = "\n".join(str(p) for p in para_chunk)
|
|
|
|
translated_html = generate_translation(
|
|
chunk_html,
|
|
i + 1,
|
|
total_chunks,
|
|
max_new_tokens=max_new_tokens,
|
|
)
|
|
|
|
# Parse the LLM's HTML response
|
|
translated_soup = BeautifulSoup(translated_html, "html.parser")
|
|
|
|
# Insert all translated elements right before the first <p> in
|
|
# this chunk, then remove the original paragraphs.
|
|
anchor = para_chunk[0]
|
|
for new_element in list(translated_soup.children):
|
|
anchor.insert_before(copy.copy(new_element))
|
|
|
|
for p in para_chunk:
|
|
p.decompose()
|
|
|
|
return str(soup)
|
|
|
|
|
|
# ──────────────────────────────────────────────────────────────
|
|
# EPUB Processing Pipeline
|
|
# ──────────────────────────────────────────────────────────────
|
|
def process_epub(
|
|
input_path: str,
|
|
output_path: str,
|
|
*,
|
|
model_id: str,
|
|
chunk_size: int,
|
|
max_new_tokens: int,
|
|
resume: bool,
|
|
load_in_8bit: bool,
|
|
) -> None:
|
|
"""Read an English EPUB, translate it, and write a bilingual EPUB."""
|
|
|
|
# ── Load model into memory (once) ────────────────────────
|
|
load_model(model_id, load_in_8bit=load_in_8bit)
|
|
|
|
print(f"📖 Reading EPUB: {input_path}")
|
|
book = epub.read_epub(input_path, {"ignore_ncx": True})
|
|
|
|
# ── Create output book & copy metadata ──────────────────
|
|
out_book = epub.EpubBook()
|
|
|
|
identifiers = book.get_metadata("DC", "identifier")
|
|
base_id = identifiers[0][0] if identifiers else "unknown"
|
|
out_book.set_identifier(f"{base_id}-bilingual-en-vi")
|
|
|
|
titles = book.get_metadata("DC", "title")
|
|
original_title = titles[0][0] if titles else "Untitled"
|
|
out_book.set_title(f"{original_title} (Bilingual EN-VI)")
|
|
out_book.set_language("en")
|
|
out_book.add_metadata("DC", "language", "vi")
|
|
|
|
for author in book.get_metadata("DC", "creator"):
|
|
out_book.add_author(author[0])
|
|
|
|
# ── Add bilingual CSS ────────────────────────────────────
|
|
css_item = epub.EpubItem(
|
|
uid="bilingual_css",
|
|
file_name="style/bilingual.css",
|
|
media_type="text/css",
|
|
content=BILINGUAL_CSS.encode("utf-8"),
|
|
)
|
|
out_book.add_item(css_item)
|
|
|
|
# ── Gather chapter items ─────────────────────────────────
|
|
all_items = list(book.get_items())
|
|
doc_items = [
|
|
item
|
|
for item in all_items
|
|
if item.get_type() == ebooklib.ITEM_DOCUMENT
|
|
]
|
|
|
|
print(f"📚 Found {len(doc_items)} document items to process.\n")
|
|
|
|
spine_items: list = ["nav"]
|
|
toc_items: list = []
|
|
|
|
for idx, item in enumerate(
|
|
tqdm(doc_items, desc="📝 Translating chapters", unit="ch")
|
|
):
|
|
item_name = item.get_name() or f"chapter_{idx + 1}.xhtml"
|
|
chapter_label = item_name.rsplit("/", 1)[-1]
|
|
|
|
html_content = item.get_content().decode("utf-8", errors="replace")
|
|
soup_check = BeautifulSoup(html_content, "html.parser")
|
|
para_count = len(soup_check.find_all("p"))
|
|
|
|
if para_count > 0:
|
|
# ── Check for cached checkpoint ──────────────────
|
|
cached = load_checkpoint(input_path, item_name) if resume else None
|
|
|
|
if cached is not None:
|
|
tqdm.write(
|
|
f" ⏩ Resuming (cached): {chapter_label} "
|
|
f"({para_count} paragraphs)"
|
|
)
|
|
translated_html = cached
|
|
else:
|
|
tqdm.write(
|
|
f"\n 📄 Translating: {chapter_label} "
|
|
f"({para_count} paragraphs)"
|
|
)
|
|
translated_html = translate_chapter(
|
|
html_content,
|
|
chapter_label,
|
|
chunk_size=chunk_size,
|
|
max_new_tokens=max_new_tokens,
|
|
)
|
|
save_checkpoint(input_path, item_name, translated_html)
|
|
|
|
# Inject CSS link into the <head>
|
|
final_soup = BeautifulSoup(translated_html, "html.parser")
|
|
head = final_soup.find("head")
|
|
if head:
|
|
link_tag = final_soup.new_tag(
|
|
"link",
|
|
rel="stylesheet",
|
|
type="text/css",
|
|
href="../style/bilingual.css",
|
|
)
|
|
head.append(link_tag)
|
|
|
|
new_item = epub.EpubHtml(
|
|
title=chapter_label,
|
|
file_name=item_name,
|
|
lang="en",
|
|
)
|
|
new_item.set_content(str(final_soup).encode("utf-8"))
|
|
new_item.add_item(css_item)
|
|
else:
|
|
# Non-content item (cover page, TOC, etc.) — copy as-is
|
|
new_item = epub.EpubHtml(
|
|
title=chapter_label,
|
|
file_name=item_name,
|
|
lang="en",
|
|
)
|
|
new_item.set_content(item.get_content())
|
|
|
|
out_book.add_item(new_item)
|
|
spine_items.append(new_item)
|
|
toc_items.append(new_item)
|
|
|
|
# ── Copy non-document items (images, fonts, etc.) ────────
|
|
for item in all_items:
|
|
if item.get_type() == ebooklib.ITEM_DOCUMENT:
|
|
continue # already handled
|
|
|
|
if item.get_type() == ebooklib.ITEM_IMAGE:
|
|
img = epub.EpubImage()
|
|
img.file_name = item.get_name()
|
|
img.media_type = item.media_type
|
|
img.content = item.get_content()
|
|
out_book.add_item(img)
|
|
|
|
elif item.get_type() == ebooklib.ITEM_STYLE:
|
|
css = epub.EpubItem(
|
|
file_name=item.get_name(),
|
|
media_type=item.media_type,
|
|
content=item.get_content(),
|
|
)
|
|
out_book.add_item(css)
|
|
|
|
elif item.get_type() == ebooklib.ITEM_FONT:
|
|
font = epub.EpubItem(
|
|
file_name=item.get_name(),
|
|
media_type=item.media_type,
|
|
content=item.get_content(),
|
|
)
|
|
out_book.add_item(font)
|
|
|
|
elif item.get_type() not in (
|
|
ebooklib.ITEM_NAVIGATION,
|
|
ebooklib.ITEM_COVER,
|
|
):
|
|
try:
|
|
other = epub.EpubItem(
|
|
file_name=item.get_name(),
|
|
media_type=item.media_type,
|
|
content=item.get_content(),
|
|
)
|
|
out_book.add_item(other)
|
|
except Exception:
|
|
pass # skip items that can't be copied
|
|
|
|
# ── Build TOC, spine, and navigation ─────────────────────
|
|
out_book.toc = toc_items
|
|
out_book.spine = spine_items
|
|
out_book.add_item(epub.EpubNcx())
|
|
out_book.add_item(epub.EpubNav())
|
|
|
|
# ── Write output EPUB ────────────────────────────────────
|
|
print(f"\n💾 Writing bilingual EPUB: {output_path}")
|
|
epub.write_epub(output_path, out_book, {})
|
|
print("✅ Done! Your bilingual EPUB is ready.\n")
|
|
|
|
|
|
# ──────────────────────────────────────────────────────────────
|
|
# CLI Entry Point
|
|
# ──────────────────────────────────────────────────────────────
|
|
def main() -> None:
|
|
parser = argparse.ArgumentParser(
|
|
description=(
|
|
"Translate an English EPUB into a bilingual "
|
|
"English-Vietnamese EPUB. Loads the LLM directly "
|
|
"via HuggingFace Transformers (no server needed)."
|
|
),
|
|
formatter_class=argparse.RawDescriptionHelpFormatter,
|
|
epilog="""\
|
|
Examples:
|
|
python translate_epub.py book.epub
|
|
python translate_epub.py book.epub -o bilingual_book.epub
|
|
python translate_epub.py book.epub --chunk-size 3
|
|
python translate_epub.py book.epub --no-8bit # load full precision
|
|
python translate_epub.py book.epub --no-resume # ignore cache
|
|
""",
|
|
)
|
|
|
|
parser.add_argument(
|
|
"input",
|
|
help="Path to the input English .epub file",
|
|
)
|
|
parser.add_argument(
|
|
"-o", "--output",
|
|
help="Output .epub path (default: <input>_bilingual.epub)",
|
|
)
|
|
parser.add_argument(
|
|
"--model",
|
|
default=DEFAULT_MODEL_ID,
|
|
help=f"HuggingFace model ID (default: {DEFAULT_MODEL_ID})",
|
|
)
|
|
parser.add_argument(
|
|
"--chunk-size",
|
|
type=int,
|
|
default=CHUNK_SIZE,
|
|
help=f"Paragraphs per inference call (default: {CHUNK_SIZE})",
|
|
)
|
|
parser.add_argument(
|
|
"--max-new-tokens",
|
|
type=int,
|
|
default=MAX_NEW_TOKENS,
|
|
help=f"Max tokens per generation (default: {MAX_NEW_TOKENS})",
|
|
)
|
|
parser.add_argument(
|
|
"--no-8bit",
|
|
action="store_true",
|
|
help="Disable INT8 quantization (uses more RAM but may be higher quality)",
|
|
)
|
|
parser.add_argument(
|
|
"--no-resume",
|
|
action="store_true",
|
|
help="Ignore cached checkpoints and re-translate everything",
|
|
)
|
|
parser.add_argument(
|
|
"--clear-cache",
|
|
action="store_true",
|
|
help="Delete all cached checkpoints for this input file and exit",
|
|
)
|
|
|
|
args = parser.parse_args()
|
|
|
|
# ── Validate input ───────────────────────────────────────
|
|
input_path = Path(args.input).resolve()
|
|
if not input_path.exists():
|
|
print(f"❌ Error: File not found: {input_path}")
|
|
sys.exit(1)
|
|
if input_path.suffix.lower() != ".epub":
|
|
print(f"❌ Error: Expected .epub file, got: {input_path.suffix}")
|
|
sys.exit(1)
|
|
|
|
# ── Handle --clear-cache ─────────────────────────────────
|
|
if args.clear_cache:
|
|
clear_checkpoints(str(input_path))
|
|
sys.exit(0)
|
|
|
|
# ── Derive output path ───────────────────────────────────
|
|
if args.output:
|
|
output_path = args.output
|
|
else:
|
|
output_path = str(input_path.with_suffix("")) + "_bilingual.epub"
|
|
|
|
# ── Print banner ─────────────────────────────────────────
|
|
resume = not args.no_resume
|
|
load_in_8bit = not args.no_8bit
|
|
print("=" * 62)
|
|
print(" 📚 Bilingual EPUB Translator (EN → EN/VI)")
|
|
print(" Direct model loading — no server needed")
|
|
print("=" * 62)
|
|
print(f" Input : {input_path.name}")
|
|
print(f" Output : {Path(output_path).name}")
|
|
print(f" Model : {args.model}")
|
|
print(f" Quantization : {'INT8' if load_in_8bit else 'auto (full precision)'}")
|
|
print(f" Chunk size : {args.chunk_size} paragraphs")
|
|
print(f" Max tokens : {args.max_new_tokens} per chunk")
|
|
print(f" Resume : {'ON (cached chapters reused)' if resume else 'OFF'}")
|
|
print("=" * 62)
|
|
|
|
process_epub(
|
|
str(input_path),
|
|
output_path,
|
|
model_id=args.model,
|
|
chunk_size=args.chunk_size,
|
|
max_new_tokens=args.max_new_tokens,
|
|
resume=resume,
|
|
load_in_8bit=load_in_8bit,
|
|
)
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main()
|