155 lines
6.6 KiB
Python
155 lines
6.6 KiB
Python
"""
|
|
双语 EPUB 构建器模块 - 安全的EPUB构建 (Manifest 兼容版)
|
|
"""
|
|
|
|
from ebooklib import epub
|
|
import ebooklib
|
|
from bs4 import BeautifulSoup
|
|
from typing import Dict, List
|
|
from pathlib import Path
|
|
from loguru import logger
|
|
import uuid
|
|
|
|
|
|
class BilingualEPUBBuilder:
|
|
"""双语 EPUB 构建器"""
|
|
|
|
def __init__(self, original_book, config: Dict):
|
|
self.original_book = original_book
|
|
self.config = config
|
|
self.output_config = config['output']
|
|
|
|
def create_bilingual_epub_with_mapping(self, translation_map: Dict[str, str],
|
|
paragraph_map: Dict[str, Dict],
|
|
output_path: str) -> str:
|
|
"""
|
|
创建双语 EPUB。使用 ordered_ids 确保与 Manifest 严格一致。
|
|
"""
|
|
try:
|
|
new_book = epub.EpubBook()
|
|
self._copy_metadata(new_book)
|
|
new_book.toc = self.original_book.toc
|
|
|
|
# 准备每个文件的有序ID列表
|
|
file_ordered_ids = {}
|
|
sorted_pids = sorted(paragraph_map.keys(), key=lambda x: int(x.split('_')[1]))
|
|
for pid in sorted_pids:
|
|
info = paragraph_map[pid]
|
|
fname = info['file_name']
|
|
if fname not in file_ordered_ids:
|
|
file_ordered_ids[fname] = []
|
|
file_ordered_ids[fname].append(pid)
|
|
|
|
processed_item_ids = set()
|
|
item_map = {}
|
|
|
|
# 复制资源
|
|
for item in self.original_book.get_items():
|
|
if item.get_type() != ebooklib.ITEM_DOCUMENT:
|
|
if item.id not in processed_item_ids:
|
|
new_book.add_item(item)
|
|
processed_item_ids.add(item.id)
|
|
item_map[item.id] = item
|
|
|
|
# 重建 Spine
|
|
new_spine = []
|
|
for spine_id, linear in self.original_book.spine:
|
|
item = self.original_book.get_item_with_id(spine_id)
|
|
if not item: continue
|
|
|
|
if item.get_type() == ebooklib.ITEM_DOCUMENT:
|
|
file_name = item.get_name()
|
|
if file_name in file_ordered_ids:
|
|
new_item = self._create_bilingual_document(
|
|
item, file_ordered_ids[file_name], translation_map
|
|
)
|
|
new_item.id = item.id
|
|
else:
|
|
new_item = item
|
|
|
|
if new_item.id not in processed_item_ids:
|
|
new_book.add_item(new_item)
|
|
processed_item_ids.add(new_item.id)
|
|
new_spine.append(new_item)
|
|
else:
|
|
if item.id in item_map:
|
|
new_spine.append(item_map[item.id])
|
|
|
|
new_book.spine = new_spine
|
|
new_book.add_item(epub.EpubNcx())
|
|
new_book.add_item(epub.EpubNav())
|
|
|
|
output_file = self._generate_output_filename(output_path)
|
|
epub.write_epub(output_file, new_book, {})
|
|
return output_file
|
|
|
|
except Exception as e:
|
|
logger.error(f"创建双语 EPUB 失败: {e}", exc_info=True)
|
|
raise
|
|
|
|
def _copy_metadata(self, new_book):
|
|
try:
|
|
for namespace, meta_dict in self.original_book.metadata.items():
|
|
for name, values in meta_dict.items():
|
|
for value, other in values:
|
|
if name and hasattr(name, 'lower') and name.lower() == 'identifier': continue
|
|
new_book.add_metadata(namespace, name, value, other)
|
|
new_book.add_metadata('DC', 'language', 'zh-CN')
|
|
new_book.set_identifier(f"bilingual-{uuid.uuid4().hex[:12]}")
|
|
|
|
cover_id_meta = self.original_book.get_metadata('OPF', 'cover')
|
|
if cover_id_meta:
|
|
cover_item = self.original_book.get_item_with_id(cover_id_meta[0][0])
|
|
if cover_item:
|
|
new_book.add_item(cover_item)
|
|
new_book.set_cover(cover_item.get_name(), cover_item.get_content())
|
|
except Exception as e:
|
|
logger.error(f"元数据复制出错: {e}")
|
|
|
|
def _create_bilingual_document(self, original_item, ordered_ids: list, translation_map: dict):
|
|
try:
|
|
from .text_processor import TextProcessor
|
|
soup = BeautifulSoup(original_item.get_content().decode('utf-8'), 'html.parser')
|
|
self._add_style_link(soup)
|
|
|
|
# 使用与 TextProcessor 相同的过滤逻辑获取元素
|
|
text_elements = TextProcessor.get_valid_text_elements(soup)
|
|
|
|
current_para_index = 0
|
|
for element in text_elements:
|
|
if TextProcessor.is_navigation_element(element): continue
|
|
if not TextProcessor.clean_element_text(element): continue
|
|
|
|
if current_para_index < len(ordered_ids):
|
|
target_id = ordered_ids[current_para_index]
|
|
translation = translation_map.get(target_id)
|
|
if translation:
|
|
self._insert_translation(element, translation, soup)
|
|
current_para_index += 1
|
|
|
|
new_item = epub.EpubHtml(title=original_item.title, file_name=original_item.get_name(), lang='zh-CN')
|
|
new_item.set_content(str(soup).encode('utf-8'))
|
|
return new_item
|
|
except Exception as e:
|
|
logger.error(f"创建双语文档失败 {original_item.get_name()}: {e}")
|
|
return original_item
|
|
|
|
def _add_style_link(self, soup):
|
|
head = soup.find('head')
|
|
if head and not head.find('link', href='style/bilingual.css'):
|
|
head.append(soup.new_tag('link', rel='stylesheet', type='text/css', href='style/bilingual.css'))
|
|
|
|
def _insert_translation(self, element, translation: str, soup):
|
|
try:
|
|
translation_p = soup.new_tag('p')
|
|
translation_p.string = translation
|
|
translation_p['class'] = ['translation-text', 'chinese']
|
|
element.insert_after(translation_p)
|
|
except: pass
|
|
|
|
def _generate_output_filename(self, output_path: str) -> str:
|
|
from .utils import sanitize_filename
|
|
title = self.original_book.get_metadata('DC', 'title')
|
|
clean_title = sanitize_filename(title[0][0]) if title else "bilingual_book"
|
|
Path(output_path).mkdir(parents=True, exist_ok=True)
|
|
return str(Path(output_path) / f"{clean_title}_bilingual.epub") |