Files
epub_bilingual_translator/archive/v0.13/src/preprocessing/epub_cleaner.py
T

185 lines
7.7 KiB
Python

import shutil
import uuid
import zipfile
from pathlib import Path
from typing import Dict, Set
import ebooklib
from bs4 import BeautifulSoup, Tag
from ebooklib import epub
from src.common.data_model import BookStructure, BookMetaData, ResourceItem
from src.common.exceptions import CleaningError
from src.common.utils import setup_logger, ensure_directory
logger = setup_logger("epub_cleaner")
class EpubCleaner:
def __init__(self, input_path: Path, work_dir: Path):
self.input_path = input_path
self.work_dir = work_dir
self.assets_dir = work_dir / "assets"
self.json_path = work_dir / "book_structure.json"
def clean(self) -> Path:
"""
Cleans the input EPUB and generates book_structure.json.
Returns the path to the JSON file.
"""
logger.info(f"Starting cleanup for {self.input_path}")
ensure_directory(self.work_dir)
ensure_directory(self.assets_dir)
try:
book = epub.read_epub(self.input_path)
# 1. Extract Metadata
metadata = self._extract_metadata(book)
# 2. Process Resources
resources = {}
# Use zipfile for binary extraction to avoid ebooklib's memory overhead/decoding issues
with zipfile.ZipFile(self.input_path, 'r') as zf:
# Map ebooklib items to zip entries isn't straightforward directly via name
# So we iterate ebooklib items and assume standard structure or handle content bytes
for item in book.get_items():
item_id = item.get_id()
file_name = item.get_name()
media_type = item.get_type() # ebooklib constant
if media_type == ebooklib.ITEM_DOCUMENT:
# Clean HTML
content_str = item.get_content().decode('utf-8')
cleaned_content = self._clean_html(content_str, file_name)
resources[item_id] = ResourceItem(
href=file_name,
media_type="application/xhtml+xml",
content=cleaned_content
)
elif media_type in (ebooklib.ITEM_IMAGE, ebooklib.ITEM_STYLE, ebooklib.ITEM_FONT, ebooklib.ITEM_COVER):
# Save asset
# Check if it is a cover
if media_type == ebooklib.ITEM_COVER:
metadata.cover_image_id = item_id
logger.info(f"Found cover image: {item_id} ({file_name})")
# Preserve directory structure to avoid collisions
asset_path = self.assets_dir / file_name
ensure_directory(asset_path.parent)
# Ebooklib might change filenames, safer to use item.get_content()
with open(asset_path, "wb") as f:
f.write(item.get_content())
resources[item_id] = ResourceItem(
href=file_name,
media_type=self._get_media_type_str(item),
file_path=str(asset_path.relative_to(self.work_dir))
)
elif media_type == ebooklib.ITEM_NAVIGATION:
# NCX or NAV document - preserve for TOC
content_bytes = item.get_content()
asset_path = self.assets_dir / file_name
ensure_directory(asset_path.parent)
with open(asset_path, "wb") as f:
f.write(content_bytes)
# Determine media type
if file_name.endswith('.ncx'):
mt = "application/x-dtbncx+xml"
else:
mt = "application/xhtml+xml"
resources[item_id] = ResourceItem(
href=file_name,
media_type=mt,
file_path=str(asset_path.relative_to(self.work_dir))
)
logger.debug(f"Preserved navigation: {file_name}")
else:
# Skip other items (scripts, etc.)
pass
# 3. Extract Spine
spine_ids = [item[0] for item in book.spine]
# 4. Construct Structure
structure = BookStructure(
metadata=metadata,
spine=spine_ids,
resources=resources
)
# 5. Serialize
with open(self.json_path, "w", encoding="utf-8") as f:
f.write(structure.model_dump_json(indent=2))
logger.info(f"Cleanup finished. Structure saved to {self.json_path}")
return self.json_path
except Exception as e:
logger.error(f"Cleaning failed: {e}")
raise CleaningError(f"Failed to clean EPUB: {e}") from e
def _extract_metadata(self, book: epub.EpubBook) -> BookMetaData:
title = book.get_metadata('DC', 'title')[0][0] if book.get_metadata('DC', 'title') else "Unknown"
author = book.get_metadata('DC', 'creator')[0][0] if book.get_metadata('DC', 'creator') else "Unknown"
lang = book.get_metadata('DC', 'language')[0][0] if book.get_metadata('DC', 'language') else "en"
ident = book.get_metadata('DC', 'identifier')[0][0] if book.get_metadata('DC', 'identifier') else ""
return BookMetaData(
title=str(title),
author=str(author),
language=str(lang),
identifier=str(ident)
)
def _get_media_type_str(self, item) -> str:
# Helper to map ebooklib type to mime string if needed
# ebooklib doesn't expose easy MIME string for all types directly on item object sometimes
if hasattr(item, 'media_type'):
return item.media_type
return "application/octet-stream"
def _clean_html(self, content: str, filename: str) -> str:
soup = BeautifulSoup(content, 'html.parser')
# 1. Provide IDs for structural/translatable elements
self._ensure_element_ids(soup)
# 2. Flatten divs (Disable to prevent style loss)
# self._flatten_divs(soup)
return str(soup)
def _ensure_element_ids(self, soup: BeautifulSoup):
"""
Injects UUIDs into p, h1-h6, li tags if they don't have an ID.
This provides the anchor for translation backfilling.
"""
targets = soup.find_all(['p', 'h1', 'h2', 'h3', 'h4', 'h5', 'h6', 'li'])
for tag in targets:
if not tag.has_attr('id'):
tag['id'] = f"uuid-{uuid.uuid4()}"
def _flatten_divs(self, soup: BeautifulSoup):
"""
Converts generic divs containing only inline text/styles to p tags.
Recursive strategies can be complex, sticking to simple heuristic from archive.
"""
inline_tags = {'span', 'em', 'strong', 'i', 'b', 'u', 'a', 'br', 'sub', 'sup', 'small'}
for div in list(soup.find_all('div')):
# If div has no block children, convert to p
has_block = any(
isinstance(c, Tag) and c.name not in inline_tags
for c in div.children
)
if not has_block:
div.name = 'p'