354 lines
11 KiB
Python
354 lines
11 KiB
Python
#!/usr/bin/env python3
|
|
"""
|
|
Chapter Translation Test Script
|
|
Translate a complete chapter to test the full pipeline.
|
|
|
|
Usage:
|
|
python scripts/translate_chapter.py --show-toc # 显示章节目录
|
|
python scripts/translate_chapter.py --chapter 5 # 翻译第5章
|
|
python scripts/translate_chapter.py --chapter 5 --test # 测试模式,只翻译前2个chunk
|
|
"""
|
|
|
|
import argparse
|
|
import asyncio
|
|
import json
|
|
import os
|
|
import sys
|
|
from pathlib import Path
|
|
|
|
sys.path.insert(0, str(Path(__file__).parent.parent))
|
|
|
|
from dotenv import load_dotenv
|
|
from src.manifest_manager import ManifestManager
|
|
from src.llm_client import LLMClient
|
|
from src.book_profiler import BookProfiler
|
|
from src.format_restorer import FormatRestorer
|
|
from src.data_model import ManifestEntry, BookStructure
|
|
|
|
load_dotenv()
|
|
|
|
# Configuration - Use unified .work directory
|
|
BOOK_NAME = "Empire of AI Dreams and Nightmares in Sam Altmans OpenAI (Karen Hao)"
|
|
WORK_DIR = Path(f".work/{BOOK_NAME}")
|
|
MANIFEST_PATH = WORK_DIR / "manifest.json"
|
|
STRUCTURE_PATH = WORK_DIR / "book_structure.json"
|
|
CHUNK_DIR = WORK_DIR / "chunks"
|
|
|
|
API_KEY = os.getenv("V3_API_KEY")
|
|
BASE_URL = "https://api.gpt.ge/v1"
|
|
MODEL = "gpt-4o-mini" # "gemini-3-flash-preview"
|
|
EXTRA_HEADERS = {"x-foo": "true"}
|
|
|
|
# Chunk config - around 5000 chars per chunk
|
|
CHUNK_SIZE_CHARS = 5000
|
|
|
|
|
|
def load_toc_from_structure() -> list:
|
|
"""Load TOC from book structure for readable chapter names."""
|
|
if not STRUCTURE_PATH.exists():
|
|
return []
|
|
|
|
try:
|
|
structure = BookStructure.load(STRUCTURE_PATH)
|
|
# Build TOC from spine order with chapter titles
|
|
toc = []
|
|
for i, item_id in enumerate(structure.spine, 1):
|
|
if item_id in structure.resources:
|
|
resource = structure.resources[item_id]
|
|
href = resource.href
|
|
|
|
# Try to extract title from content
|
|
title = extract_title_from_html(resource.content) if resource.content else None
|
|
|
|
toc.append({
|
|
'index': i,
|
|
'item_id': item_id,
|
|
'href': href,
|
|
'title': title or f"Chapter {i}"
|
|
})
|
|
return toc
|
|
except Exception as e:
|
|
print(f"警告: 无法加载书籍结构: {e}")
|
|
return []
|
|
|
|
|
|
def extract_title_from_html(html: str) -> str:
|
|
"""Extract title from HTML content."""
|
|
from bs4 import BeautifulSoup
|
|
soup = BeautifulSoup(html, 'html.parser')
|
|
|
|
# Try h1, h2, h3 in order
|
|
for tag in ['h1', 'h2', 'h3']:
|
|
elem = soup.find(tag)
|
|
if elem:
|
|
return elem.get_text().strip()[:50]
|
|
|
|
# Try first paragraph
|
|
p = soup.find('p')
|
|
if p:
|
|
text = p.get_text().strip()[:50]
|
|
if text:
|
|
return text + "..."
|
|
|
|
return None
|
|
|
|
|
|
def build_toc_from_manifest(manager: ManifestManager) -> list:
|
|
"""Build TOC from manifest entries."""
|
|
files = {}
|
|
for entry in manager.entries:
|
|
fp = entry.file_path
|
|
if fp not in files:
|
|
files[fp] = {
|
|
'count': 0,
|
|
'first_text': '',
|
|
'total_chars': 0
|
|
}
|
|
files[fp]['count'] += 1
|
|
files[fp]['total_chars'] += len(entry.original_text)
|
|
if not files[fp]['first_text'] and entry.original_text:
|
|
files[fp]['first_text'] = entry.original_text[:40].replace('\n', ' ')
|
|
|
|
toc = []
|
|
for i, (fp, info) in enumerate(sorted(files.items()), 1):
|
|
toc.append({
|
|
'index': i,
|
|
'href': fp,
|
|
'title': info['first_text'] or f"File {i}",
|
|
'paragraphs': info['count'],
|
|
'chars': info['total_chars']
|
|
})
|
|
return toc
|
|
|
|
|
|
def show_toc(manager: ManifestManager):
|
|
"""Display TOC with chapter numbers."""
|
|
toc = build_toc_from_manifest(manager)
|
|
|
|
print("\n" + "=" * 70)
|
|
print("章节目录 (Table of Contents)")
|
|
print("=" * 70)
|
|
print(f"{'#':>3} | {'段落':>5} | {'字符':>6} | 章节标题")
|
|
print("-" * 70)
|
|
|
|
for item in toc:
|
|
title = item['title'][:45] if len(item['title']) > 45 else item['title']
|
|
print(f"{item['index']:3d} | {item['paragraphs']:5d} | {item['chars']:6d} | {title}")
|
|
|
|
print("-" * 70)
|
|
print(f"共 {len(toc)} 个章节")
|
|
print("\n用法: python scripts/translate_chapter.py --chapter <编号>")
|
|
print("示例: python scripts/translate_chapter.py --chapter 5")
|
|
|
|
|
|
def get_chapter_entries(manager: ManifestManager, chapter_index: int) -> tuple:
|
|
"""Get entries for a specific chapter by index."""
|
|
toc = build_toc_from_manifest(manager)
|
|
|
|
if chapter_index < 1 or chapter_index > len(toc):
|
|
print(f"错误: 章节编号 {chapter_index} 无效 (范围: 1-{len(toc)})")
|
|
return None, None
|
|
|
|
chapter = toc[chapter_index - 1]
|
|
href = chapter['href']
|
|
|
|
entries = [e for e in manager.entries if e.file_path == href]
|
|
return chapter, entries
|
|
|
|
|
|
def create_char_based_chunks(entries: list, chunk_size: int = CHUNK_SIZE_CHARS) -> list:
|
|
"""
|
|
Create chunks based on character count (~5000 chars each).
|
|
Returns list of entry lists.
|
|
"""
|
|
chunks = []
|
|
current_chunk = []
|
|
current_size = 0
|
|
|
|
for entry in entries:
|
|
text_len = len(entry.original_text)
|
|
|
|
# If adding this entry exceeds limit and we have content, start new chunk
|
|
if current_size + text_len > chunk_size and current_chunk:
|
|
chunks.append(current_chunk)
|
|
current_chunk = []
|
|
current_size = 0
|
|
|
|
current_chunk.append(entry)
|
|
current_size += text_len
|
|
|
|
if current_chunk:
|
|
chunks.append(current_chunk)
|
|
|
|
return chunks
|
|
|
|
|
|
async def translate_chapter(chapter: dict, entries: list, manager: ManifestManager,
|
|
llm_client: LLMClient, profile, test_mode: bool = False):
|
|
"""Translate a complete chapter."""
|
|
|
|
print(f"\n开始翻译章节 #{chapter['index']}: {chapter['title'][:40]}...")
|
|
print(f" 文件: {chapter['href']}")
|
|
print(f" 总段落: {len(entries)}")
|
|
|
|
# Filter untranslated
|
|
untranslated = [e for e in entries if not e.translated_text]
|
|
print(f" 待翻译: {len(untranslated)}")
|
|
|
|
if not untranslated:
|
|
print(" ✅ 该章节已全部翻译!")
|
|
return
|
|
|
|
# Create character-based chunks
|
|
chunks = create_char_based_chunks(untranslated)
|
|
print(f" 分块: {len(chunks)} 个 Chunk (约{CHUNK_SIZE_CHARS}字符/块)")
|
|
|
|
if test_mode:
|
|
print(" [测试模式] 只翻译前2个 Chunk")
|
|
chunks = chunks[:2]
|
|
|
|
# Show chunk stats
|
|
for i, chunk in enumerate(chunks, 1):
|
|
total_chars = sum(len(e.original_text) for e in chunk)
|
|
print(f" Chunk {i}: {len(chunk)} 段落, {total_chars} 字符")
|
|
|
|
# Translate
|
|
restorer = FormatRestorer()
|
|
total_success = 0
|
|
total_failed = 0
|
|
|
|
for i, chunk in enumerate(chunks, 1):
|
|
chunk_chars = sum(len(e.original_text) for e in chunk)
|
|
print(f"\n 翻译 Chunk {i}/{len(chunks)} ({len(chunk)} 段, {chunk_chars} 字符)...")
|
|
|
|
try:
|
|
results = await llm_client.translate_chunk(
|
|
chunk,
|
|
instruction=profile.style_guide if hasattr(profile, 'style_guide') else None,
|
|
mode="bilingual"
|
|
)
|
|
|
|
# Apply results
|
|
chunk_success = 0
|
|
chunk_failed = 0
|
|
|
|
for entry in chunk:
|
|
if entry.entry_id in results:
|
|
translated = results[entry.entry_id]
|
|
entry.translated_text = translated
|
|
|
|
# Verify placeholder restoration
|
|
if entry.placeholders:
|
|
_, success = restorer.restore(translated, entry.placeholders)
|
|
if success:
|
|
chunk_success += 1
|
|
else:
|
|
chunk_failed += 1
|
|
print(f" ⚠️ 占位符还原警告: {entry.entry_id[-30:]}")
|
|
else:
|
|
chunk_success += 1
|
|
else:
|
|
chunk_failed += 1
|
|
print(f" ❌ 缺失: {entry.entry_id[-30:]}")
|
|
|
|
total_success += chunk_success
|
|
total_failed += chunk_failed
|
|
print(f" ✓ 成功: {chunk_success}, 失败: {chunk_failed}")
|
|
|
|
# Save after each chunk
|
|
manager.save()
|
|
|
|
except Exception as e:
|
|
print(f" ❌ Chunk {i} 翻译失败: {e}")
|
|
total_failed += len(chunk)
|
|
|
|
print(f"\n翻译完成:")
|
|
print(f" ✅ 成功: {total_success}")
|
|
print(f" ❌ 失败: {total_failed}")
|
|
|
|
# Show sample results
|
|
print(f"\n翻译样例 (前3段):")
|
|
print("-" * 60)
|
|
translated_entries = [e for e in entries if e.translated_text][:3]
|
|
for entry in translated_entries:
|
|
orig = entry.original_text[:40].replace('\n', ' ')
|
|
trans = entry.translated_text[:40].replace('\n', ' ') if entry.translated_text else "(无)"
|
|
print(f" 原: {orig}...")
|
|
print(f" 译: {trans}...")
|
|
print()
|
|
|
|
|
|
async def main():
|
|
parser = argparse.ArgumentParser(description="翻译指定章节")
|
|
parser.add_argument("--show-toc", action="store_true", help="显示章节目录")
|
|
parser.add_argument("--chapter", "-c", type=int, help="章节编号 (从1开始)")
|
|
parser.add_argument("--test", "-t", action="store_true", help="测试模式 (只翻译前2个chunk)")
|
|
args = parser.parse_args()
|
|
|
|
if not MANIFEST_PATH.exists():
|
|
print(f"错误: Manifest 不存在: {MANIFEST_PATH}")
|
|
print("请先运行主管道生成 manifest。")
|
|
return
|
|
|
|
# Load manifest
|
|
manager = ManifestManager(MANIFEST_PATH)
|
|
manager.load()
|
|
print(f"已加载 manifest: {len(manager.entries)} 条目")
|
|
|
|
# Show TOC
|
|
if args.show_toc or not args.chapter:
|
|
show_toc(manager)
|
|
return
|
|
|
|
if not API_KEY:
|
|
print("错误: V3_API_KEY 未设置")
|
|
return
|
|
|
|
# Get chapter entries
|
|
chapter, entries = get_chapter_entries(manager, args.chapter)
|
|
if not chapter:
|
|
return
|
|
|
|
# Initialize LLM client
|
|
from src.utils import ensure_directory
|
|
ensure_directory(CHUNK_DIR)
|
|
|
|
llm_client = LLMClient(
|
|
api_key=API_KEY,
|
|
base_url=BASE_URL,
|
|
model=MODEL,
|
|
extra_headers=EXTRA_HEADERS,
|
|
chunk_dir=CHUNK_DIR
|
|
)
|
|
|
|
try:
|
|
# Generate profile
|
|
print("\n生成书籍 Profile... (Skipping for debug)")
|
|
# profiler = BookProfiler(llm_client)
|
|
# profile = await profiler.analyze(manager.entries)
|
|
# print(f" 风格: {profile.style_guide[:80] if profile.style_guide else '(无)'}...")
|
|
|
|
class DummyProfile:
|
|
style_guide = "Keep technical terms. Translate accurately."
|
|
profile = DummyProfile()
|
|
|
|
# Translate chapter
|
|
await translate_chapter(chapter, entries, manager, llm_client, profile, args.test)
|
|
|
|
print(f"\n✅ Manifest 已保存: {MANIFEST_PATH}")
|
|
print(f"✅ Chunk 文件保存在: {CHUNK_DIR}")
|
|
|
|
finally:
|
|
await llm_client.close()
|
|
|
|
|
|
if __name__ == "__main__":
|
|
print("DEBUG: Script started execution")
|
|
try:
|
|
asyncio.run(main())
|
|
print("DEBUG: Script finished execution")
|
|
except Exception as e:
|
|
import traceback
|
|
traceback.print_exc()
|
|
print(f"CRITICAL ERROR: {e}")
|