#!/usr/bin/env python3 """Phase 4 成稿阶段:统一入口。 从 final_zh_polished.md(或指定的 Markdown)+ manifest.json 生成: - .pdf PDF(ReportLab 或 Quarto/xelatex) - <title>.docx Pandoc 出 DOCX - <title>-en.pdf 如果存在 final_en.md 也一并出英文版(可选) 文件名来自 manifest.report_title(去掉非法字符),不再用 "final.pdf" 这种通用名。 用法: uv run python scripts/build_report.py <project_slug> # 使用 Quarto/xelatex 引擎(推荐,更好的中文+宽表支持): uv run python scripts/build_report.py <project_slug> --engine quarto # 只生成 PDF: uv run python scripts/build_report.py <project_slug> --no-docx # 从自定义 md 生成: uv run python scripts/build_report.py <project_slug> --input phase4/final_zh.md 环境依赖: - reportlab, pypandoc, 思源字体(bash .opencode/templates/fonts/download-fonts.sh) - pandoc 可执行文件在 PATH - Quarto(可选,--engine quarto 时需要):https://quarto.org/docs/get-started/ 安装后运行:quarto install tinytex """ from __future__ import annotations import argparse import json import re import shutil import subprocess import sys import textwrap from pathlib import Path sys.path.insert(0, str(Path(__file__).resolve().parent.parent)) from scripts.lib.zenmux_client import load_secrets # noqa: F401 (为一致性) from scripts.reporting.fonts import resolve_quarto_fonts from scripts.reporting.references import build_references_block REPO_ROOT = Path(__file__).resolve().parent.parent PDF_TEMPLATE = REPO_ROOT / ".opencode" / "templates" / "report-template.py" DEFAULT_FONTS_DIR = REPO_ROOT / ".opencode" / "templates" / "fonts" def resolve_project(arg: str) -> Path: p = Path(arg) if p.is_dir(): return p.resolve() cand = REPO_ROOT / "projects" / arg if cand.is_dir(): return cand.resolve() raise SystemExit(f"project not found: {arg}") # 文件系统对文件名的常见限制:Windows 更严格,按最小公倍数来 _FS_ILLEGAL_RE = re.compile(r'[\\/:*?"<>|\r\n\t]+') _WHITESPACE_RE = re.compile(r"\s+") def sanitize_filename(name: str, max_len: int = 120) -> str: """把报告标题变成可跨 OS 使用的文件名。""" if not name: return "report" # 去掉非法字符 cleaned = _FS_ILLEGAL_RE.sub(" ", name) # 合并空白 cleaned = _WHITESPACE_RE.sub(" ", cleaned).strip() # 首尾 . 空格 . (Windows 要求) cleaned = cleaned.strip(". ").strip() if len(cleaned) > max_len: cleaned = cleaned[:max_len].rstrip() return cleaned or "report" def load_manifest(path: Path) -> dict: return json.loads(path.read_text(encoding="utf-8")) def determine_input_md(project_root: Path, preferred: str | None) -> Path: """决定用哪个 Markdown 出稿。 优先级:--input 指定 > final_zh_polished.md > final_zh.md > final_en.md """ if preferred: p = project_root / preferred if not p.exists(): raise SystemExit(f"指定的 --input 不存在:{p}") return p for candidate in ( "phase4/final_zh_polished.md", "phase4/final_zh.md", "phase4/final_en.md", ): p = project_root / candidate if p.exists(): return p raise SystemExit( f"找不到任何 Markdown 源。项目根:{project_root}\n" "先跑 translate.py / polish.py 或使用 --input 指定路径。" ) def build_pdf( md_path: Path, manifest_path: Path, output_pdf: Path, fonts_dir: Path, sources_path: Path | None, ) -> None: """调用 report-template.py 生成 PDF。""" cmd = [ sys.executable, str(PDF_TEMPLATE), "--input", str(md_path), "--manifest", str(manifest_path), "--output", str(output_pdf), "--fonts-dir", str(fonts_dir), ] if sources_path and sources_path.exists(): cmd += ["--sources", str(sources_path)] print(f"\n→ 生成 PDF:{output_pdf.name}") result = subprocess.run(cmd, check=False) if result.returncode != 0: raise SystemExit(f"PDF 生成失败,返回码 {result.returncode}") def _detect_wide_tables(md_text: str, min_cols: int = 8) -> list[tuple[int, int]]: """返回所有列数 >= min_cols 的 Markdown 表格的 (start_line, end_line) 区间(0-based)。""" lines = md_text.split("\n") ranges = [] i = 0 while i < len(lines): line = lines[i] if line.startswith("|") and line.count("|") - 1 >= min_cols: # Possible table header — next line should be separator if i + 1 < len(lines) and re.match(r"^\|[\s\-:|]+\|", lines[i + 1]): start = i j = i + 2 while j < len(lines) and lines[j].strip().startswith("|"): j += 1 ranges.append((start, j)) i = j continue i += 1 return ranges def prepare_qmd( md_path: Path, manifest: dict, output_qmd: Path, fonts_dir: Path, sources_path: Path | None, wide_table_cols: int = 8, ) -> None: """将普通 Markdown 转换为带 Quarto front matter 的 .qmd 文件。 主要处理: 1. 插入 YAML front matter(标题、字体、页面设置等) 2. 用 {.landscape} div 包裹列数 >= wide_table_cols 的宽表 3. 将 [TOC will be generated...] 占位符替换为真实 TOC 指令 4. 将 [REFERENCES will be filled...] 占位符替换为参考文献内容 """ title = manifest.get("report_title", "报告") subtitle = manifest.get("report_subtitle", "") date = manifest.get("date", "") # 决定字体名称:Quarto/xelatex 使用系统字体 family name。 fonts = resolve_quarto_fonts(fonts_dir) main_font = fonts.main_font sans_font = fonts.sans_font # Write LaTeX header file for CJK font setup. # Using a separate .tex file avoids YAML escape issues with backslashes. mf = main_font # "Source Han Serif CN" sf = sans_font # "Source Han Sans CN" front_matter = textwrap.dedent(f"""\ --- title: "{title}" subtitle: "{subtitle}" date: "{date}" lang: zh format: pdf: pdf-engine: xelatex CJKmainfont: "{mf}" mainfont: "{mf}" mainfontoptions: - BoldFont={mf} - ItalicFont={mf} - BoldItalicFont={mf} CJKoptions: - BoldFont={mf} - ItalicFont={mf} - BoldItalicFont={mf} sansfont: "{sf}" sansfontoptions: - BoldFont={sf} - ItalicFont={sf} - BoldItalicFont={sf} monofont: "Liberation Mono" papersize: a4 documentclass: scrartcl classoption: - DIV=11 - headinclude toc: true toc-depth: 2 toc-title: "目录" number-sections: false colorlinks: true linkcolor: NavyBlue urlcolor: NavyBlue geometry: - top=25mm - bottom=25mm - left=25mm - right=20mm pdf-engine-opts: - "-stack-size=32768" - "-extra-mem-top=2000000" include-in-header: - file: _preamble.tex --- """) md_text = md_path.read_text(encoding="utf-8") # Remove existing YAML front matter if any (between first two ---) if md_text.startswith("---"): end = md_text.find("\n---", 3) if end != -1: md_text = md_text[end + 4:].lstrip("\n") # Quarto already renders the title from YAML; drop a duplicated leading H1. md_text = re.sub( rf"^#\s+{re.escape(title)}\s*\n+", "", md_text, count=1, ) # Replace TOC placeholder md_text = re.sub( r"\[TOC will be generated.*?\]", "", # Quarto handles TOC via front matter md_text, ) # Replace REFERENCES placeholder with actual references from sources.jsonl ref_block = build_references_block(sources_path, md_text) md_text = re.sub( r"\[REFERENCES will be filled.*?\]", ref_block, md_text, ) # Wrap wide tables in {.landscape} divs lines = md_text.split("\n") wide_ranges = _detect_wide_tables(md_text, min_cols=wide_table_cols) if wide_ranges: # Insert landscape wrappers from bottom up (so line numbers stay valid) for start, end in reversed(wide_ranges): lines.insert(end, "\n:::") lines.insert(start, "::: {.landscape}\n") md_text = "\n".join(lines) # Write LaTeX preamble file (table + landscape support) preamble_tex = output_qmd.parent / "_preamble.tex" preamble_tex.write_text( "\\usepackage{longtable}\n" "\\usepackage{booktabs}\n" "\\usepackage{array}\n" "\\usepackage{xcolor}\n" "\\usepackage{titlesec}\n" "\\definecolor{DRBlue}{HTML}{1E3A8A}\n" "\\definecolor{DRSlate}{HTML}{374151}\n" "\\definecolor{DRMuted}{HTML}{6B7280}\n" # Use lscape instead of pdflscape to avoid \LS@makefcolumn recursion # which exhausts TeX param_size on large longtables. # lscape rotates content without changing page media box (reader must rotate). "\\usepackage{lscape}\n" "\\setlength{\\LTpre}{6pt}\n" "\\setlength{\\LTpost}{6pt}\n" "\\setlength{\\tabcolsep}{3pt}\n" "\\linespread{1.18}\n" "\\setlength{\\parindent}{2em}\n" "\\setlength{\\parskip}{0.25em}\n" "\\newcommand{\\sectionbreak}{\\clearpage}\n" "\\titleformat{\\section}[display]\n" " {\\centering\\Large\\bfseries\\sffamily\\color{DRBlue}}\n" " {}{0pt}{}\n" "\\titlespacing*{\\section}{0pt}{0pt}{1.1em}\n" "\\titleformat{\\subsection}\n" " {\\large\\bfseries\\sffamily\\color{DRBlue}}\n" " {}{0pt}{}\n" "\\titlespacing*{\\subsection}{0pt}{1.1em}{0.45em}\n" "\\titleformat{\\subsubsection}\n" " {\\normalsize\\bfseries\\sffamily\\color{DRSlate}}\n" " {}{0pt}{}\n" "\\titlespacing*{\\subsubsection}{0pt}{0.9em}{0.35em}\n" "\\renewcommand{\\contentsname}{目录}\n" "\\setcounter{tocdepth}{2}\n", encoding="utf-8", ) output_qmd.write_text(front_matter + md_text, encoding="utf-8") print(f" .qmd prepared: {output_qmd.name} ({len(wide_ranges)} landscape table(s))") def build_pdf_quarto( md_path: Path, manifest: dict, output_pdf: Path, fonts_dir: Path, sources_path: Path | None, ) -> None: """使用 Quarto + xelatex 生成 PDF。""" if not shutil.which("quarto"): raise SystemExit( "quarto 命令未找到。请先安装 Quarto:https://quarto.org/docs/get-started/\n" "安装后运行:quarto install tinytex" ) # Prepare .qmd in the same dir as output_pdf qmd_path = output_pdf.parent / (output_pdf.stem + ".qmd") prepare_qmd(md_path, manifest, qmd_path, fonts_dir, sources_path) print(f"\n→ 生成 PDF(Quarto/xelatex):{output_pdf.name}") cmd = [ "quarto", "render", str(qmd_path), "--to", "pdf", "--output", output_pdf.name, ] result = subprocess.run(cmd, cwd=str(output_pdf.parent), check=False) if result.returncode != 0: raise SystemExit(f"Quarto PDF 生成失败,返回码 {result.returncode}") # Clean up auxiliary files Quarto leaves behind for ext in (".tex", ".log", ".aux", ".toc", ".out", "-files"): candidate = output_pdf.parent / (output_pdf.stem + ext) if candidate.exists(): candidate.unlink(missing_ok=True) def build_docx(md_path: Path, output_docx: Path, title: str) -> None: """用 pandoc 生成 DOCX。""" if not shutil.which("pandoc"): print(f" ⚠ pandoc 不在 PATH,跳过 DOCX 生成", file=sys.stderr) return print(f"\n→ 生成 DOCX:{output_docx.name}") # 关闭 tex_math_dollars/tex_math_single_backslash 防止文中 "$100" "$10^6" 被当数学公式 cmd = [ "pandoc", str(md_path), "-o", str(output_docx), "--from=markdown-tex_math_dollars-tex_math_single_backslash-raw_tex", "--to=docx", "--standalone", "-M", f"title={title}", "--wrap=preserve", ] # reference-doc 如果存在就用 ref_doc = REPO_ROOT / ".opencode" / "templates" / "reference.docx" if ref_doc.exists(): cmd += ["--reference-doc", str(ref_doc)] result = subprocess.run(cmd, check=False) if result.returncode != 0: print(f" ⚠ DOCX 生成失败(返回码 {result.returncode}),但不阻断流程", file=sys.stderr) def main() -> int: parser = argparse.ArgumentParser(description="Phase 4 成稿(PDF + DOCX)") parser.add_argument("project", help="项目 slug 或完整路径") parser.add_argument( "--input", default=None, help="Markdown 源(默认自动寻找 phase4/final_zh_polished.md)", ) parser.add_argument( "--output-dir", default="phase4", help="输出目录(相对项目根,默认 phase4)", ) parser.add_argument( "--fonts-dir", default=str(DEFAULT_FONTS_DIR), help="字体目录", ) parser.add_argument( "--sources", default=None, help="sources.jsonl 路径(默认 phase2/sources.jsonl)", ) parser.add_argument("--no-docx", action="store_true", help="跳过 DOCX 生成") parser.add_argument("--no-pdf", action="store_true", help="跳过 PDF 生成") parser.add_argument( "--engine", choices=["reportlab", "quarto"], default="reportlab", help="PDF 渲染引擎:reportlab(默认,Python 原生)或 quarto(xelatex,更好的中文+宽表支持)", ) parser.add_argument( "--basename", default=None, help="文件名 stem(不带扩展名),默认从 manifest.report_title 生成", ) args = parser.parse_args() project_root = resolve_project(args.project) manifest_path = project_root / "manifest.json" if not manifest_path.exists(): raise SystemExit(f"manifest.json not found: {manifest_path}") manifest = load_manifest(manifest_path) md_path = determine_input_md(project_root, args.input) title = manifest.get("report_title") or manifest.get("topic") or "Deep Research Report" basename = args.basename or sanitize_filename(title) output_dir = project_root / args.output_dir output_dir.mkdir(parents=True, exist_ok=True) pdf_path = output_dir / f"{basename}.pdf" docx_path = output_dir / f"{basename}.docx" fonts_dir = Path(args.fonts_dir) if not fonts_dir.is_absolute(): fonts_dir = REPO_ROOT / fonts_dir sources_path: Path | None = None if args.sources: sources_path = Path(args.sources) else: default_src = project_root / "phase2" / "sources.jsonl" if default_src.exists(): sources_path = default_src print("========== Phase 4 成稿 ==========") print(f"项目: {project_root.name}") print(f"Markdown: {md_path.relative_to(project_root)}") print(f"标题: {title}") print(f"输出名: {basename}") print(f"字体目录: {fonts_dir}") print(f"Sources: {sources_path if sources_path else '(缺失)'}") if not args.no_pdf: if args.engine == "quarto": build_pdf_quarto(md_path, manifest, pdf_path, fonts_dir, sources_path) else: build_pdf(md_path, manifest_path, pdf_path, fonts_dir, sources_path) if not args.no_docx: build_docx(md_path, docx_path, title) print("\n========== 完成 ==========") if pdf_path.exists(): print(f" PDF: {pdf_path.relative_to(project_root)} ({pdf_path.stat().st_size // 1024} KB)") if docx_path.exists(): print(f" DOCX: {docx_path.relative_to(project_root)} ({docx_path.stat().st_size // 1024} KB)") return 0 if __name__ == "__main__": sys.exit(main())