v0.6-wip: polish pipeline + PDF template fixes
Phase 4 \u6da6\u8272\u5c42\u4e0e PDF \u6a21\u677f\u4fee\u590d\uff0c\u63a5\u7740\u4e0a\u4e00\u4e2a commit\u3002 polish.py\uff08\u65b0\u589e\uff09\uff1a - \u548c translate.py \u5bf9\u79f0\uff0c\u6309 H2 section \u5207\u5757 \u2192 \u5faa\u73af\u6da6\u8272 \u2192 \u62fc\u63a5 - \u4f7f\u7528 <<<POLISHED>>>/<<<NOTES>>> \u5206\u9694\u7b26 prompt\uff08\u907f\u5f00 Markdown-in-JSON \u95ee\u9898\uff09 - \u65ad\u70b9\u7eed\u4f20\u3001\u6a21\u578b\u81ea\u8bc4\u6ce8\u8bb0\u843d\u76d8 polish_notes.jsonl - \u5728\u53cc\u9776\u70b9 RNAi \u9879\u76ee\u8dd1\u901a\uff1a60 \u5757\u5168\u6210\u529f\uff0c10.7 \u5206\u949f\uff0c$1.20\uff0c\u5b57\u6570 -0.2% report-template.py\uff08\u5927\u6539\u4e00\u6279 P0 bug\uff09\uff1a - \u5b57\u4f53\u6ce8\u518c\u652f\u6301 fonts/ttf/ \u5b50\u76ee\u5f55\uff08\u89e3\u51b3 OTF PostScript outlines \u4e0d\u517c\u5bb9\uff09 - \u5220\u9664 build_disclaimer \u91cd\u590d\u8c03\u7528\uff08\u514d\u8d23\u58f0\u660e\u4ece Markdown \u8bfb\uff0cmanifest \u4e0d\u518d\u91cd\u590d\uff09 - build_body \u81ea\u52a8\u8df3\u8fc7\u6b63\u6587\u9996\u4e2a H1+\u5c01\u9762\u5143\u4fe1\u606f\u6bb5\uff08\u4e0e\u5c01\u9762\u91cd\u590d\uff09 - \u5360\u4f4d\u7b26 \u201c\u76ee\u5f55\u5c06\u5728\u6700\u7ec8\u6e32\u67d3\u65f6\u81ea\u52a8\u751f\u6210\u201d \u2192 \u81ea\u52a8\u751f\u6210 TOC - \u5360\u4f4d\u7b26 \u201c\u5b8c\u6574\u7f16\u53f7\u53c2\u8003\u6587\u732e\u5217\u8868\u2026\u201d \u2192 \u4ece phase2/sources.jsonl \u81ea\u52a8\u751f\u6210 GB/T 7714 \u683c\u5f0f\u5f15\u6587 - src \u4e0a\u6807\u6b63\u5219\u6269\u5c55\uff1a\u652f\u6301 src_A14 / src_B-18 \u7b49\u5b57\u6bcd+\u6570\u5b57\u7ec4\u5408 ID\uff08\u539f\u53ea\u652f\u6301 src_\d+\uff09 - Unicode \u4e0a/\u4e0b\u6807\u8f6c <super>/<sub>\uff1a10\u2076 \u2192 10<super>6</super>\uff08\u601d\u6e90\u5b57\u4f53\u5b50\u96c6\u4e0d\u542b\u4e0a\u6807\u5b57\u5f62\uff0c\u5426\u5219\u6e32\u67d3\u65b9\u6846\uff09 - \u4e2d\u82f1\u6df7\u6392\u81ea\u52a8\u52a0\u7a7a\u683c\uff08CJK \u2194 [A-Za-z0-9] \u8fb9\u754c\uff09 - \u8868\u683c\u6837\u5f0f\u91cd\u505a\uff1atable-header/table-cell/table-cell-center\uff1b\u5782\u76f4\u5c45\u4e2d\uff1b\u77ed cell\uff08\u7eaf\u6570\u5b57/\u77ed\u6807\u7b7e\uff09\u6c34\u5e73\u5c45\u4e2d\uff1b\u957f cell \u81ea\u52a8 CJK \u6362\u884c - TOC \u672b\u5c3e PageBreak\uff08\u76ee\u5f55\u72ec\u5360\u6574\u9875\uff09 \u5df2\u77e5\u672a\u4fee\u590d\uff1a - Maywavee \u662f LLM \u5728 dr-analyst \u9636\u6bb5\u7f16\u9020\uff0c\u6b63\u786e\u4e3a Mabwell\uff08\u8fc8\u5a01\u751f\u7269\uff09\u3002\u4fe1\u6e90\u4fa7 bug\uff0c\u9700\u5728\u540e\u7eed build_glossary.py \u4e2d\u505a\u4e8b\u5b9e\u6838\u67e5\u3002 - \u6b63\u6587 101 \u4e2a src_id\u3001sources.jsonl \u53ea\u670947 \u4e2a\u3001\u4ea4\u96c6 39 \u4e2a\u2014\u2014\u662f v0.4 \u9057\u7559\u7684\u6ce8\u5165 bug\uff0cbuild_references \u73b0\u5728\u4f1a\u5217\u51fa\u7f3a\u5931\u7684 id \u4f9b\u4eba\u5de5\u6838\u5bf9 - \u53cd\u9a73\u8bc1\u636e\u6bb5\u683c\u5f0f\u4e0d\u7edf\u4e00\u662f dr-analyst/skill \u89c4\u8303\u95ee\u9898\uff0c\u4e0b\u4e00\u6279\u6539 skill Co-authored-by: User <human>
This commit is contained in:
@@ -69,18 +69,37 @@ FONT_MAP = {
|
||||
}
|
||||
|
||||
|
||||
def _resolve_font(fonts_dir: Path, fname: str) -> Optional[Path]:
|
||||
"""查找字体文件:先在 fonts_dir 根下找 OTF/TTF,再看 ttf/ 子目录的 TTF 兜底。
|
||||
|
||||
ReportLab 的 TTFont 只支持 TrueType(无 PostScript outlines)。
|
||||
思源字体的 OTF 是 PS outlines 版本,注册会失败,必须用 TTF 版本。
|
||||
"""
|
||||
# 优先级:
|
||||
# 1. 直接给的文件名(例如已经是 .ttf)
|
||||
direct = fonts_dir / fname
|
||||
if direct.exists() and direct.suffix.lower() == ".ttf":
|
||||
return direct
|
||||
# 2. 如果 fname 是 .otf,尝试在 ttf/ 子目录找同 stem 的 .ttf
|
||||
if fname.lower().endswith(".otf"):
|
||||
ttf_candidate = fonts_dir / "ttf" / (fname[:-4] + ".ttf")
|
||||
if ttf_candidate.exists():
|
||||
return ttf_candidate
|
||||
# 3. 原始 OTF 文件(让调用者自己处理错误)
|
||||
if direct.exists():
|
||||
return direct
|
||||
return None
|
||||
|
||||
|
||||
def register_fonts(fonts_dir: Path) -> None:
|
||||
missing = []
|
||||
resolved: dict[str, Path] = {}
|
||||
for logical, fname in FONT_MAP.items():
|
||||
path = fonts_dir / fname
|
||||
if not path.exists():
|
||||
missing.append(str(path))
|
||||
path = _resolve_font(fonts_dir, fname)
|
||||
if not path:
|
||||
missing.append(f"{logical} (looked for {fname} / ttf/{fname.replace('.otf','.ttf')})")
|
||||
continue
|
||||
try:
|
||||
pdfmetrics.registerFont(TTFont(logical, str(path)))
|
||||
except Exception as e:
|
||||
print(f"ERROR: font registration failed {logical} ({path}): {e}", file=sys.stderr)
|
||||
sys.exit(1)
|
||||
resolved[logical] = path
|
||||
|
||||
if missing:
|
||||
print("ERROR: missing fonts:", file=sys.stderr)
|
||||
@@ -89,6 +108,18 @@ def register_fonts(fonts_dir: Path) -> None:
|
||||
print("\nRun: bash .opencode/templates/fonts/download-fonts.sh", file=sys.stderr)
|
||||
sys.exit(1)
|
||||
|
||||
for logical, path in resolved.items():
|
||||
try:
|
||||
pdfmetrics.registerFont(TTFont(logical, str(path)))
|
||||
except Exception as e:
|
||||
print(
|
||||
f"ERROR: font registration failed {logical} ({path}): {e}\n"
|
||||
f"Hint: ReportLab needs TrueType outlines. "
|
||||
f"If this is an .otf with PostScript outlines, use the TTF version in fonts/ttf/.",
|
||||
file=sys.stderr,
|
||||
)
|
||||
sys.exit(1)
|
||||
|
||||
pdfmetrics.registerFontFamily(
|
||||
"SrcSerif",
|
||||
normal="SrcSerif",
|
||||
@@ -218,6 +249,75 @@ def build_styles() -> StyleSheet1:
|
||||
allowOrphans=0,
|
||||
))
|
||||
|
||||
# Table cell - no first-line indent, smaller font, CJK wrap for auto line break
|
||||
ss.add(ParagraphStyle(
|
||||
name="table-cell",
|
||||
fontName="SrcSerif",
|
||||
fontSize=9,
|
||||
leading=13,
|
||||
alignment=TA_LEFT,
|
||||
firstLineIndent=0,
|
||||
spaceBefore=0,
|
||||
spaceAfter=0,
|
||||
textColor=colors.HexColor("#1a1a1a"),
|
||||
wordWrap="CJK",
|
||||
))
|
||||
|
||||
# 表头水平居中,略加粗
|
||||
ss.add(ParagraphStyle(
|
||||
name="table-header",
|
||||
fontName="SrcSans-Bold",
|
||||
fontSize=9.5,
|
||||
leading=14,
|
||||
alignment=TA_CENTER,
|
||||
firstLineIndent=0,
|
||||
spaceBefore=0,
|
||||
spaceAfter=0,
|
||||
textColor=colors.HexColor("#1e3a8a"),
|
||||
wordWrap="CJK",
|
||||
))
|
||||
|
||||
# 短文本数字 cell(用于纯数字/短标签列,水平居中)
|
||||
ss.add(ParagraphStyle(
|
||||
name="table-cell-center",
|
||||
fontName="SrcSerif",
|
||||
fontSize=9,
|
||||
leading=13,
|
||||
alignment=TA_CENTER,
|
||||
firstLineIndent=0,
|
||||
spaceBefore=0,
|
||||
spaceAfter=0,
|
||||
textColor=colors.HexColor("#1a1a1a"),
|
||||
wordWrap="CJK",
|
||||
))
|
||||
|
||||
# TOC entry styles
|
||||
ss.add(ParagraphStyle(
|
||||
name="toc-h1",
|
||||
fontName="SrcSans-Bold",
|
||||
fontSize=11,
|
||||
leading=18,
|
||||
alignment=TA_LEFT,
|
||||
firstLineIndent=0,
|
||||
spaceBefore=6,
|
||||
spaceAfter=2,
|
||||
textColor=colors.HexColor("#1e3a8a"),
|
||||
wordWrap="CJK",
|
||||
))
|
||||
ss.add(ParagraphStyle(
|
||||
name="toc-h2",
|
||||
fontName="SrcSerif",
|
||||
fontSize=10,
|
||||
leading=16,
|
||||
alignment=TA_LEFT,
|
||||
leftIndent=18,
|
||||
firstLineIndent=0,
|
||||
spaceBefore=1,
|
||||
spaceAfter=1,
|
||||
textColor=colors.HexColor("#374151"),
|
||||
wordWrap="CJK",
|
||||
))
|
||||
|
||||
# Cover - main title (heavy, centered, large)
|
||||
ss.add(ParagraphStyle(
|
||||
name="cover-title",
|
||||
@@ -397,12 +497,76 @@ def parse_markdown(md_text: str) -> List[Block]:
|
||||
return blocks
|
||||
|
||||
|
||||
def md_inline_to_rl(text: str) -> str:
|
||||
_CJK_RE = re.compile(r"[\u4e00-\u9fff\u3400-\u4dbf]")
|
||||
_CJK_ASCII_SPACE_RE = re.compile(
|
||||
r"(?<=[\u4e00-\u9fff])(?=[A-Za-z0-9])|(?<=[A-Za-z0-9)\]])(?=[\u4e00-\u9fff])"
|
||||
)
|
||||
|
||||
|
||||
def _add_cjk_spaces(text: str) -> str:
|
||||
"""在中文字符与 ASCII(英文/数字)交界处加半角空格,提升可读性。
|
||||
|
||||
作用范围故意保守:只在 CJK ↔ [A-Za-z0-9] 的边界插入空格,不影响
|
||||
`[src_xxx]` 这种方括号内部,也不影响数字紧跟单位(如 "300 µg",因为
|
||||
µ 是非 ASCII)。
|
||||
"""
|
||||
return _CJK_ASCII_SPACE_RE.sub(" ", text)
|
||||
|
||||
|
||||
# Unicode 上标 → 常规数字/字母的映射(思源字体子集不含上标字形,需显式用 <super> 渲染)
|
||||
_UNICODE_SUPER = {
|
||||
"⁰": "0", "¹": "1", "²": "2", "³": "3", "⁴": "4",
|
||||
"⁵": "5", "⁶": "6", "⁷": "7", "⁸": "8", "⁹": "9",
|
||||
"⁺": "+", "⁻": "-", "⁼": "=", "⁽": "(", "⁾": ")",
|
||||
"ⁱ": "i", "ⁿ": "n",
|
||||
}
|
||||
_UNICODE_SUB = {
|
||||
"₀": "0", "₁": "1", "₂": "2", "₃": "3", "₄": "4",
|
||||
"₅": "5", "₆": "6", "₇": "7", "₈": "8", "₉": "9",
|
||||
"₊": "+", "₋": "-", "₌": "=", "₍": "(", "₎": ")",
|
||||
}
|
||||
_SUPER_CHARS_RE = re.compile(f"([{''.join(_UNICODE_SUPER)}]+)")
|
||||
_SUB_CHARS_RE = re.compile(f"([{''.join(_UNICODE_SUB)}]+)")
|
||||
|
||||
|
||||
def _replace_unicode_superscripts(text: str) -> str:
|
||||
"""把连续的 Unicode 上标字符替换为 ReportLab <super> 标签。
|
||||
|
||||
例:10⁶ → 10<super>6</super>
|
||||
H₂O → H<sub>2</sub>O
|
||||
思源字体子集不包含这些字形,直接放会渲染成方框。
|
||||
"""
|
||||
def _sup(m: "re.Match") -> str:
|
||||
payload = "".join(_UNICODE_SUPER.get(c, c) for c in m.group(1))
|
||||
return f"<super>{payload}</super>"
|
||||
|
||||
def _sub(m: "re.Match") -> str:
|
||||
payload = "".join(_UNICODE_SUB.get(c, c) for c in m.group(1))
|
||||
return f"<sub>{payload}</sub>"
|
||||
|
||||
text = _SUPER_CHARS_RE.sub(_sup, text)
|
||||
text = _SUB_CHARS_RE.sub(_sub, text)
|
||||
return text
|
||||
|
||||
|
||||
def md_inline_to_rl(text: str, *, add_cjk_space: bool = True) -> str:
|
||||
"""Markdown inline → ReportLab mini HTML."""
|
||||
# 先做 Unicode 上/下标归一(字体子集不含这些字形,否则渲染为方框)
|
||||
text = _replace_unicode_superscripts(text)
|
||||
# 然后在中英交界处加空格
|
||||
if add_cjk_space:
|
||||
text = _add_cjk_spaces(text)
|
||||
text = re.sub(r"\*\*([^*]+)\*\*", r"<b>\1</b>", text)
|
||||
text = re.sub(r"(?<!\*)\*([^*]+)\*(?!\*)", r"<i>\1</i>", text)
|
||||
text = re.sub(r"`([^`]+)`", r'<font face="Courier">\1</font>', text)
|
||||
text = re.sub(r"\[(src_\d+)\]", r"<super><font size=8>[\1]</font></super>", text)
|
||||
# 引用 ID 支持字母+数字(src_042 / src_A14 / src_B-18)
|
||||
text = re.sub(
|
||||
r"\[((?:src_[A-Za-z0-9_-]+)(?:\s*,\s*src_[A-Za-z0-9_-]+)*)\]",
|
||||
lambda m: '<super><font size=7>['
|
||||
+ m.group(1).replace(" ", "")
|
||||
+ ']</font></super>',
|
||||
text,
|
||||
)
|
||||
text = re.sub(r"\[([^\]]+)\]\(([^)]+)\)", r"\1", text)
|
||||
return text
|
||||
|
||||
@@ -464,90 +628,438 @@ def make_page_decorator(manifest: Manifest):
|
||||
return draw
|
||||
|
||||
|
||||
def build_cover(manifest: Manifest, styles: StyleSheet1) -> List:
|
||||
def build_cover(manifest: Manifest, blocks: List[Block], styles: StyleSheet1) -> List:
|
||||
"""构建封面:以 manifest 为准,完全不依赖正文第一段。
|
||||
|
||||
正文里的 H1 标题 + 元信息段会在 build_body 阶段被识别并跳过,
|
||||
避免"封面和第一页重复"的 v0.5 老问题。
|
||||
"""
|
||||
story = []
|
||||
story.append(Spacer(1, 6 * cm))
|
||||
story.append(Spacer(1, 5 * cm))
|
||||
story.append(Paragraph(manifest.report_title, styles["cover-title"]))
|
||||
if manifest.report_subtitle:
|
||||
story.append(Spacer(1, 0.5 * cm))
|
||||
story.append(Paragraph(manifest.report_subtitle, styles["cover-subtitle"]))
|
||||
story.append(Spacer(1, 5 * cm))
|
||||
story.append(Spacer(1, 4.5 * cm))
|
||||
|
||||
if manifest.confidentiality:
|
||||
story.append(Paragraph(manifest.confidentiality, styles["cover-confidential"]))
|
||||
story.append(Spacer(1, 2 * cm))
|
||||
story.append(Spacer(1, 1.5 * cm))
|
||||
|
||||
story.append(Paragraph(f"类型:{manifest.type}", styles["cover-meta"]))
|
||||
if manifest.type:
|
||||
story.append(Paragraph(f"类型:{manifest.type}", styles["cover-meta"]))
|
||||
story.append(Paragraph(f"作者:{manifest.author}", styles["cover-meta"]))
|
||||
story.append(Paragraph(f"编制日期:{manifest.date}", styles["cover-meta"]))
|
||||
if manifest.date:
|
||||
story.append(Paragraph(f"编制日期:{manifest.date}", styles["cover-meta"]))
|
||||
story.append(Paragraph(f"版本:v{manifest.version}", styles["cover-meta"]))
|
||||
|
||||
story.append(PageBreak())
|
||||
return story
|
||||
|
||||
|
||||
def build_disclaimer(manifest: Manifest, styles: StyleSheet1) -> List:
|
||||
story = []
|
||||
story.append(Paragraph("免责声明", styles["h1"]))
|
||||
story.append(Spacer(1, 0.5 * cm))
|
||||
story.append(Paragraph(manifest.disclaimer, styles["body"]))
|
||||
# ============================================================
|
||||
# 占位符识别 & 自动生成内容
|
||||
# ============================================================
|
||||
|
||||
# 识别"目录将在最终渲染时自动生成"这类占位段落(dr-editor-in-chief 写的模板行)
|
||||
_TOC_PLACEHOLDER_RE = re.compile(r"目录将在最终渲染时自动生成|TOC will be generated|\[TOC\]", re.IGNORECASE)
|
||||
_REF_PLACEHOLDER_RE = re.compile(
|
||||
r"完整编号参考文献列表将在此处呈现|将正文中每个.*src_xxx.*标识符映射|\[REFERENCES\]",
|
||||
re.IGNORECASE,
|
||||
)
|
||||
# 跳过封面 H1(正文第一个 H1 + 其后直到第一个 "---" 或 "## " 的所有段落)
|
||||
# 这部分内容由 build_cover 从 manifest 生成。
|
||||
_COVER_FRONTMATTER_PATTERNS = (
|
||||
"confidentiality", "date:", "version:", "system:", "机密",
|
||||
)
|
||||
|
||||
|
||||
def is_cover_frontmatter(text: str) -> bool:
|
||||
"""判断一段正文是否是封面元信息(Confidentiality/Date/Version/System 混排)。"""
|
||||
low = text.lower()
|
||||
hits = sum(1 for pat in _COVER_FRONTMATTER_PATTERNS if pat in low)
|
||||
return hits >= 2
|
||||
|
||||
|
||||
def collect_toc_entries(blocks: List[Block]) -> List[tuple[int, str]]:
|
||||
"""从 blocks 里抽 H1/H2 生成 TOC 条目。返回 [(level, title)]。
|
||||
|
||||
跳过一些不该进 TOC 的标题:目录本身、免责声明、摘要、术语表、参考文献、版本历史、附录。
|
||||
"""
|
||||
skip_titles_substr = (
|
||||
"目录", "table of contents",
|
||||
"免责声明", "disclaimer",
|
||||
"执行摘要", "executive summary",
|
||||
"摘要", "abstract",
|
||||
"术语表", "glossary",
|
||||
"参考文献", "references",
|
||||
"版本历史", "version history",
|
||||
"附录", "appendix",
|
||||
)
|
||||
entries: list[tuple[int, str]] = []
|
||||
for b in blocks:
|
||||
if b.kind not in ("h1", "h2"):
|
||||
continue
|
||||
title = b.content.strip()
|
||||
if any(s in title.lower() for s in skip_titles_substr):
|
||||
continue
|
||||
level = 1 if b.kind == "h1" else 2
|
||||
entries.append((level, title))
|
||||
return entries
|
||||
|
||||
|
||||
def build_toc(blocks: List[Block], styles: StyleSheet1) -> List:
|
||||
"""生成目录条目。
|
||||
|
||||
目录末尾 PageBreak 让后续内容独立成页。开头不 PageBreak,
|
||||
调用方(H1 分支)已经负责在 H1 前另起一页。
|
||||
"""
|
||||
story: list = []
|
||||
story.append(Paragraph("目录", styles["h1"]))
|
||||
story.append(Spacer(1, 0.4 * cm))
|
||||
for level, title in collect_toc_entries(blocks):
|
||||
style_name = "toc-h1" if level == 1 else "toc-h2"
|
||||
story.append(Paragraph(md_inline_to_rl(title), styles[style_name]))
|
||||
story.append(PageBreak())
|
||||
return story
|
||||
|
||||
|
||||
# ============================================================
|
||||
# sources.jsonl → 参考文献列表
|
||||
# ============================================================
|
||||
|
||||
_SRC_ID_RE = re.compile(r"\[(src_[A-Za-z0-9_-]+(?:\s*,\s*src_[A-Za-z0-9_-]+)*)\]")
|
||||
|
||||
|
||||
def collect_cited_src_ids(blocks: List[Block]) -> List[str]:
|
||||
"""扫描全文收集被引用的 src_xxx(保序去重)。"""
|
||||
seen: set[str] = set()
|
||||
order: list[str] = []
|
||||
for b in blocks:
|
||||
if b.kind in ("image",):
|
||||
continue
|
||||
for m in _SRC_ID_RE.finditer(b.content):
|
||||
for sid in m.group(1).split(","):
|
||||
sid = sid.strip()
|
||||
if sid and sid not in seen:
|
||||
seen.add(sid)
|
||||
order.append(sid)
|
||||
return order
|
||||
|
||||
|
||||
def load_sources_jsonl(path: Path) -> dict[str, dict]:
|
||||
"""加载 sources.jsonl,返回 {src_id: record}。"""
|
||||
if not path or not path.exists():
|
||||
return {}
|
||||
out: dict[str, dict] = {}
|
||||
for line in path.read_text(encoding="utf-8").splitlines():
|
||||
line = line.strip()
|
||||
if not line:
|
||||
continue
|
||||
try:
|
||||
rec = json.loads(line)
|
||||
sid = rec.get("id")
|
||||
if sid:
|
||||
out[sid] = rec
|
||||
except Exception:
|
||||
continue
|
||||
return out
|
||||
|
||||
|
||||
def format_gb7714(rec: dict) -> str:
|
||||
"""按 GB/T 7714-2015 生成参考文献条目(简化版)。
|
||||
|
||||
字段容错:authors/title/year/venue/url/doi/type 都可能缺失。
|
||||
"""
|
||||
authors = rec.get("authors") or rec.get("author") or ""
|
||||
title = rec.get("title", "").strip()
|
||||
year = rec.get("year", "")
|
||||
venue = rec.get("venue", "")
|
||||
url = rec.get("url", "")
|
||||
doi = rec.get("doi", "")
|
||||
rec_type = (rec.get("type") or "").lower()
|
||||
type_tag = {
|
||||
"journal": "[J]",
|
||||
"article": "[J]",
|
||||
"book": "[M]",
|
||||
"report": "[R]",
|
||||
"patent": "[P]",
|
||||
"thesis": "[D]",
|
||||
"standard": "[S]",
|
||||
"news": "[N/OL]",
|
||||
"web": "[EB/OL]",
|
||||
"preprint": "[J/OL]",
|
||||
"database": "[DB/OL]",
|
||||
}.get(rec_type, "[EB/OL]")
|
||||
|
||||
parts: list[str] = []
|
||||
if authors:
|
||||
parts.append(str(authors).rstrip("."))
|
||||
if title:
|
||||
parts.append(f"{title}{type_tag}")
|
||||
tail: list[str] = []
|
||||
if venue:
|
||||
tail.append(str(venue))
|
||||
if year:
|
||||
tail.append(str(year))
|
||||
if tail:
|
||||
parts.append(", ".join(tail) + ".")
|
||||
if doi:
|
||||
parts.append(f"DOI: {doi}.")
|
||||
if url:
|
||||
parts.append(f"[{rec.get('accessed_at', '')}]. {url}" if rec.get("accessed_at") else url)
|
||||
body = " ".join(p for p in parts if p).strip()
|
||||
return body
|
||||
|
||||
|
||||
def build_references(
|
||||
blocks: List[Block],
|
||||
sources_path: Optional[Path],
|
||||
styles: StyleSheet1,
|
||||
) -> List:
|
||||
"""生成参考文献段落。
|
||||
|
||||
引用顺序:按正文首次出现的先后排列(GB/T 7714 顺序编码制)。
|
||||
"""
|
||||
story: list = []
|
||||
story.append(Paragraph("参考文献", styles["h1"]))
|
||||
story.append(Spacer(1, 0.4 * cm))
|
||||
|
||||
sources = load_sources_jsonl(sources_path) if sources_path else {}
|
||||
cited_ids = collect_cited_src_ids(blocks)
|
||||
|
||||
if not cited_ids:
|
||||
story.append(Paragraph(
|
||||
"(正文未发现 [src_xxx] 引用标注)",
|
||||
styles["caption"],
|
||||
))
|
||||
return story
|
||||
|
||||
if not sources:
|
||||
# 至少列出所有被引用的 ID,供人工回填
|
||||
story.append(Paragraph(
|
||||
f"(未找到 sources.jsonl 或其内容为空。以下为正文出现的 {len(cited_ids)} 个引用标识符)",
|
||||
styles["caption"],
|
||||
))
|
||||
for i, sid in enumerate(cited_ids, 1):
|
||||
story.append(Paragraph(f"[{i}] {sid}", styles["footnote"]))
|
||||
return story
|
||||
|
||||
missing: list[str] = []
|
||||
for i, sid in enumerate(cited_ids, 1):
|
||||
rec = sources.get(sid)
|
||||
if not rec:
|
||||
missing.append(sid)
|
||||
story.append(Paragraph(
|
||||
f"[{i}] {sid}(来源记录缺失,请核查 sources.jsonl)",
|
||||
styles["footnote"],
|
||||
))
|
||||
continue
|
||||
text = format_gb7714(rec)
|
||||
# 前面加序号,后面追加 [sid] 便于正文回溯
|
||||
entry = f"[{i}] {text} <font color='#6b7280' size=7>【{sid}】</font>"
|
||||
story.append(Paragraph(entry, styles["footnote"]))
|
||||
|
||||
if missing:
|
||||
print(
|
||||
f"WARNING: {len(missing)} cited src_ids not found in sources.jsonl: "
|
||||
f"{', '.join(missing[:5])}{'...' if len(missing) > 5 else ''}",
|
||||
file=sys.stderr,
|
||||
)
|
||||
return story
|
||||
|
||||
|
||||
def _is_short_cell(text: str) -> bool:
|
||||
"""判断 cell 文本是否短到适合居中(简单表格风格)。
|
||||
|
||||
规则:
|
||||
- 纯数字/范围(含 ±, %, –, nm, µg 等常见单位)居中
|
||||
- 短标签(<= 10 字符,不含标点)居中
|
||||
- 其它(段落级文本)左对齐
|
||||
"""
|
||||
t = text.strip()
|
||||
if not t:
|
||||
return True
|
||||
# 纯数字 / 范围 / 单位
|
||||
if re.match(r"^[\d.,\s\-\–\—±×/%]+\s*[A-Za-zµ°%]*$", t):
|
||||
return True
|
||||
# 短标签(排除常见句末标点)
|
||||
if len(t) <= 10 and not any(p in t for p in ",。;:!?,.;:!?"):
|
||||
return True
|
||||
return False
|
||||
|
||||
|
||||
def _render_table_cell(text: str, styles: StyleSheet1) -> "Paragraph":
|
||||
content = md_inline_to_rl(text)
|
||||
style = styles["table-cell-center"] if _is_short_cell(text) else styles["table-cell"]
|
||||
return Paragraph(content, style)
|
||||
|
||||
|
||||
def render_table(md_table: str, styles: StyleSheet1) -> Table:
|
||||
rows = []
|
||||
"""渲染 Markdown 表格为 ReportLab Table。
|
||||
|
||||
规则:
|
||||
- 首行用 table-header 样式(水平居中 + 加粗 + 深蓝色)
|
||||
- 短 cell(纯数字/单位/短标签)水平居中
|
||||
- 长 cell(段落级文本)左对齐
|
||||
- 所有 cell 垂直居中
|
||||
- 长文字自动 CJK 换行
|
||||
- 长表自动按行分页
|
||||
"""
|
||||
rows: list[list] = []
|
||||
raw_rows = []
|
||||
for line in md_table.strip().split("\n"):
|
||||
line = line.strip().strip("|")
|
||||
cells = [c.strip() for c in line.split("|")]
|
||||
rows.append([Paragraph(md_inline_to_rl(c), styles["body"]) for c in cells])
|
||||
raw_rows.append(cells)
|
||||
|
||||
if not raw_rows:
|
||||
return Table([[""]])
|
||||
|
||||
header_cells = raw_rows[0]
|
||||
rows.append([
|
||||
Paragraph(md_inline_to_rl(c), styles["table-header"]) for c in header_cells
|
||||
])
|
||||
for cells in raw_rows[1:]:
|
||||
# 补齐列数(防御性)
|
||||
while len(cells) < len(header_cells):
|
||||
cells.append("")
|
||||
rows.append([_render_table_cell(c, styles) for c in cells])
|
||||
|
||||
table = Table(rows, repeatRows=1, splitByRow=True)
|
||||
table.setStyle(TableStyle([
|
||||
("BACKGROUND", (0, 0), (-1, 0), colors.HexColor("#e0e7ff")),
|
||||
("FONTNAME", (0, 0), (-1, 0), "SrcSans-Bold"),
|
||||
("FONTSIZE", (0, 0), (-1, -1), 9.5),
|
||||
("GRID", (0, 0), (-1, -1), 0.5, colors.HexColor("#cbd5e1")),
|
||||
("VALIGN", (0, 0), (-1, -1), "MIDDLE"),
|
||||
("LEFTPADDING", (0, 0), (-1, -1), 6),
|
||||
("RIGHTPADDING", (0, 0), (-1, -1), 6),
|
||||
("TOPPADDING", (0, 0), (-1, -1), 4),
|
||||
("BOTTOMPADDING", (0, 0), (-1, -1), 4),
|
||||
("TOPPADDING", (0, 0), (-1, -1), 5),
|
||||
("BOTTOMPADDING", (0, 0), (-1, -1), 5),
|
||||
]))
|
||||
return table
|
||||
|
||||
|
||||
def build_body(blocks: List[Block], base_dir: Path, styles: StyleSheet1) -> List:
|
||||
"""
|
||||
Render markdown blocks to flowables.
|
||||
def build_body(
|
||||
blocks: List[Block],
|
||||
base_dir: Path,
|
||||
styles: StyleSheet1,
|
||||
*,
|
||||
sources_path: Optional[Path] = None,
|
||||
) -> List:
|
||||
"""把 Markdown blocks 渲染为 flowable。
|
||||
|
||||
v0.5 upgrade: h1 triggers PageBreak; h2/h3 use keepWithNext; tables splitByRow.
|
||||
v0.6 升级:
|
||||
- 跳过正文开头的封面 H1 + 紧跟的元信息段(由 build_cover 独立生成,避免重复)
|
||||
- 识别"目录"占位段落 → 自动生成 TOC
|
||||
- 识别"参考文献"占位段落 → 自动读 sources.jsonl 生成 GB/T 7714 列表
|
||||
- H1 triggers PageBreak;H2/H3 keepWithNext;表格 splitByRow
|
||||
"""
|
||||
story = []
|
||||
first_h1 = True
|
||||
story: list = []
|
||||
first_h1_seen = False # 是否已跳过正文首个 H1
|
||||
skipping_cover_meta = False # 是否在吞掉封面元信息段
|
||||
|
||||
# Track whether we're in a special section that uses different body style
|
||||
# Summary 样式
|
||||
in_summary = False
|
||||
|
||||
for block in blocks:
|
||||
if block.kind == "h1":
|
||||
# PageBreak before every h1 EXCEPT the very first
|
||||
if not first_h1:
|
||||
story.append(PageBreak())
|
||||
first_h1 = False
|
||||
# 准备跳过标志:标题级别下一个 "目录""参考文献" 见到时替换掉它(包含其下紧跟的占位段)
|
||||
# 采用简单索引遍历以便向前看。
|
||||
i = 0
|
||||
n = len(blocks)
|
||||
while i < n:
|
||||
block = blocks[i]
|
||||
|
||||
# Check if this is Executive Summary / 执行摘要 - use summary style for following body
|
||||
# --- 跳过正文首个 H1(封面标题)+ 紧跟的元信息/hr ---
|
||||
if not first_h1_seen and block.kind == "h1":
|
||||
first_h1_seen = True
|
||||
skipping_cover_meta = True
|
||||
i += 1
|
||||
continue
|
||||
if skipping_cover_meta:
|
||||
# 吞掉 p(元信息)、hr、quote(副标题可能被当成加粗段)
|
||||
# 遇到 h1/h2/h3 就停止吞
|
||||
if block.kind in ("h1", "h2", "h3"):
|
||||
skipping_cover_meta = False
|
||||
# 不 continue,让当前 block 正常处理
|
||||
elif block.kind == "p" and is_cover_frontmatter(block.content):
|
||||
i += 1
|
||||
continue
|
||||
elif block.kind in ("hr", "quote", "p", "bullet"):
|
||||
# 第一个 hr 标记封面结束
|
||||
if block.kind == "hr":
|
||||
skipping_cover_meta = False
|
||||
i += 1
|
||||
continue
|
||||
# 普通段落:如果不是封面元信息,就认为封面已结束
|
||||
if block.kind == "p":
|
||||
skipping_cover_meta = False
|
||||
# fall through to normal handling
|
||||
else:
|
||||
i += 1
|
||||
continue
|
||||
else:
|
||||
i += 1
|
||||
continue
|
||||
|
||||
# --- H1 处理(非首个)---
|
||||
if block.kind == "h1":
|
||||
story.append(PageBreak())
|
||||
content = block.content
|
||||
if any(keyword in content for keyword in ["执行摘要", "Executive Summary", "管理层摘要"]):
|
||||
if any(k in content for k in ("执行摘要", "Executive Summary", "管理层摘要")):
|
||||
in_summary = True
|
||||
else:
|
||||
in_summary = False
|
||||
|
||||
# 目录 / 参考文献:替换为自动生成的内容
|
||||
title_low = content.strip().lower()
|
||||
if any(s in title_low for s in ("目录", "table of contents")):
|
||||
story.extend(build_toc(blocks, styles))
|
||||
# 跳过紧随其后的占位段
|
||||
j = i + 1
|
||||
while j < n and blocks[j].kind == "p" and _TOC_PLACEHOLDER_RE.search(blocks[j].content):
|
||||
j += 1
|
||||
i = j
|
||||
continue
|
||||
if any(s in title_low for s in ("参考文献", "references")):
|
||||
story.extend(build_references(blocks, sources_path, styles))
|
||||
j = i + 1
|
||||
while j < n and blocks[j].kind == "p" and _REF_PLACEHOLDER_RE.search(blocks[j].content):
|
||||
j += 1
|
||||
i = j
|
||||
continue
|
||||
|
||||
story.append(Paragraph(md_inline_to_rl(content), styles["h1"]))
|
||||
elif block.kind == "h2":
|
||||
i += 1
|
||||
continue
|
||||
|
||||
# --- H2 同样检测占位符 ---
|
||||
if block.kind == "h2":
|
||||
title_low = block.content.strip().lower()
|
||||
if any(s in title_low for s in ("目录", "table of contents")):
|
||||
story.extend(build_toc(blocks, styles))
|
||||
j = i + 1
|
||||
while j < n and blocks[j].kind == "p" and _TOC_PLACEHOLDER_RE.search(blocks[j].content):
|
||||
j += 1
|
||||
i = j
|
||||
continue
|
||||
if any(s in title_low for s in ("参考文献", "references")):
|
||||
story.extend(build_references(blocks, sources_path, styles))
|
||||
j = i + 1
|
||||
while j < n and blocks[j].kind == "p" and _REF_PLACEHOLDER_RE.search(blocks[j].content):
|
||||
j += 1
|
||||
i = j
|
||||
continue
|
||||
|
||||
story.append(Paragraph(md_inline_to_rl(block.content), styles["h2"]))
|
||||
elif block.kind == "h3":
|
||||
i += 1
|
||||
continue
|
||||
|
||||
if block.kind == "h3":
|
||||
story.append(Paragraph(md_inline_to_rl(block.content), styles["h3"]))
|
||||
elif block.kind == "p":
|
||||
# 跳过已识别但没有标题的孤立占位符(防御性)
|
||||
if _TOC_PLACEHOLDER_RE.search(block.content) or _REF_PLACEHOLDER_RE.search(block.content):
|
||||
i += 1
|
||||
continue
|
||||
style = styles["summary"] if in_summary else styles["body"]
|
||||
story.append(Paragraph(md_inline_to_rl(block.content), style))
|
||||
elif block.kind == "quote":
|
||||
@@ -580,6 +1092,8 @@ def build_body(blocks: List[Block], base_dir: Path, styles: StyleSheet1) -> List
|
||||
except Exception as e:
|
||||
story.append(Paragraph(f"[表格渲染失败: {e}]", styles["caption"]))
|
||||
|
||||
i += 1
|
||||
|
||||
return story
|
||||
|
||||
|
||||
@@ -588,8 +1102,8 @@ def build_body(blocks: List[Block], base_dir: Path, styles: StyleSheet1) -> List
|
||||
# ============================================================
|
||||
|
||||
def main():
|
||||
parser = argparse.ArgumentParser(description="Deep Research PDF Generator (v0.5)")
|
||||
parser.add_argument("--input", required=True, help="Input markdown (final_zh.md)")
|
||||
parser = argparse.ArgumentParser(description="Deep Research PDF Generator (v0.6)")
|
||||
parser.add_argument("--input", required=True, help="Input markdown (final_zh_polished.md)")
|
||||
parser.add_argument("--manifest", required=True, help="manifest.json path")
|
||||
parser.add_argument("--output", required=True, help="Output PDF path")
|
||||
parser.add_argument(
|
||||
@@ -597,6 +1111,11 @@ def main():
|
||||
default=".opencode/templates/fonts",
|
||||
help="Fonts directory",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--sources",
|
||||
default=None,
|
||||
help="sources.jsonl 路径(默认自动在 <project>/phase2/sources.jsonl 查找)",
|
||||
)
|
||||
args = parser.parse_args()
|
||||
|
||||
md_path = Path(args.input)
|
||||
@@ -610,6 +1129,19 @@ def main():
|
||||
print(f"ERROR: {label} not found: {p}", file=sys.stderr)
|
||||
sys.exit(1)
|
||||
|
||||
# Sources.jsonl 自动发现
|
||||
if args.sources:
|
||||
sources_path = Path(args.sources)
|
||||
else:
|
||||
# 默认 <project_root>/phase2/sources.jsonl
|
||||
project_root = manifest_path.parent
|
||||
candidate = project_root / "phase2" / "sources.jsonl"
|
||||
sources_path = candidate if candidate.exists() else None
|
||||
|
||||
if sources_path and not sources_path.exists():
|
||||
print(f"WARNING: sources file not found: {sources_path}", file=sys.stderr)
|
||||
sources_path = None
|
||||
|
||||
# Register fonts and build styles
|
||||
register_fonts(fonts_dir)
|
||||
styles = build_styles()
|
||||
@@ -650,11 +1182,12 @@ def main():
|
||||
])
|
||||
|
||||
# Assemble story
|
||||
# 注意:封面只从 manifest 构建,正文中的封面 H1+元信息会被 build_body 自动跳过。
|
||||
# 免责声明来自 Markdown(## 免责声明),不再从 manifest 额外构建(避免重复)。
|
||||
story: List = []
|
||||
story.extend(build_cover(manifest, styles))
|
||||
story.extend(build_cover(manifest, blocks, styles))
|
||||
story.append(NextPageTemplate("normal"))
|
||||
story.extend(build_disclaimer(manifest, styles))
|
||||
story.extend(build_body(blocks, md_path.parent, styles))
|
||||
story.extend(build_body(blocks, md_path.parent, styles, sources_path=sources_path))
|
||||
|
||||
# Build
|
||||
doc.build(story)
|
||||
@@ -665,6 +1198,10 @@ def main():
|
||||
print(f" Size: {size / 1024:.1f} KB")
|
||||
print(f" Fonts: {len(FONT_MAP)}")
|
||||
print(f" Blocks: {len(blocks)}")
|
||||
if sources_path:
|
||||
print(f" Sources: {sources_path}")
|
||||
else:
|
||||
print(f" Sources: (none — references section will show placeholder)")
|
||||
|
||||
if size < 500 * 1024:
|
||||
print(f" WARNING: PDF size < 500KB, fonts may not be properly embedded", file=sys.stderr)
|
||||
|
||||
Reference in New Issue
Block a user