Files
deep_research/scripts/polish.py
T
kaiandUser <human> a86010e9a7 v0.6-wip: polish pipeline + PDF template fixes
Phase 4 \u6da6\u8272\u5c42\u4e0e PDF \u6a21\u677f\u4fee\u590d\uff0c\u63a5\u7740\u4e0a\u4e00\u4e2a commit\u3002

polish.py\uff08\u65b0\u589e\uff09\uff1a
- \u548c translate.py \u5bf9\u79f0\uff0c\u6309 H2 section \u5207\u5757 \u2192 \u5faa\u73af\u6da6\u8272 \u2192 \u62fc\u63a5
- \u4f7f\u7528 <<<POLISHED>>>/<<<NOTES>>> \u5206\u9694\u7b26 prompt\uff08\u907f\u5f00 Markdown-in-JSON \u95ee\u9898\uff09
- \u65ad\u70b9\u7eed\u4f20\u3001\u6a21\u578b\u81ea\u8bc4\u6ce8\u8bb0\u843d\u76d8 polish_notes.jsonl
- \u5728\u53cc\u9776\u70b9 RNAi \u9879\u76ee\u8dd1\u901a\uff1a60 \u5757\u5168\u6210\u529f\uff0c10.7 \u5206\u949f\uff0c$1.20\uff0c\u5b57\u6570 -0.2%

report-template.py\uff08\u5927\u6539\u4e00\u6279 P0 bug\uff09\uff1a
- \u5b57\u4f53\u6ce8\u518c\u652f\u6301 fonts/ttf/ \u5b50\u76ee\u5f55\uff08\u89e3\u51b3 OTF PostScript outlines \u4e0d\u517c\u5bb9\uff09
- \u5220\u9664 build_disclaimer \u91cd\u590d\u8c03\u7528\uff08\u514d\u8d23\u58f0\u660e\u4ece Markdown \u8bfb\uff0cmanifest \u4e0d\u518d\u91cd\u590d\uff09
- build_body \u81ea\u52a8\u8df3\u8fc7\u6b63\u6587\u9996\u4e2a H1+\u5c01\u9762\u5143\u4fe1\u606f\u6bb5\uff08\u4e0e\u5c01\u9762\u91cd\u590d\uff09
- \u5360\u4f4d\u7b26 \u201c\u76ee\u5f55\u5c06\u5728\u6700\u7ec8\u6e32\u67d3\u65f6\u81ea\u52a8\u751f\u6210\u201d \u2192 \u81ea\u52a8\u751f\u6210 TOC
- \u5360\u4f4d\u7b26 \u201c\u5b8c\u6574\u7f16\u53f7\u53c2\u8003\u6587\u732e\u5217\u8868\u2026\u201d \u2192 \u4ece phase2/sources.jsonl \u81ea\u52a8\u751f\u6210 GB/T 7714 \u683c\u5f0f\u5f15\u6587
- src \u4e0a\u6807\u6b63\u5219\u6269\u5c55\uff1a\u652f\u6301 src_A14 / src_B-18 \u7b49\u5b57\u6bcd+\u6570\u5b57\u7ec4\u5408 ID\uff08\u539f\u53ea\u652f\u6301 src_\d+\uff09
- Unicode \u4e0a/\u4e0b\u6807\u8f6c <super>/<sub>\uff1a10\u2076 \u2192 10<super>6</super>\uff08\u601d\u6e90\u5b57\u4f53\u5b50\u96c6\u4e0d\u542b\u4e0a\u6807\u5b57\u5f62\uff0c\u5426\u5219\u6e32\u67d3\u65b9\u6846\uff09
- \u4e2d\u82f1\u6df7\u6392\u81ea\u52a8\u52a0\u7a7a\u683c\uff08CJK \u2194 [A-Za-z0-9] \u8fb9\u754c\uff09
- \u8868\u683c\u6837\u5f0f\u91cd\u505a\uff1atable-header/table-cell/table-cell-center\uff1b\u5782\u76f4\u5c45\u4e2d\uff1b\u77ed cell\uff08\u7eaf\u6570\u5b57/\u77ed\u6807\u7b7e\uff09\u6c34\u5e73\u5c45\u4e2d\uff1b\u957f cell \u81ea\u52a8 CJK \u6362\u884c
- TOC \u672b\u5c3e PageBreak\uff08\u76ee\u5f55\u72ec\u5360\u6574\u9875\uff09

\u5df2\u77e5\u672a\u4fee\u590d\uff1a
- Maywavee \u662f LLM \u5728 dr-analyst \u9636\u6bb5\u7f16\u9020\uff0c\u6b63\u786e\u4e3a Mabwell\uff08\u8fc8\u5a01\u751f\u7269\uff09\u3002\u4fe1\u6e90\u4fa7 bug\uff0c\u9700\u5728\u540e\u7eed build_glossary.py \u4e2d\u505a\u4e8b\u5b9e\u6838\u67e5\u3002
- \u6b63\u6587 101 \u4e2a src_id\u3001sources.jsonl \u53ea\u670947 \u4e2a\u3001\u4ea4\u96c6 39 \u4e2a\u2014\u2014\u662f v0.4 \u9057\u7559\u7684\u6ce8\u5165 bug\uff0cbuild_references \u73b0\u5728\u4f1a\u5217\u51fa\u7f3a\u5931\u7684 id \u4f9b\u4eba\u5de5\u6838\u5bf9
- \u53cd\u9a73\u8bc1\u636e\u6bb5\u683c\u5f0f\u4e0d\u7edf\u4e00\u662f dr-analyst/skill \u89c4\u8303\u95ee\u9898\uff0c\u4e0b\u4e00\u6279\u6539 skill

Co-authored-by: User <human>
2026-04-22 12:58:07 +08:00

263 lines
9.2 KiB
Python
Raw Permalink Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
#!/usr/bin/env python3
"""Phase 4 中文润色:按 H2 section 切块 → 逐块润色 → 拼接 → 写盘。
用法:
uv run python scripts/polish.py <project_slug>
架构跟 translate.py 一致:
- Python 做切块/循环/重试/断点续传
- LLM 只做"润色这一段",单次 output token 远低于上限
- 结果落在 phase4/zh_polished_chunks/<anchor>.md,重跑只补缺
- 最终拼接写入 phase4/final_zh_polished.md(默认覆写 final_zh.md 的副本)
区别只在:
- 输入是中文(final_zh.md),输出还是中文
- 不维护术语表(翻译阶段已经固定了)
- 会附带一份 notes.jsonl 记录模型发现的异常
"""
from __future__ import annotations
import argparse
import json
import sys
import time
from pathlib import Path
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
from scripts.lib.markdown_chunker import (
MarkdownBlock,
count_chinese_chars,
split_by_headers,
)
from scripts.lib.zenmux_client import ZenMuxClient, ZenMuxError, load_secrets
DEFAULT_MODEL = "anthropic/claude-sonnet-4.6"
MODEL_MAX_TOKENS = {
"anthropic/claude-sonnet-4.6": 32000,
"anthropic/claude-sonnet-4.5": 32000,
"anthropic/claude-opus-4.7": 32000,
"anthropic/claude-opus-4.6": 32000,
"anthropic/claude-haiku-4.5": 16000,
}
PROMPT_FILE = Path(__file__).parent / "prompts" / "polish_system.txt"
def resolve_project(arg: str) -> Path:
p = Path(arg)
if p.is_dir():
return p
cand = Path.cwd() / "projects" / arg
if cand.is_dir():
return cand
raise SystemExit(f"project not found: {arg}")
def parse_polish_response(text: str) -> tuple[str, str]:
"""解析 <<<POLISHED>>>/<<<NOTES>>> 格式。返回 (polished, notes)。"""
p_start = text.find("<<<POLISHED>>>")
p_end = text.find("<<<END_POLISHED>>>")
if p_start == -1 or p_end == -1 or p_end <= p_start:
raise ValueError(f"missing <<<POLISHED>>> markers: {text[:300]}")
polished = text[p_start + len("<<<POLISHED>>>"): p_end].strip("\r\n")
n_start = text.find("<<<NOTES>>>")
n_end = text.find("<<<END_NOTES>>>")
notes = ""
if n_start != -1 and n_end != -1 and n_end > n_start:
notes = text[n_start + len("<<<NOTES>>>"): n_end].strip()
return polished, notes
def build_user_prompt(block: MarkdownBlock) -> str:
level_hint = f"H{block.level}" if block.level >= 1 else "frontmatter (无标题)"
return (
f"# 待润色的中文块(Markdown, {level_hint}\n"
"请按系统提示的规则润色下面这段中文。严格使用指定的分隔符格式输出。\n\n"
"----- BEGIN BLOCK -----\n"
f"{block.content}\n"
"----- END BLOCK -----\n"
)
def polish_block(
client: ZenMuxClient,
block: MarkdownBlock,
*,
model: str,
system_prompt: str,
temperature: float,
) -> tuple[str, str]:
user = build_user_prompt(block)
max_tok = MODEL_MAX_TOKENS.get(model, 16000)
raw = client.chat_complete(
model=model,
system=system_prompt,
user=user,
temperature=temperature,
max_tokens=max_tok,
tag=f"polish:{block.anchor}",
)
try:
polished, notes = parse_polish_response(raw)
except Exception as e:
raise RuntimeError(
f"bad response format for block {block.anchor}: {e}\nraw head: {raw[:300]}"
)
if not polished.strip():
raise RuntimeError(f"empty polished content for block {block.anchor}")
return polished.rstrip(), notes
def main() -> int:
parser = argparse.ArgumentParser(description="Phase 4 中文润色(按 H2 切块循环)")
parser.add_argument("project", help="项目 slug 或完整路径")
parser.add_argument(
"--source",
default="phase4/final_zh.md",
help="中文源(默认 phase4/final_zh.md,即 translate.py 的产物)",
)
parser.add_argument(
"--output",
default="phase4/final_zh_polished.md",
help="润色后输出(默认 phase4/final_zh_polished.md",
)
parser.add_argument("--model", default=DEFAULT_MODEL)
parser.add_argument("--temperature", type=float, default=0.4)
parser.add_argument(
"--force",
action="store_true",
help="忽略 zh_polished_chunks 缓存,强制重润",
)
parser.add_argument(
"--only",
default=None,
help="只润色指定 order(逗号分隔),例如 --only 2,7,18",
)
parser.add_argument(
"--limit",
type=int,
default=None,
help="最多润色前 N 个未缓存的块(调试用)",
)
args = parser.parse_args()
load_secrets()
project_root = resolve_project(args.project)
src_path = project_root / args.source
out_path = project_root / args.output
if not src_path.exists():
raise SystemExit(f"source not found: {src_path}. 请先跑 translate.py")
chunks_dir = project_root / "phase4" / "zh_polished_chunks"
chunks_dir.mkdir(parents=True, exist_ok=True)
logs_dir = project_root / "phase4" / "logs"
log_file = logs_dir / "polish.jsonl"
notes_file = project_root / "phase4" / "polish_notes.jsonl"
system_prompt = PROMPT_FILE.read_text(encoding="utf-8")
text = src_path.read_text(encoding="utf-8")
blocks = split_by_headers(text, max_level=2)
only_orders: set[int] | None = None
if args.only:
only_orders = {int(a.strip()) for a in args.only.split(",") if a.strip()}
total_cn = sum(count_chinese_chars(b.content) for b in blocks)
print(f"Source: {src_path.relative_to(project_root)}")
print(f"Blocks: {len(blocks)} | total Chinese chars: {total_cn:,}")
print(f"Model: {args.model} | temperature: {args.temperature}")
print()
start = time.time()
translated_this_run = 0
notes_records: list[dict] = []
with ZenMuxClient(log_file=log_file) as client:
for b in blocks:
chunk_path = chunks_dir / f"{b.order:03d}-{b.anchor}.md"
if only_orders is not None and b.order not in only_orders:
continue
if chunk_path.exists() and not args.force:
print(f" [ok ] #{b.order:03d} {b.short_title} (cached)")
continue
if args.limit is not None and translated_this_run >= args.limit:
continue
label = f"#{b.order:03d} L{b.level} {count_chinese_chars(b.content):>4}{b.short_title}"
print(f" [... ] {label} ", end="", flush=True)
t0 = time.time()
try:
polished, notes = polish_block(
client,
b,
model=args.model,
system_prompt=system_prompt,
temperature=args.temperature,
)
except (ZenMuxError, RuntimeError) as e:
print(f"\n [FAIL] {label}\n {e}")
continue
elapsed = time.time() - t0
chunk_path.write_text(polished + "\n", encoding="utf-8")
cn = count_chinese_chars(polished)
before_cn = count_chinese_chars(b.content)
delta = cn - before_cn
sign = "+" if delta >= 0 else ""
translated_this_run += 1
if notes:
notes_records.append(
{"order": b.order, "anchor": b.anchor, "title": b.short_title, "notes": notes}
)
print(f"\r [done] {label}{cn}字 ({sign}{delta}, {elapsed:4.1f}s)")
# 汇总
merged: list[str] = []
missing: list[str] = []
for b in blocks:
chunk_path = chunks_dir / f"{b.order:03d}-{b.anchor}.md"
if not chunk_path.exists():
missing.append(f"#{b.order:03d} {b.short_title}")
continue
merged.append(chunk_path.read_text(encoding="utf-8").rstrip())
partial = only_orders is not None or args.limit is not None
if missing:
print(f"\n⚠ 缺失 {len(missing)} 块:")
for m in missing[:10]:
print(f" - {m}")
if len(missing) > 10:
print(f" ... 还有 {len(missing) - 10} 块")
if partial:
print("partial 模式:--only / --limit 生效,未生成 final_zh_polished.md")
else:
print("重新运行本脚本即可补润(已润的会跳过)。")
print(client.usage.summary())
return 1
out_path.parent.mkdir(parents=True, exist_ok=True)
out_path.write_text("\n\n".join(merged) + "\n", encoding="utf-8")
if notes_records:
notes_file.write_text(
"\n".join(json.dumps(r, ensure_ascii=False) for r in notes_records) + "\n",
encoding="utf-8",
)
print(f"\n发现 {len(notes_records)} 条润色笔记 → {notes_file.relative_to(project_root)}")
final_text = out_path.read_text(encoding="utf-8")
cn = count_chinese_chars(final_text)
total_time = time.time() - start
print()
print(f"✓ 输出:{out_path.relative_to(project_root)}")
print(f" 润色前字数: {total_cn:,}")
print(f" 润色后字数: {cn:,} (变化 {cn - total_cn:+d}, {(cn/total_cn - 1)*100:+.1f}%)")
print(f" 耗时: {total_time:.1f}s")
print(client.usage.summary())
return 0
if __name__ == "__main__":
sys.exit(main())