v0.6-wip: polish pipeline + PDF template fixes
Phase 4 \u6da6\u8272\u5c42\u4e0e PDF \u6a21\u677f\u4fee\u590d\uff0c\u63a5\u7740\u4e0a\u4e00\u4e2a commit\u3002 polish.py\uff08\u65b0\u589e\uff09\uff1a - \u548c translate.py \u5bf9\u79f0\uff0c\u6309 H2 section \u5207\u5757 \u2192 \u5faa\u73af\u6da6\u8272 \u2192 \u62fc\u63a5 - \u4f7f\u7528 <<<POLISHED>>>/<<<NOTES>>> \u5206\u9694\u7b26 prompt\uff08\u907f\u5f00 Markdown-in-JSON \u95ee\u9898\uff09 - \u65ad\u70b9\u7eed\u4f20\u3001\u6a21\u578b\u81ea\u8bc4\u6ce8\u8bb0\u843d\u76d8 polish_notes.jsonl - \u5728\u53cc\u9776\u70b9 RNAi \u9879\u76ee\u8dd1\u901a\uff1a60 \u5757\u5168\u6210\u529f\uff0c10.7 \u5206\u949f\uff0c$1.20\uff0c\u5b57\u6570 -0.2% report-template.py\uff08\u5927\u6539\u4e00\u6279 P0 bug\uff09\uff1a - \u5b57\u4f53\u6ce8\u518c\u652f\u6301 fonts/ttf/ \u5b50\u76ee\u5f55\uff08\u89e3\u51b3 OTF PostScript outlines \u4e0d\u517c\u5bb9\uff09 - \u5220\u9664 build_disclaimer \u91cd\u590d\u8c03\u7528\uff08\u514d\u8d23\u58f0\u660e\u4ece Markdown \u8bfb\uff0cmanifest \u4e0d\u518d\u91cd\u590d\uff09 - build_body \u81ea\u52a8\u8df3\u8fc7\u6b63\u6587\u9996\u4e2a H1+\u5c01\u9762\u5143\u4fe1\u606f\u6bb5\uff08\u4e0e\u5c01\u9762\u91cd\u590d\uff09 - \u5360\u4f4d\u7b26 \u201c\u76ee\u5f55\u5c06\u5728\u6700\u7ec8\u6e32\u67d3\u65f6\u81ea\u52a8\u751f\u6210\u201d \u2192 \u81ea\u52a8\u751f\u6210 TOC - \u5360\u4f4d\u7b26 \u201c\u5b8c\u6574\u7f16\u53f7\u53c2\u8003\u6587\u732e\u5217\u8868\u2026\u201d \u2192 \u4ece phase2/sources.jsonl \u81ea\u52a8\u751f\u6210 GB/T 7714 \u683c\u5f0f\u5f15\u6587 - src \u4e0a\u6807\u6b63\u5219\u6269\u5c55\uff1a\u652f\u6301 src_A14 / src_B-18 \u7b49\u5b57\u6bcd+\u6570\u5b57\u7ec4\u5408 ID\uff08\u539f\u53ea\u652f\u6301 src_\d+\uff09 - Unicode \u4e0a/\u4e0b\u6807\u8f6c <super>/<sub>\uff1a10\u2076 \u2192 10<super>6</super>\uff08\u601d\u6e90\u5b57\u4f53\u5b50\u96c6\u4e0d\u542b\u4e0a\u6807\u5b57\u5f62\uff0c\u5426\u5219\u6e32\u67d3\u65b9\u6846\uff09 - \u4e2d\u82f1\u6df7\u6392\u81ea\u52a8\u52a0\u7a7a\u683c\uff08CJK \u2194 [A-Za-z0-9] \u8fb9\u754c\uff09 - \u8868\u683c\u6837\u5f0f\u91cd\u505a\uff1atable-header/table-cell/table-cell-center\uff1b\u5782\u76f4\u5c45\u4e2d\uff1b\u77ed cell\uff08\u7eaf\u6570\u5b57/\u77ed\u6807\u7b7e\uff09\u6c34\u5e73\u5c45\u4e2d\uff1b\u957f cell \u81ea\u52a8 CJK \u6362\u884c - TOC \u672b\u5c3e PageBreak\uff08\u76ee\u5f55\u72ec\u5360\u6574\u9875\uff09 \u5df2\u77e5\u672a\u4fee\u590d\uff1a - Maywavee \u662f LLM \u5728 dr-analyst \u9636\u6bb5\u7f16\u9020\uff0c\u6b63\u786e\u4e3a Mabwell\uff08\u8fc8\u5a01\u751f\u7269\uff09\u3002\u4fe1\u6e90\u4fa7 bug\uff0c\u9700\u5728\u540e\u7eed build_glossary.py \u4e2d\u505a\u4e8b\u5b9e\u6838\u67e5\u3002 - \u6b63\u6587 101 \u4e2a src_id\u3001sources.jsonl \u53ea\u670947 \u4e2a\u3001\u4ea4\u96c6 39 \u4e2a\u2014\u2014\u662f v0.4 \u9057\u7559\u7684\u6ce8\u5165 bug\uff0cbuild_references \u73b0\u5728\u4f1a\u5217\u51fa\u7f3a\u5931\u7684 id \u4f9b\u4eba\u5de5\u6838\u5bf9 - \u53cd\u9a73\u8bc1\u636e\u6bb5\u683c\u5f0f\u4e0d\u7edf\u4e00\u662f dr-analyst/skill \u89c4\u8303\u95ee\u9898\uff0c\u4e0b\u4e00\u6279\u6539 skill Co-authored-by: User <human>
This commit is contained in:
@@ -0,0 +1,262 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Phase 4 中文润色:按 H2 section 切块 → 逐块润色 → 拼接 → 写盘。
|
||||
|
||||
用法:
|
||||
uv run python scripts/polish.py <project_slug>
|
||||
|
||||
架构跟 translate.py 一致:
|
||||
- Python 做切块/循环/重试/断点续传
|
||||
- LLM 只做"润色这一段",单次 output token 远低于上限
|
||||
- 结果落在 phase4/zh_polished_chunks/<anchor>.md,重跑只补缺
|
||||
- 最终拼接写入 phase4/final_zh_polished.md(默认覆写 final_zh.md 的副本)
|
||||
|
||||
区别只在:
|
||||
- 输入是中文(final_zh.md),输出还是中文
|
||||
- 不维护术语表(翻译阶段已经固定了)
|
||||
- 会附带一份 notes.jsonl 记录模型发现的异常
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import sys
|
||||
import time
|
||||
from pathlib import Path
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
|
||||
|
||||
from scripts.lib.markdown_chunker import (
|
||||
MarkdownBlock,
|
||||
count_chinese_chars,
|
||||
split_by_headers,
|
||||
)
|
||||
from scripts.lib.zenmux_client import ZenMuxClient, ZenMuxError, load_secrets
|
||||
|
||||
DEFAULT_MODEL = "anthropic/claude-sonnet-4.6"
|
||||
MODEL_MAX_TOKENS = {
|
||||
"anthropic/claude-sonnet-4.6": 32000,
|
||||
"anthropic/claude-sonnet-4.5": 32000,
|
||||
"anthropic/claude-opus-4.7": 32000,
|
||||
"anthropic/claude-opus-4.6": 32000,
|
||||
"anthropic/claude-haiku-4.5": 16000,
|
||||
}
|
||||
PROMPT_FILE = Path(__file__).parent / "prompts" / "polish_system.txt"
|
||||
|
||||
|
||||
def resolve_project(arg: str) -> Path:
|
||||
p = Path(arg)
|
||||
if p.is_dir():
|
||||
return p
|
||||
cand = Path.cwd() / "projects" / arg
|
||||
if cand.is_dir():
|
||||
return cand
|
||||
raise SystemExit(f"project not found: {arg}")
|
||||
|
||||
|
||||
def parse_polish_response(text: str) -> tuple[str, str]:
|
||||
"""解析 <<<POLISHED>>>/<<<NOTES>>> 格式。返回 (polished, notes)。"""
|
||||
p_start = text.find("<<<POLISHED>>>")
|
||||
p_end = text.find("<<<END_POLISHED>>>")
|
||||
if p_start == -1 or p_end == -1 or p_end <= p_start:
|
||||
raise ValueError(f"missing <<<POLISHED>>> markers: {text[:300]}")
|
||||
polished = text[p_start + len("<<<POLISHED>>>"): p_end].strip("\r\n")
|
||||
|
||||
n_start = text.find("<<<NOTES>>>")
|
||||
n_end = text.find("<<<END_NOTES>>>")
|
||||
notes = ""
|
||||
if n_start != -1 and n_end != -1 and n_end > n_start:
|
||||
notes = text[n_start + len("<<<NOTES>>>"): n_end].strip()
|
||||
return polished, notes
|
||||
|
||||
|
||||
def build_user_prompt(block: MarkdownBlock) -> str:
|
||||
level_hint = f"H{block.level}" if block.level >= 1 else "frontmatter (无标题)"
|
||||
return (
|
||||
f"# 待润色的中文块(Markdown, {level_hint})\n"
|
||||
"请按系统提示的规则润色下面这段中文。严格使用指定的分隔符格式输出。\n\n"
|
||||
"----- BEGIN BLOCK -----\n"
|
||||
f"{block.content}\n"
|
||||
"----- END BLOCK -----\n"
|
||||
)
|
||||
|
||||
|
||||
def polish_block(
|
||||
client: ZenMuxClient,
|
||||
block: MarkdownBlock,
|
||||
*,
|
||||
model: str,
|
||||
system_prompt: str,
|
||||
temperature: float,
|
||||
) -> tuple[str, str]:
|
||||
user = build_user_prompt(block)
|
||||
max_tok = MODEL_MAX_TOKENS.get(model, 16000)
|
||||
raw = client.chat_complete(
|
||||
model=model,
|
||||
system=system_prompt,
|
||||
user=user,
|
||||
temperature=temperature,
|
||||
max_tokens=max_tok,
|
||||
tag=f"polish:{block.anchor}",
|
||||
)
|
||||
try:
|
||||
polished, notes = parse_polish_response(raw)
|
||||
except Exception as e:
|
||||
raise RuntimeError(
|
||||
f"bad response format for block {block.anchor}: {e}\nraw head: {raw[:300]}"
|
||||
)
|
||||
if not polished.strip():
|
||||
raise RuntimeError(f"empty polished content for block {block.anchor}")
|
||||
return polished.rstrip(), notes
|
||||
|
||||
|
||||
def main() -> int:
|
||||
parser = argparse.ArgumentParser(description="Phase 4 中文润色(按 H2 切块循环)")
|
||||
parser.add_argument("project", help="项目 slug 或完整路径")
|
||||
parser.add_argument(
|
||||
"--source",
|
||||
default="phase4/final_zh.md",
|
||||
help="中文源(默认 phase4/final_zh.md,即 translate.py 的产物)",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--output",
|
||||
default="phase4/final_zh_polished.md",
|
||||
help="润色后输出(默认 phase4/final_zh_polished.md)",
|
||||
)
|
||||
parser.add_argument("--model", default=DEFAULT_MODEL)
|
||||
parser.add_argument("--temperature", type=float, default=0.4)
|
||||
parser.add_argument(
|
||||
"--force",
|
||||
action="store_true",
|
||||
help="忽略 zh_polished_chunks 缓存,强制重润",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--only",
|
||||
default=None,
|
||||
help="只润色指定 order(逗号分隔),例如 --only 2,7,18",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--limit",
|
||||
type=int,
|
||||
default=None,
|
||||
help="最多润色前 N 个未缓存的块(调试用)",
|
||||
)
|
||||
args = parser.parse_args()
|
||||
|
||||
load_secrets()
|
||||
project_root = resolve_project(args.project)
|
||||
src_path = project_root / args.source
|
||||
out_path = project_root / args.output
|
||||
if not src_path.exists():
|
||||
raise SystemExit(f"source not found: {src_path}. 请先跑 translate.py")
|
||||
|
||||
chunks_dir = project_root / "phase4" / "zh_polished_chunks"
|
||||
chunks_dir.mkdir(parents=True, exist_ok=True)
|
||||
logs_dir = project_root / "phase4" / "logs"
|
||||
log_file = logs_dir / "polish.jsonl"
|
||||
notes_file = project_root / "phase4" / "polish_notes.jsonl"
|
||||
|
||||
system_prompt = PROMPT_FILE.read_text(encoding="utf-8")
|
||||
text = src_path.read_text(encoding="utf-8")
|
||||
blocks = split_by_headers(text, max_level=2)
|
||||
|
||||
only_orders: set[int] | None = None
|
||||
if args.only:
|
||||
only_orders = {int(a.strip()) for a in args.only.split(",") if a.strip()}
|
||||
|
||||
total_cn = sum(count_chinese_chars(b.content) for b in blocks)
|
||||
print(f"Source: {src_path.relative_to(project_root)}")
|
||||
print(f"Blocks: {len(blocks)} | total Chinese chars: {total_cn:,}")
|
||||
print(f"Model: {args.model} | temperature: {args.temperature}")
|
||||
print()
|
||||
|
||||
start = time.time()
|
||||
translated_this_run = 0
|
||||
notes_records: list[dict] = []
|
||||
with ZenMuxClient(log_file=log_file) as client:
|
||||
for b in blocks:
|
||||
chunk_path = chunks_dir / f"{b.order:03d}-{b.anchor}.md"
|
||||
if only_orders is not None and b.order not in only_orders:
|
||||
continue
|
||||
if chunk_path.exists() and not args.force:
|
||||
print(f" [ok ] #{b.order:03d} {b.short_title} (cached)")
|
||||
continue
|
||||
if args.limit is not None and translated_this_run >= args.limit:
|
||||
continue
|
||||
|
||||
label = f"#{b.order:03d} L{b.level} {count_chinese_chars(b.content):>4}字 {b.short_title}"
|
||||
print(f" [... ] {label} ", end="", flush=True)
|
||||
t0 = time.time()
|
||||
try:
|
||||
polished, notes = polish_block(
|
||||
client,
|
||||
b,
|
||||
model=args.model,
|
||||
system_prompt=system_prompt,
|
||||
temperature=args.temperature,
|
||||
)
|
||||
except (ZenMuxError, RuntimeError) as e:
|
||||
print(f"\n [FAIL] {label}\n {e}")
|
||||
continue
|
||||
elapsed = time.time() - t0
|
||||
|
||||
chunk_path.write_text(polished + "\n", encoding="utf-8")
|
||||
cn = count_chinese_chars(polished)
|
||||
before_cn = count_chinese_chars(b.content)
|
||||
delta = cn - before_cn
|
||||
sign = "+" if delta >= 0 else ""
|
||||
translated_this_run += 1
|
||||
if notes:
|
||||
notes_records.append(
|
||||
{"order": b.order, "anchor": b.anchor, "title": b.short_title, "notes": notes}
|
||||
)
|
||||
print(f"\r [done] {label} → {cn}字 ({sign}{delta}, {elapsed:4.1f}s)")
|
||||
|
||||
# 汇总
|
||||
merged: list[str] = []
|
||||
missing: list[str] = []
|
||||
for b in blocks:
|
||||
chunk_path = chunks_dir / f"{b.order:03d}-{b.anchor}.md"
|
||||
if not chunk_path.exists():
|
||||
missing.append(f"#{b.order:03d} {b.short_title}")
|
||||
continue
|
||||
merged.append(chunk_path.read_text(encoding="utf-8").rstrip())
|
||||
|
||||
partial = only_orders is not None or args.limit is not None
|
||||
if missing:
|
||||
print(f"\n⚠ 缺失 {len(missing)} 块:")
|
||||
for m in missing[:10]:
|
||||
print(f" - {m}")
|
||||
if len(missing) > 10:
|
||||
print(f" ... 还有 {len(missing) - 10} 块")
|
||||
if partial:
|
||||
print("(partial 模式:--only / --limit 生效,未生成 final_zh_polished.md)")
|
||||
else:
|
||||
print("重新运行本脚本即可补润(已润的会跳过)。")
|
||||
print(client.usage.summary())
|
||||
return 1
|
||||
|
||||
out_path.parent.mkdir(parents=True, exist_ok=True)
|
||||
out_path.write_text("\n\n".join(merged) + "\n", encoding="utf-8")
|
||||
|
||||
if notes_records:
|
||||
notes_file.write_text(
|
||||
"\n".join(json.dumps(r, ensure_ascii=False) for r in notes_records) + "\n",
|
||||
encoding="utf-8",
|
||||
)
|
||||
print(f"\n发现 {len(notes_records)} 条润色笔记 → {notes_file.relative_to(project_root)}")
|
||||
|
||||
final_text = out_path.read_text(encoding="utf-8")
|
||||
cn = count_chinese_chars(final_text)
|
||||
total_time = time.time() - start
|
||||
print()
|
||||
print(f"✓ 输出:{out_path.relative_to(project_root)}")
|
||||
print(f" 润色前字数: {total_cn:,}")
|
||||
print(f" 润色后字数: {cn:,} (变化 {cn - total_cn:+d}, {(cn/total_cn - 1)*100:+.1f}%)")
|
||||
print(f" 耗时: {total_time:.1f}s")
|
||||
print(client.usage.summary())
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
@@ -3,40 +3,45 @@
|
||||
## 不可违反的规则
|
||||
|
||||
1. **保留所有引用标注** `[src_xxx]`,位置可以微调但不得删除或改写。
|
||||
2. **保留所有数字、百分比、日期、单位、化学式、药物代号**,一字不改。
|
||||
2. **保留所有数字、百分比、日期、单位、化学式、药物代号、机构名称**,一字不改。
|
||||
3. **保留 Markdown 结构**:输入是什么标题层级(#/##/###)输出就是什么。表格的 `|` 分隔符和列数不变。列表符号(-, *, 1.)不变。
|
||||
4. **保留段落数量**:不要合并或拆分段落。每段原文输出一段译文。
|
||||
5. **专有名词首次出现保持"中文(English)"格式**;如果译文里这个术语已经这样标了就别改。
|
||||
4. **保留段落数量**:不要合并或拆分段落。每段原文对应一段输出。
|
||||
5. **专有名词首次出现保持"中文(English)"格式**;如果译文里这个术语已经这样标了就别改,也不要删掉。
|
||||
6. **不改变论点、结论、数据、案例**。只改语言表达。
|
||||
|
||||
## 要去掉的"AI 味/翻译腔"表征
|
||||
|
||||
- 空泛套话:随着…不断发展、综上所述、本质上、从根本上、跃迁、赋能、落地、抓手
|
||||
- 翻译腔:对于…来说、在…方面、…的话、值得注意的是、众所周知、毫无疑问
|
||||
- 翻译腔开头:值得注意的是、众所周知、毫无疑问、不难看出
|
||||
- 冗余连词开头:此外、而且、并且、再者(英文 moreover / furthermore / additionally 的直译残留)
|
||||
- 过度强调:非常、十分、极其、特别(没有数据支撑时)
|
||||
- 介词短语套用:对于…来说、在…方面、…的话、关于…这一点
|
||||
- 过度强调:非常、十分、极其、特别、尤为(没有数据支撑时)
|
||||
- 长串的"的"字("X 的 Y 的 Z 的 W")改为短句
|
||||
- 被动语态("被…所…")尽量改主动
|
||||
- 把"我们"去掉,除非是真的作者第一人称立场
|
||||
- 被动语态("被…所…"、"……得以……")尽量改主动
|
||||
- 把"我们"删掉,除非真是作者第一人称立场
|
||||
|
||||
## 要加强的中文表达特征
|
||||
|
||||
- 句子节奏变化:短句和中句交替,避免一路长句
|
||||
- 句子节奏变化:短句与中句交替,避免一路长句
|
||||
- 动词前置:中文偏好动词驱动,不要像英文那样把名词短语堆在主语
|
||||
- 具体化:如果翻译留下了模糊的"相关", "一定的", "较大的",尽量换成源文里的具体含义
|
||||
- 段落内逻辑词(因此、相比之下、代价是)用得准确
|
||||
- 具体化:如果译文留下了模糊的"相关"、"一定的"、"较大的",根据上下文换成源文里的具体含义;若无依据,保留原样不硬改
|
||||
- 段落内逻辑词(因此、相比之下、代价是)用得准确、用得克制
|
||||
|
||||
## 特殊情况
|
||||
|
||||
- 如果段落里有"译者注"、"TRANSLATOR_NOTE:" 之类残留,删除后自然连接上下文
|
||||
- 如果出现明显的翻译错误(中文表达反了意思),修正它,但在输出的 `notes` 字段里记一笔
|
||||
- 如果出现明显的翻译错误(中文表达反了意思),修正它,但在输出的 `notes` 中记一笔
|
||||
- 如果某句过于生硬又不确定原意,保守处理(小改),不要激进重写
|
||||
|
||||
## 输出格式
|
||||
## 输出格式(严格)
|
||||
|
||||
返回一行 JSON,两个键:
|
||||
输出完全按下面的分隔符格式,前后不得有任何多余字符、说明或代码围栏。
|
||||
|
||||
- `polished`: 完整润色后的 Markdown 块,字符串。
|
||||
- `notes`: 字符串,最多两句。若无异常就给空串。
|
||||
<<<POLISHED>>>
|
||||
...润色后的完整 Markdown 块,逐字复制,包括标题行...
|
||||
<<<END_POLISHED>>>
|
||||
<<<NOTES>>>
|
||||
...最多两句话的异常说明;若无异常就留空...
|
||||
<<<END_NOTES>>>
|
||||
|
||||
**不要**用 ```json 包裹。不要加 JSON 外的任何字符。
|
||||
`<<<POLISHED>>>...<<<END_POLISHED>>>` 之间是原生 Markdown(无需转义)。
|
||||
|
||||
Reference in New Issue
Block a user