#!/usr/bin/env python3 """一次性脚本:从用户 Markdown 书单解析并导入 SQLite。""" import re import sys import os # 让 import db 能找到项目模块 sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) os.chdir(os.path.dirname(os.path.abspath(__file__))) import db # ── 分组映射 ────────────────────────────────────────────────────────── GROUP_MAP = { "历史学": "历史学", "伊朗": "历史学", "中东": "历史学", "政治哲学": "政治哲学", "社会思想": "政治哲学", "科学": "科学&进化", "进化": "科学&进化", "人生意义": "心理&思维", "心理学": "心理&思维", "经济学": "经济&金融", "金融": "经济&金融", "文学": "文学", "传记": "文学", "古典诗学": "古典诗学", "文化": "古典诗学", "投资": "经济&金融", "交易": "经济&金融", "思维": "心理&思维", "工具": "心理&思维", "科技": "科技&AI", "AI": "科技&AI", "赫拉利": "科技&AI", "数学": "数学&科学史", "科学史": "数学&科学史", "伯克利": "推荐书单", "中国史": "中国史", "其他": "其他", "已读书目": "已读归档", "阅读优先级": "", } def infer_group(section_title): for key, group in GROUP_MAP.items(): if key in section_title: return group return "" def parse_status(content): if "✅" in content and "已读" in content: return "read" if "🔖" in content or "在读" in content: return "reading" return "to-read" def extract_scores(content): douban = None gr = None m = re.search(r'豆瓣[^0-9]*?(\d\.\d)', content) if m: douban = float(m.group(1)) m = re.search(r'GR[^0-9]*?(\d\.\d+)', content) if m: gr = float(m.group(1)) return douban, gr def extract_author(content): # 尝试 · 作者 · 模式 m = re.search(r'·\s*([^·\n]{2,30}?)\s*·', content) if m: author = m.group(1).strip() skip_words = ['豆瓣', 'GR', '评分', '文学', '历史', '科技', '哲学', '经济', '科学', '传记', '政治', '社会', '心理', '宇宙', '物理', '数学', '⭐', '在读', '已读', '待读', 'FT', 'NYT', '年度', '美国', '中国', '伊朗', '金融', '制度', '方志远', '经济制裁', '复杂', '海权', '越南', '中东', '法律', '清代'] if not any(kw in author for kw in skip_words): return author # 尝试 [] 内国籍后的作者: [美] 作者名 m = re.search(r'\[[\u4e00-\u9fff]+\]\s*([\u4e00-\u9fff·A-Za-z ]{2,20})', content) if m: return m.group(1).strip() return None def clean_title(raw): return raw.replace("**", "").strip() def parse_booklist(text): books = [] seen_titles = set() current_group = "" current_priority = "" lines = text.splitlines() i = 0 while i < len(lines): line = lines[i].rstrip() # ## 大标题 if line.startswith("## "): section = line.lstrip("# ").strip() g = infer_group(section) if g: current_group = g # 优先级检测 if "⭐⭐⭐" in section or "强烈推荐" in section: current_priority = "⭐⭐⭐" elif "⭐⭐" in section or "值得读" in section: current_priority = "⭐⭐" elif "⭐" in section or "可选" in section: current_priority = "⭐" else: current_priority = "" i += 1 continue # ⭐ 优先级行 if re.match(r'^⭐⭐⭐\s', line): current_priority = "⭐⭐⭐" i += 1 continue if re.match(r'^⭐⭐\s', line) and not line.startswith("⭐⭐⭐"): current_priority = "⭐⭐" i += 1 continue if re.match(r'^⭐\s', line) and not line.startswith("⭐⭐"): current_priority = "⭐" i += 1 continue # ### 子分组 if line.startswith("### "): sub = line.lstrip("# ").strip() sub_g = infer_group(sub) if sub_g: current_group = sub_g i += 1 continue # 书目行 m = re.match(r'^-\s*\[([ x/])\]\s*(.*)', line) if not m: i += 1 continue content = m.group(2).strip() # 提取书名 title_match = re.search(r'\*\*(.+?)\*\*', content) if not title_match: i += 1 continue title = clean_title(title_match.group(1)) if title in seen_titles: i += 1 continue seen_titles.add(title) status = parse_status(content) douban, gr = extract_scores(content) author = extract_author(content) # 收集推荐理由 reasons = [] j = i + 1 while j < len(lines): nl = lines[j] if nl.startswith(" ") and not nl.strip().startswith("- ["): reason = nl.strip().lstrip("- ") if reason.startswith("推荐理由:") or reason.startswith("推荐理由:"): reason = reason.split(":", 1)[-1].split(":", 1)[-1].strip() if reason and len(reason) > 2: reasons.append(reason) j += 1 else: break read_reason = " ".join(reasons) if reasons else "" # group 推断 group = current_group if not group: for key, g in [("中国史", "中国史"), ("历史", "历史学"), ("科技", "科技&AI"), ("AI", "科技&AI"), ("文学", "文学"), ("哲学", "政治哲学"), ("经济", "经济&金融"), ("科学", "科学&进化"), ("心理", "心理&思维"), ("传记", "文学"), ("物理", "科学&进化"), ("宇宙", "科学&进化"), ("金融", "经济&金融"), ("投资", "经济&金融")]: if key in content: group = g break books.append({ "title": title, "author": author, "status": status, "douban_score": douban, "goodreads_score": gr, "group": group, "priority": current_priority, "read_reason": read_reason, "format": "paper", "tags": "", }) i = j continue return books def main(): booklist_path = os.path.join(os.path.dirname(__file__), "booklist.md") if not os.path.exists(booklist_path): print("❌ 未找到 booklist.md,请将书单保存到项目目录") sys.exit(1) with open(booklist_path, "r", encoding="utf-8") as f: text = f.read() books = parse_booklist(text) db.init_db() # 清空(一次性操作) conn = db._connect() existing = conn.execute("SELECT COUNT(*) FROM books").fetchone()[0] if existing > 0: print(f"⚠️ 数据库已有 {existing} 条记录,将先清空") conn.execute("DELETE FROM books") conn.execute("DELETE FROM sqlite_sequence WHERE name='books'") conn.commit() conn.close() # 批量插入 inserted = 0 for b in books: book_id = db.add_book( b["title"], author=b.get("author"), status=b["status"], douban_score=b.get("douban_score"), goodreads_score=b.get("goodreads_score"), group=b.get("group", ""), priority=b.get("priority", ""), read_reason=b.get("read_reason", ""), format=b["format"], tags=b.get("tags", ""), ) icon = {"read": "✅", "reading": "🔖", "to-read": "📖"}.get(b["status"], "📕") g = b.get("group", "") p = b.get("priority", "") print(f" {icon} [{book_id:>3}] 《{b['title']}》 [{g}] {p}") inserted += 1 print(f"\n✅ 导入完成:共 {inserted} 本书") stats = db.get_stats() print(f" 状态: {dict(stats['by_status'])}") if stats.get("by_group"): print(f" 分组: {dict(stats['by_group'])}") if __name__ == "__main__": main()