Files
bookshelf/import_booklist.py
T
kai 69c34d4faf feat: schema upgrade + import 159 books from reading list
- db.py: add group, isbn, priority, read_reason, read_order, cover_blob fields
- Import 159 books from booklist.md with group/priority/status metadata
- Include books.db in repo (personal use)
2026-03-25 14:55:23 +08:00

280 lines
8.5 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
#!/usr/bin/env python3
"""一次性脚本:从用户 Markdown 书单解析并导入 SQLite。"""
import re
import sys
import os
# 让 import db 能找到项目模块
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
os.chdir(os.path.dirname(os.path.abspath(__file__)))
import db
# ── 分组映射 ──────────────────────────────────────────────────────────
GROUP_MAP = {
"历史学": "历史学",
"伊朗": "历史学",
"中东": "历史学",
"政治哲学": "政治哲学",
"社会思想": "政治哲学",
"科学": "科学&进化",
"进化": "科学&进化",
"人生意义": "心理&思维",
"心理学": "心理&思维",
"经济学": "经济&金融",
"金融": "经济&金融",
"文学": "文学",
"传记": "文学",
"古典诗学": "古典诗学",
"文化": "古典诗学",
"投资": "经济&金融",
"交易": "经济&金融",
"思维": "心理&思维",
"工具": "心理&思维",
"科技": "科技&AI",
"AI": "科技&AI",
"赫拉利": "科技&AI",
"数学": "数学&科学史",
"科学史": "数学&科学史",
"伯克利": "推荐书单",
"中国史": "中国史",
"其他": "其他",
"已读书目": "已读归档",
"阅读优先级": "",
}
def infer_group(section_title):
for key, group in GROUP_MAP.items():
if key in section_title:
return group
return ""
def parse_status(content):
if "✅" in content and "已读" in content:
return "read"
if "🔖" in content or "在读" in content:
return "reading"
return "to-read"
def extract_scores(content):
douban = None
gr = None
m = re.search(r'豆瓣[^0-9]*?(\d\.\d)', content)
if m:
douban = float(m.group(1))
m = re.search(r'GR[^0-9]*?(\d\.\d+)', content)
if m:
gr = float(m.group(1))
return douban, gr
def extract_author(content):
# 尝试 · 作者 · 模式
m = re.search(r'·\s*([^·\n]{2,30}?)\s*·', content)
if m:
author = m.group(1).strip()
skip_words = ['豆瓣', 'GR', '评分', '文学', '历史', '科技',
'哲学', '经济', '科学', '传记', '政治',
'社会', '心理', '宇宙', '物理', '数学',
'⭐', '在读', '已读', '待读', 'FT', 'NYT',
'年度', '美国', '中国', '伊朗', '金融',
'制度', '方志远', '经济制裁', '复杂',
'海权', '越南', '中东', '法律', '清代']
if not any(kw in author for kw in skip_words):
return author
# 尝试 [] 内国籍后的作者: [美] 作者名
m = re.search(r'\[[\u4e00-\u9fff]+\]\s*([\u4e00-\u9fff·A-Za-z ]{2,20})', content)
if m:
return m.group(1).strip()
return None
def clean_title(raw):
return raw.replace("**", "").strip()
def parse_booklist(text):
books = []
seen_titles = set()
current_group = ""
current_priority = ""
lines = text.splitlines()
i = 0
while i < len(lines):
line = lines[i].rstrip()
# ## 大标题
if line.startswith("## "):
section = line.lstrip("# ").strip()
g = infer_group(section)
if g:
current_group = g
# 优先级检测
if "⭐⭐⭐" in section or "强烈推荐" in section:
current_priority = "⭐⭐⭐"
elif "⭐⭐" in section or "值得读" in section:
current_priority = "⭐⭐"
elif "⭐" in section or "可选" in section:
current_priority = "⭐"
else:
current_priority = ""
i += 1
continue
# ⭐ 优先级行
if re.match(r'^⭐⭐⭐\s', line):
current_priority = "⭐⭐⭐"
i += 1
continue
if re.match(r'^⭐⭐\s', line) and not line.startswith("⭐⭐⭐"):
current_priority = "⭐⭐"
i += 1
continue
if re.match(r'^⭐\s', line) and not line.startswith("⭐⭐"):
current_priority = "⭐"
i += 1
continue
# ### 子分组
if line.startswith("### "):
sub = line.lstrip("# ").strip()
sub_g = infer_group(sub)
if sub_g:
current_group = sub_g
i += 1
continue
# 书目行
m = re.match(r'^-\s*\[([ x/])\]\s*(.*)', line)
if not m:
i += 1
continue
content = m.group(2).strip()
# 提取书名
title_match = re.search(r'\*\*(.+?)\*\*', content)
if not title_match:
i += 1
continue
title = clean_title(title_match.group(1))
if title in seen_titles:
i += 1
continue
seen_titles.add(title)
status = parse_status(content)
douban, gr = extract_scores(content)
author = extract_author(content)
# 收集推荐理由
reasons = []
j = i + 1
while j < len(lines):
nl = lines[j]
if nl.startswith(" ") and not nl.strip().startswith("- ["):
reason = nl.strip().lstrip("- ")
if reason.startswith("推荐理由:") or reason.startswith("推荐理由:"):
reason = reason.split("", 1)[-1].split(":", 1)[-1].strip()
if reason and len(reason) > 2:
reasons.append(reason)
j += 1
else:
break
read_reason = " ".join(reasons) if reasons else ""
# group 推断
group = current_group
if not group:
for key, g in [("中国史", "中国史"), ("历史", "历史学"),
("科技", "科技&AI"), ("AI", "科技&AI"),
("文学", "文学"), ("哲学", "政治哲学"),
("经济", "经济&金融"), ("科学", "科学&进化"),
("心理", "心理&思维"), ("传记", "文学"),
("物理", "科学&进化"), ("宇宙", "科学&进化"),
("金融", "经济&金融"), ("投资", "经济&金融")]:
if key in content:
group = g
break
books.append({
"title": title,
"author": author,
"status": status,
"douban_score": douban,
"goodreads_score": gr,
"group": group,
"priority": current_priority,
"read_reason": read_reason,
"format": "paper",
"tags": "",
})
i = j
continue
return books
def main():
booklist_path = os.path.join(os.path.dirname(__file__), "booklist.md")
if not os.path.exists(booklist_path):
print("❌ 未找到 booklist.md,请将书单保存到项目目录")
sys.exit(1)
with open(booklist_path, "r", encoding="utf-8") as f:
text = f.read()
books = parse_booklist(text)
db.init_db()
# 清空(一次性操作)
conn = db._connect()
existing = conn.execute("SELECT COUNT(*) FROM books").fetchone()[0]
if existing > 0:
print(f"⚠️ 数据库已有 {existing} 条记录,将先清空")
conn.execute("DELETE FROM books")
conn.execute("DELETE FROM sqlite_sequence WHERE name='books'")
conn.commit()
conn.close()
# 批量插入
inserted = 0
for b in books:
book_id = db.add_book(
b["title"],
author=b.get("author"),
status=b["status"],
douban_score=b.get("douban_score"),
goodreads_score=b.get("goodreads_score"),
group=b.get("group", ""),
priority=b.get("priority", ""),
read_reason=b.get("read_reason", ""),
format=b["format"],
tags=b.get("tags", ""),
)
icon = {"read": "✅", "reading": "🔖", "to-read": "📖"}.get(b["status"], "📕")
g = b.get("group", "")
p = b.get("priority", "")
print(f" {icon} [{book_id:>3}] 《{b['title']}》 [{g}] {p}")
inserted += 1
print(f"\n✅ 导入完成:共 {inserted} 本书")
stats = db.get_stats()
print(f" 状态: {dict(stats['by_status'])}")
if stats.get("by_group"):
print(f" 分组: {dict(stats['by_group'])}")
if __name__ == "__main__":
main()