feat: add book scraping (Douban via Playwright + Goodreads + Google Books)

- scrape.py: Douban scraping with Playwright (multi-strategy browser launch)
- scrape.py: Goodreads scraping with requests + BeautifulSoup
- scrape.py: Google Books API + Open Library as fallback
- bookshelf.py: add 'fetch' subcommand (ISBN / Douban URL / GR URL)
- Fix Douban #info parsing using span.pl tags
- Add requirements.txt (requests, beautifulsoup4, playwright)
This commit is contained in:
kai
2026-03-25 14:20:12 +08:00
parent e87be58ba3
commit 8c8095d9ac
4 changed files with 635 additions and 0 deletions
+3
View File
@@ -1 +1,4 @@
__pycache__/ __pycache__/
.venv/
books.db
output/
+76
View File
@@ -5,6 +5,7 @@ import argparse
import sys import sys
import db import db
import web import web
import scrape
STATUS_CHOICES = ["to-read", "reading", "read"] STATUS_CHOICES = ["to-read", "reading", "read"]
@@ -78,6 +79,72 @@ def cmd_delete(args):
print(f"🗑️ 已删除:《{book['title']}》(ID: {args.id})") print(f"🗑️ 已删除:《{book['title']}》(ID: {args.id})")
def cmd_fetch(args):
"""刮削图书元数据,可选择直接写入数据库。"""
identifier = args.identifier
# 选择刮削策略
if "douban.com" in identifier:
print(f"🔍 正在从豆瓣刮削:{identifier}")
data = scrape.fetch_douban(identifier)
source = "豆瓣"
elif "goodreads.com" in identifier:
print(f"🔍 正在从 Goodreads 刮削:{identifier}")
data = scrape.fetch_goodreads(identifier)
source = "Goodreads"
elif scrape._is_isbn(identifier):
print(f"🔍 正在刮削 ISBN{identifier}")
data, source = scrape.fetch_by_isbn(identifier)
else:
print("❌ 请提供有效的 ISBN(10/13位)、豆瓣链接或 Goodreads 链接。")
sys.exit(1)
if not data:
print("❌ 刮削失败,未能获取到图书信息。")
print(" 提示:豆瓣偶尔会拒绝请求(418/403),请稍后重试。")
sys.exit(1)
# 打印预览
print(scrape.preview(data, source))
print()
if args.update_id:
# 更新已有记录
db.init_db()
book = db.get_book(args.update_id)
if not book:
print(f"❌ 未找到 ID 为 {args.update_id} 的书目。")
sys.exit(1)
# 只覆盖刮削到的字段,保留用户已设置的字段
updates = {k: v for k, v in data.items() if v}
if args.format:
updates["format"] = args.format
if args.status:
updates["status"] = args.status
db.update_book(args.update_id, **updates)
print(f"✅ 已更新:《{book['title']}》→《{data.get('title', book['title'])}》(ID: {args.update_id})")
elif args.add:
# 写入新记录
db.init_db()
book_id = db.add_book(
data.get("title", ""),
author=data.get("author"),
translator=data.get("translator"),
publisher=data.get("publisher"),
pub_date=data.get("pub_date"),
cover_url=data.get("cover_url"),
format=args.format or "paper",
status=args.status or "to-read",
douban_score=data.get("douban_score"),
goodreads_score=data.get("goodreads_score"),
tags=data.get("tags", ""),
)
print(f"✅ 已添加:《{data.get('title')}》(ID: {book_id})")
else:
print("💡 加 --add 选项可直接写入数据库,或用 --update-id <ID> 更新已有记录。")
def cmd_build(args): def cmd_build(args):
db.init_db() db.init_db()
output_path = web.build() output_path = web.build()
@@ -142,6 +209,15 @@ def main():
p_del.add_argument("id", type=int, help="书目 ID") p_del.add_argument("id", type=int, help="书目 ID")
p_del.set_defaults(func=cmd_delete) p_del.set_defaults(func=cmd_delete)
# --- fetch ---
p_fetch = sub.add_parser("fetch", help="从豆瓣/Google Books 刮削书目信息")
p_fetch.add_argument("identifier", help="ISBN10/13位)或豆瓣图书链接")
p_fetch.add_argument("--add", action="store_true", help="刮削后直接写入数据库")
p_fetch.add_argument("--update-id", type=int, metavar="ID", help="刮削后更新指定 ID 的书目")
p_fetch.add_argument("--format", choices=FORMAT_CHOICES, help="书籍格式(入库时使用)")
p_fetch.add_argument("--status", choices=STATUS_CHOICES, help="阅读状态(入库时使用)")
p_fetch.set_defaults(func=cmd_fetch)
# --- build --- # --- build ---
p_build = sub.add_parser("build", help="生成展示网页") p_build = sub.add_parser("build", help="生成展示网页")
p_build.set_defaults(func=cmd_build) p_build.set_defaults(func=cmd_build)
+3
View File
@@ -0,0 +1,3 @@
requests>=2.31.0
beautifulsoup4>=4.12.0
playwright>=1.40.0
+553
View File
@@ -0,0 +1,553 @@
"""图书元数据刮削模块。
数据源策略:
- 中文书:豆瓣网页爬取 + Goodreads 补评分
- 英文书:Goodreads 为主力
- 兜底:Google Books API / Open Library ISBN API
"""
import json
import re
import time
import random
try:
import requests as _requests
_HAS_REQUESTS = True
except ImportError:
import urllib.request
_HAS_REQUESTS = False
# ── 常量 ────────────────────────────────────────────────────────────
_UA = (
"Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) "
"AppleWebKit/537.36 (KHTML, like Gecko) "
"Chrome/123.0.0.0 Safari/537.36"
)
DOUBAN_ISBN_URL = "https://book.douban.com/isbn/{isbn}"
DOUBAN_SUBJECT_URL = "https://book.douban.com/subject/{sid}/"
GOODREADS_BOOK_URL = "https://www.goodreads.com/book/show/{book_id}"
GOODREADS_SEARCH_URL = "https://www.goodreads.com/search?q={query}"
GOOGLE_BOOKS_URL = "https://www.googleapis.com/books/v1/volumes?q=isbn:{isbn}"
OPEN_LIBRARY_URL = (
"https://openlibrary.org/api/books?bibkeys=ISBN:{isbn}&format=json&jscmd=data"
)
OPEN_LIBRARY_COVER = "https://covers.openlibrary.org/b/isbn/{isbn}-L.jpg"
# ── 工具函数 ─────────────────────────────────────────────────────────
def _get(url, timeout=15):
"""带 UA 的 HTTP GET,返回响应文本或 None。优先使用 requests 库。"""
headers = {"User-Agent": _UA}
if _HAS_REQUESTS:
try:
resp = _requests.get(url, headers=headers, timeout=timeout,
allow_redirects=True)
if resp.status_code == 200:
return resp.text
return None
except Exception:
return None
else:
import urllib.request
req = urllib.request.Request(url, headers=headers)
try:
with urllib.request.urlopen(req, timeout=timeout) as resp:
return resp.read().decode("utf-8", errors="replace")
except Exception:
return None
def _get_with_playwright(url, timeout=30000):
"""用 Playwright 无头浏览器获取页面 HTML(用于豆瓣等反爬严格的站点)。"""
try:
from playwright.sync_api import sync_playwright
except ImportError:
print("⚠️ 需要安装 Playwrightpip install playwright && playwright install")
return None
html = None
# 多种启动策略:系统 Chrome → Playwright Chromium → 不同 headless 模式
launch_options = [
{"channel": "chrome", "headless": True}, # 系统 Chrome
{"channel": "chromium", "headless": True}, # 系统 Chromium
{"headless": True, # Playwright 自带
"args": ["--disable-blink-features=AutomationControlled"]},
{"headless": False, # 非 headless(最后手段)
"args": ["--headless=new"]}, # Chrome 新 headless 模式
]
try:
with sync_playwright() as p:
browser = None
for opts in launch_options:
try:
browser = p.chromium.launch(**opts)
break
except Exception:
continue
if not browser:
print("⚠️ 无法启动浏览器。请确保已安装 Chrome 或运行:playwright install")
return None
context = browser.new_context(
user_agent=_UA,
viewport={"width": 1280, "height": 800},
locale="zh-CN",
)
page = context.new_page()
page.goto(url, wait_until="domcontentloaded", timeout=timeout)
# 等待核心内容出现
try:
page.wait_for_selector("#info", timeout=8000)
except Exception:
pass # 即使超时也继续,页面可能已部分加载
html = page.content()
browser.close()
except Exception as e:
print(f"⚠️ Playwright 访问失败:{e}")
return None
return html
def _sleep():
"""礼貌延时,避免被封。"""
time.sleep(random.uniform(1.0, 2.5))
def _is_isbn(s):
digits = re.sub(r"[-\s]", "", s)
return re.fullmatch(r"\d{10}|\d{13}", digits) is not None
def _clean_isbn(s):
return re.sub(r"[-\s]", "", s)
def _is_chinese_isbn(isbn):
"""以 978-7 / 7 开头的 ISBN 通常是中文出版物。"""
d = _clean_isbn(isbn)
return d.startswith("9787") or d.startswith("7")
# ── 豆瓣刮削 ─────────────────────────────────────────────────────────
def _parse_douban_html(html):
"""解析豆瓣图书详情页 HTML,返回 dict。需要 beautifulsoup4。"""
try:
from bs4 import BeautifulSoup
except ImportError:
raise RuntimeError("请先安装 beautifulsoup4pip install beautifulsoup4")
soup = BeautifulSoup(html, "html.parser")
result = {}
# 书名
title_tag = soup.find("h1")
if title_tag:
span = title_tag.find("span")
result["title"] = (span or title_tag).get_text(strip=True)
# 封面
cover_tag = soup.find("a", attrs={"class": "nbg"})
if cover_tag:
img = cover_tag.find("img")
if img:
result["cover_url"] = img.get("src", "")
# 豆瓣评分
rating_tag = soup.find("strong", attrs={"class": "ll rating_num"})
if rating_tag:
txt = rating_tag.get_text(strip=True)
try:
result["douban_score"] = float(txt)
except ValueError:
pass
# 作者、出版社等在 #info 块
info = soup.find("div", id="info")
if info:
# 方法1:通过 <span class="pl"> 标签精确提取
for span in info.find_all("span", class_="pl"):
label = span.get_text(strip=True).rstrip(": ")
# 获取同级后续文本和链接
values = []
for sib in span.next_siblings:
if hasattr(sib, 'name'):
if sib.name == 'br':
break
if sib.name == 'span' and 'pl' in (sib.get('class') or []):
break
if sib.name == 'a':
values.append(sib.get_text(strip=True))
elif sib.name == 'span':
values.append(sib.get_text(strip=True))
else:
t = str(sib).strip().strip(":/·, ")
if t:
values.append(t)
value = " ".join(v for v in values if v).strip(" /:")
if not value:
continue
if "作者" in label:
# 去掉 [国籍] 标注
author_clean = re.sub(r"\[.*?\]", "", value).strip(" /")
# 清理多余空格
author_clean = re.sub(r"\s+", " ", author_clean).strip()
if author_clean:
result["author"] = author_clean
elif "译者" in label:
result["translator"] = re.sub(r"\s+", " ", value).strip(" /")
elif "出版社" in label:
result["publisher"] = value
elif "出版年" in label:
result["pub_date"] = value
elif "ISBN" in label.upper():
result["isbn"] = value
# 方法2:兜底 — 如果 span.pl 没有拿到,用纯文本
if not result.get("author"):
text = info.get_text(separator="\n")
for line in text.splitlines():
line = line.strip()
if "作者" in line:
parts = re.split(r"[:]", line, 1)
if len(parts) == 2 and parts[1].strip():
author = re.sub(r"\[.*?\]", "", parts[1]).strip(" /")
if author:
result["author"] = author
break
# 标签
tag_links = soup.select("a.tag")
if tag_links:
tags = [a.get_text(strip=True) for a in tag_links[:6]]
result["tags"] = ",".join(tags)
return result if result.get("title") else None
def fetch_douban(identifier: str):
"""
从豆瓣刮削图书信息。
identifier 可以是:
- ISBN10位或13位)
- 豆瓣图书链接(含 subject/id
- 豆瓣 subject id(纯数字)
"""
identifier = identifier.strip()
if _is_isbn(identifier):
url = DOUBAN_ISBN_URL.format(isbn=_clean_isbn(identifier))
elif "douban.com/subject/" in identifier:
url = identifier.split("?")[0].rstrip("/") + "/"
elif re.fullmatch(r"\d+", identifier):
url = DOUBAN_SUBJECT_URL.format(sid=identifier)
else:
return None
_sleep()
html = _get_with_playwright(url)
if not html or "豆瓣" not in html:
# Playwright 失败时尝试普通 requests(偶尔可能成功)
html = _get(url)
if not html or "豆瓣" not in html:
return None
data = _parse_douban_html(html)
return data
# ── Goodreads 刮削 ────────────────────────────────────────────────────
def _parse_goodreads_html(html):
"""解析 Goodreads 图书详情页 HTML,返回 dict。"""
try:
from bs4 import BeautifulSoup
except ImportError:
raise RuntimeError("请先安装 beautifulsoup4pip install beautifulsoup4")
soup = BeautifulSoup(html, "html.parser")
result = {}
# 书名 — <h1 class="Text Text__title1" ...>
title_tag = soup.find("h1", attrs={"data-testid": "bookTitle"})
if not title_tag:
title_tag = soup.find("h1")
if title_tag:
result["title"] = title_tag.get_text(strip=True)
# 作者 — <span class="ContributorLink__name" ...>
author_tags = soup.select("span.ContributorLink__name")
if author_tags:
authors = []
for a in author_tags:
name = a.get_text(strip=True)
# 排除角色标注如 "(Primary Contributor)"
if name and "Contributor" not in name:
authors.append(name)
if authors:
result["author"] = " / ".join(dict.fromkeys(authors)) # 去重保序
# 评分 — 在页面文本中匹配 "X.XX" 紧跟 ratings 数字
# 格式通常是 "3.8015,305 ratings" 或类似
rating_match = re.search(
r'"ratingValue"\s*:\s*([\d.]+)', html
)
if rating_match:
try:
result["goodreads_score"] = float(rating_match.group(1))
except ValueError:
pass
if not result.get("goodreads_score"):
# 备选:从可见文本中提取
rating_div = soup.find("div", attrs={"class": re.compile(r"RatingStatistics__rating")})
if rating_div:
txt = rating_div.get_text(strip=True)
try:
result["goodreads_score"] = float(txt)
except ValueError:
pass
if not result.get("goodreads_score"):
# 再备选:页面全文正则("4.18" 后跟 "ratings")
fallback = re.search(r'(\d\.\d{1,2})\s*[\d,]+\s*rating', html)
if fallback:
try:
result["goodreads_score"] = float(fallback.group(1))
except ValueError:
pass
# 分类标签 — genre links
genre_links = soup.select("span.BookPageMetadataSection__genreButton a")
if not genre_links:
genre_links = soup.select("a[href*='/genres/']")
if genre_links:
tags = []
for a in genre_links:
tag = a.get_text(strip=True)
if tag and tag not in tags and len(tags) < 6:
tags.append(tag)
if tags:
result["tags"] = ",".join(tags)
# 出版信息 — 在页面文本中查找
pub_match = re.search(
r'(?:First published|Published)\s+(\w+\s+\d{1,2},\s*\d{4}|\w+\s+\d{4}|\d{4})',
html
)
if pub_match:
result["pub_date"] = pub_match.group(1)
# 封面
cover_tag = soup.find("img", attrs={"class": re.compile(r"ResponsiveImage")})
if cover_tag:
src = cover_tag.get("src", "")
if src and "nophoto" not in src:
result["cover_url"] = src
return result if result.get("title") else None
def fetch_goodreads(identifier: str):
"""
从 Goodreads 刮削图书信息。
identifier 可以是:
- Goodreads 链接(含 /book/show/
- ISBN(会用搜索页查找)
"""
identifier = identifier.strip()
if "goodreads.com/book/show/" in identifier:
url = identifier.split("?")[0].split("&")[0]
elif _is_isbn(identifier):
# 先通过搜索页找到书籍链接
search_url = GOODREADS_SEARCH_URL.format(query=_clean_isbn(identifier))
_sleep()
search_html = _get(search_url)
if not search_html:
return None
# 从搜索结果中提取第一个 /book/show/ 链接
match = re.search(r'/book/show/(\d+[^"\s\']*)', search_html)
if not match:
return None
url = f"https://www.goodreads.com/book/show/{match.group(1)}"
else:
return None
_sleep()
html = _get(url)
if not html or "goodreads" not in html.lower():
return None
return _parse_goodreads_html(html)
def fetch_goodreads_score(isbn: str):
"""只获取 Goodreads 评分(用于补全其他来源的数据)。"""
data = fetch_goodreads(isbn)
if data and data.get("goodreads_score"):
return data["goodreads_score"]
return None
# ── Google Books API ──────────────────────────────────────────────────
def fetch_google_books(isbn: str):
"""通过 Google Books API 获取图书信息(无需 key)。"""
isbn = _clean_isbn(isbn)
url = GOOGLE_BOOKS_URL.format(isbn=isbn)
html = _get(url)
if not html:
return None
try:
data = json.loads(html)
except json.JSONDecodeError:
return None
if data.get("totalItems", 0) == 0:
return None
item = data["items"][0]
vol = item.get("volumeInfo", {})
result = {"title": vol.get("title", "")}
authors = vol.get("authors", [])
if authors:
result["author"] = " / ".join(authors)
if vol.get("publisher"):
result["publisher"] = vol["publisher"]
if vol.get("publishedDate"):
result["pub_date"] = vol["publishedDate"][:7] # YYYY-MM
# 封面
images = vol.get("imageLinks", {})
cover = images.get("thumbnail") or images.get("smallThumbnail")
if cover:
# 换成更大图
result["cover_url"] = cover.replace("zoom=1", "zoom=3")
# 分类标签
cats = vol.get("categories", [])
if cats:
result["tags"] = ",".join(cats[:4])
# 评分(Google Books 有 averageRating 字段)
if vol.get("averageRating"):
result["goodreads_score"] = float(vol["averageRating"])
return result if result.get("title") else None
# ── Open Library 兜底 ─────────────────────────────────────────────────
def fetch_open_library(isbn: str):
"""Open Library ISBN API 兜底。"""
isbn = _clean_isbn(isbn)
url = OPEN_LIBRARY_URL.format(isbn=isbn)
html = _get(url)
if not html:
return None
try:
data = json.loads(html)
except json.JSONDecodeError:
return None
key = f"ISBN:{isbn}"
if key not in data:
return None
book = data[key]
result = {"title": book.get("title", "")}
authors = book.get("authors", [])
if authors:
result["author"] = " / ".join(a.get("name", "") for a in authors)
pubs = book.get("publishers", [])
if pubs:
result["publisher"] = pubs[0].get("name", "")
if book.get("publish_date"):
result["pub_date"] = book["publish_date"]
# 封面:直接用 covers API(即使信息里没有封面也能试)
cover_url = f"https://covers.openlibrary.org/b/isbn/{isbn}-L.jpg"
result["cover_url"] = cover_url
subjects = book.get("subjects", [])
if subjects:
tags = [s.get("name", "") for s in subjects[:4]]
result["tags"] = ",".join(t for t in tags if t)
return result if result.get("title") else None
# ── 主入口 ────────────────────────────────────────────────────────────
def fetch_by_isbn(isbn: str):
"""
根据 ISBN 自动选择数据源。
中文 ISBN → 豆瓣 + Goodreads 补 GR 评分。
英文 ISBN → Goodreads 为主力 → Google Books 兜底。
"""
if _is_chinese_isbn(isbn):
result = fetch_douban(isbn)
if result:
# 尝试补全 Goodreads 评分
if not result.get("goodreads_score"):
gr_score = fetch_goodreads_score(isbn)
if gr_score:
result["goodreads_score"] = gr_score
return result, "豆瓣"
result = fetch_open_library(isbn)
if result:
return result, "Open Library"
else:
# 英文书:Goodreads 为主力
result = fetch_goodreads(isbn)
if result:
return result, "Goodreads"
# 兜底
result = fetch_google_books(isbn)
if result:
return result, "Google Books"
result = fetch_open_library(isbn)
if result:
return result, "Open Library"
return None, None
def preview(data: dict, source: str):
"""格式化打印刮削结果预览。"""
lines = [f"\n📖 《{data.get('title', '未知')}》 [来源: {source}]"]
if data.get("author"):
lines.append(f" 作者:{data['author']}")
if data.get("translator"):
lines.append(f" 译者:{data['translator']}")
if data.get("publisher") or data.get("pub_date"):
pub = " · ".join(filter(None, [data.get("publisher"), data.get("pub_date")]))
lines.append(f" 出版:{pub}")
if data.get("douban_score"):
lines.append(f" 豆瓣:{data['douban_score']}")
if data.get("goodreads_score"):
lines.append(f" GR{data['goodreads_score']}")
if data.get("tags"):
lines.append(f" 标签:{data['tags']}")
if data.get("cover_url"):
lines.append(f" 封面:{data['cover_url']}")
return "\n".join(lines)