feat: add book scraping (Douban via Playwright + Goodreads + Google Books)
- scrape.py: Douban scraping with Playwright (multi-strategy browser launch) - scrape.py: Goodreads scraping with requests + BeautifulSoup - scrape.py: Google Books API + Open Library as fallback - bookshelf.py: add 'fetch' subcommand (ISBN / Douban URL / GR URL) - Fix Douban #info parsing using span.pl tags - Add requirements.txt (requests, beautifulsoup4, playwright)
This commit is contained in:
@@ -1 +1,4 @@
|
||||
__pycache__/
|
||||
.venv/
|
||||
books.db
|
||||
output/
|
||||
|
||||
@@ -5,6 +5,7 @@ import argparse
|
||||
import sys
|
||||
import db
|
||||
import web
|
||||
import scrape
|
||||
|
||||
|
||||
STATUS_CHOICES = ["to-read", "reading", "read"]
|
||||
@@ -78,6 +79,72 @@ def cmd_delete(args):
|
||||
print(f"🗑️ 已删除:《{book['title']}》(ID: {args.id})")
|
||||
|
||||
|
||||
def cmd_fetch(args):
|
||||
"""刮削图书元数据,可选择直接写入数据库。"""
|
||||
identifier = args.identifier
|
||||
|
||||
# 选择刮削策略
|
||||
if "douban.com" in identifier:
|
||||
print(f"🔍 正在从豆瓣刮削:{identifier}")
|
||||
data = scrape.fetch_douban(identifier)
|
||||
source = "豆瓣"
|
||||
elif "goodreads.com" in identifier:
|
||||
print(f"🔍 正在从 Goodreads 刮削:{identifier}")
|
||||
data = scrape.fetch_goodreads(identifier)
|
||||
source = "Goodreads"
|
||||
elif scrape._is_isbn(identifier):
|
||||
print(f"🔍 正在刮削 ISBN:{identifier}")
|
||||
data, source = scrape.fetch_by_isbn(identifier)
|
||||
else:
|
||||
print("❌ 请提供有效的 ISBN(10/13位)、豆瓣链接或 Goodreads 链接。")
|
||||
sys.exit(1)
|
||||
|
||||
if not data:
|
||||
print("❌ 刮削失败,未能获取到图书信息。")
|
||||
print(" 提示:豆瓣偶尔会拒绝请求(418/403),请稍后重试。")
|
||||
sys.exit(1)
|
||||
|
||||
# 打印预览
|
||||
print(scrape.preview(data, source))
|
||||
print()
|
||||
|
||||
if args.update_id:
|
||||
# 更新已有记录
|
||||
db.init_db()
|
||||
book = db.get_book(args.update_id)
|
||||
if not book:
|
||||
print(f"❌ 未找到 ID 为 {args.update_id} 的书目。")
|
||||
sys.exit(1)
|
||||
# 只覆盖刮削到的字段,保留用户已设置的字段
|
||||
updates = {k: v for k, v in data.items() if v}
|
||||
if args.format:
|
||||
updates["format"] = args.format
|
||||
if args.status:
|
||||
updates["status"] = args.status
|
||||
db.update_book(args.update_id, **updates)
|
||||
print(f"✅ 已更新:《{book['title']}》→《{data.get('title', book['title'])}》(ID: {args.update_id})")
|
||||
|
||||
elif args.add:
|
||||
# 写入新记录
|
||||
db.init_db()
|
||||
book_id = db.add_book(
|
||||
data.get("title", ""),
|
||||
author=data.get("author"),
|
||||
translator=data.get("translator"),
|
||||
publisher=data.get("publisher"),
|
||||
pub_date=data.get("pub_date"),
|
||||
cover_url=data.get("cover_url"),
|
||||
format=args.format or "paper",
|
||||
status=args.status or "to-read",
|
||||
douban_score=data.get("douban_score"),
|
||||
goodreads_score=data.get("goodreads_score"),
|
||||
tags=data.get("tags", ""),
|
||||
)
|
||||
print(f"✅ 已添加:《{data.get('title')}》(ID: {book_id})")
|
||||
else:
|
||||
print("💡 加 --add 选项可直接写入数据库,或用 --update-id <ID> 更新已有记录。")
|
||||
|
||||
|
||||
def cmd_build(args):
|
||||
db.init_db()
|
||||
output_path = web.build()
|
||||
@@ -142,6 +209,15 @@ def main():
|
||||
p_del.add_argument("id", type=int, help="书目 ID")
|
||||
p_del.set_defaults(func=cmd_delete)
|
||||
|
||||
# --- fetch ---
|
||||
p_fetch = sub.add_parser("fetch", help="从豆瓣/Google Books 刮削书目信息")
|
||||
p_fetch.add_argument("identifier", help="ISBN(10/13位)或豆瓣图书链接")
|
||||
p_fetch.add_argument("--add", action="store_true", help="刮削后直接写入数据库")
|
||||
p_fetch.add_argument("--update-id", type=int, metavar="ID", help="刮削后更新指定 ID 的书目")
|
||||
p_fetch.add_argument("--format", choices=FORMAT_CHOICES, help="书籍格式(入库时使用)")
|
||||
p_fetch.add_argument("--status", choices=STATUS_CHOICES, help="阅读状态(入库时使用)")
|
||||
p_fetch.set_defaults(func=cmd_fetch)
|
||||
|
||||
# --- build ---
|
||||
p_build = sub.add_parser("build", help="生成展示网页")
|
||||
p_build.set_defaults(func=cmd_build)
|
||||
|
||||
@@ -0,0 +1,3 @@
|
||||
requests>=2.31.0
|
||||
beautifulsoup4>=4.12.0
|
||||
playwright>=1.40.0
|
||||
@@ -0,0 +1,553 @@
|
||||
"""图书元数据刮削模块。
|
||||
|
||||
数据源策略:
|
||||
- 中文书:豆瓣网页爬取 + Goodreads 补评分
|
||||
- 英文书:Goodreads 为主力
|
||||
- 兜底:Google Books API / Open Library ISBN API
|
||||
"""
|
||||
|
||||
import json
|
||||
import re
|
||||
import time
|
||||
import random
|
||||
|
||||
try:
|
||||
import requests as _requests
|
||||
_HAS_REQUESTS = True
|
||||
except ImportError:
|
||||
import urllib.request
|
||||
_HAS_REQUESTS = False
|
||||
|
||||
# ── 常量 ────────────────────────────────────────────────────────────
|
||||
|
||||
_UA = (
|
||||
"Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) "
|
||||
"AppleWebKit/537.36 (KHTML, like Gecko) "
|
||||
"Chrome/123.0.0.0 Safari/537.36"
|
||||
)
|
||||
|
||||
DOUBAN_ISBN_URL = "https://book.douban.com/isbn/{isbn}"
|
||||
DOUBAN_SUBJECT_URL = "https://book.douban.com/subject/{sid}/"
|
||||
GOODREADS_BOOK_URL = "https://www.goodreads.com/book/show/{book_id}"
|
||||
GOODREADS_SEARCH_URL = "https://www.goodreads.com/search?q={query}"
|
||||
GOOGLE_BOOKS_URL = "https://www.googleapis.com/books/v1/volumes?q=isbn:{isbn}"
|
||||
OPEN_LIBRARY_URL = (
|
||||
"https://openlibrary.org/api/books?bibkeys=ISBN:{isbn}&format=json&jscmd=data"
|
||||
)
|
||||
OPEN_LIBRARY_COVER = "https://covers.openlibrary.org/b/isbn/{isbn}-L.jpg"
|
||||
|
||||
# ── 工具函数 ─────────────────────────────────────────────────────────
|
||||
|
||||
def _get(url, timeout=15):
|
||||
"""带 UA 的 HTTP GET,返回响应文本或 None。优先使用 requests 库。"""
|
||||
headers = {"User-Agent": _UA}
|
||||
if _HAS_REQUESTS:
|
||||
try:
|
||||
resp = _requests.get(url, headers=headers, timeout=timeout,
|
||||
allow_redirects=True)
|
||||
if resp.status_code == 200:
|
||||
return resp.text
|
||||
return None
|
||||
except Exception:
|
||||
return None
|
||||
else:
|
||||
import urllib.request
|
||||
req = urllib.request.Request(url, headers=headers)
|
||||
try:
|
||||
with urllib.request.urlopen(req, timeout=timeout) as resp:
|
||||
return resp.read().decode("utf-8", errors="replace")
|
||||
except Exception:
|
||||
return None
|
||||
|
||||
|
||||
def _get_with_playwright(url, timeout=30000):
|
||||
"""用 Playwright 无头浏览器获取页面 HTML(用于豆瓣等反爬严格的站点)。"""
|
||||
try:
|
||||
from playwright.sync_api import sync_playwright
|
||||
except ImportError:
|
||||
print("⚠️ 需要安装 Playwright:pip install playwright && playwright install")
|
||||
return None
|
||||
|
||||
html = None
|
||||
# 多种启动策略:系统 Chrome → Playwright Chromium → 不同 headless 模式
|
||||
launch_options = [
|
||||
{"channel": "chrome", "headless": True}, # 系统 Chrome
|
||||
{"channel": "chromium", "headless": True}, # 系统 Chromium
|
||||
{"headless": True, # Playwright 自带
|
||||
"args": ["--disable-blink-features=AutomationControlled"]},
|
||||
{"headless": False, # 非 headless(最后手段)
|
||||
"args": ["--headless=new"]}, # Chrome 新 headless 模式
|
||||
]
|
||||
|
||||
try:
|
||||
with sync_playwright() as p:
|
||||
browser = None
|
||||
for opts in launch_options:
|
||||
try:
|
||||
browser = p.chromium.launch(**opts)
|
||||
break
|
||||
except Exception:
|
||||
continue
|
||||
|
||||
if not browser:
|
||||
print("⚠️ 无法启动浏览器。请确保已安装 Chrome 或运行:playwright install")
|
||||
return None
|
||||
|
||||
context = browser.new_context(
|
||||
user_agent=_UA,
|
||||
viewport={"width": 1280, "height": 800},
|
||||
locale="zh-CN",
|
||||
)
|
||||
page = context.new_page()
|
||||
page.goto(url, wait_until="domcontentloaded", timeout=timeout)
|
||||
# 等待核心内容出现
|
||||
try:
|
||||
page.wait_for_selector("#info", timeout=8000)
|
||||
except Exception:
|
||||
pass # 即使超时也继续,页面可能已部分加载
|
||||
html = page.content()
|
||||
browser.close()
|
||||
except Exception as e:
|
||||
print(f"⚠️ Playwright 访问失败:{e}")
|
||||
return None
|
||||
|
||||
return html
|
||||
|
||||
|
||||
def _sleep():
|
||||
"""礼貌延时,避免被封。"""
|
||||
time.sleep(random.uniform(1.0, 2.5))
|
||||
|
||||
|
||||
def _is_isbn(s):
|
||||
digits = re.sub(r"[-\s]", "", s)
|
||||
return re.fullmatch(r"\d{10}|\d{13}", digits) is not None
|
||||
|
||||
|
||||
def _clean_isbn(s):
|
||||
return re.sub(r"[-\s]", "", s)
|
||||
|
||||
|
||||
def _is_chinese_isbn(isbn):
|
||||
"""以 978-7 / 7 开头的 ISBN 通常是中文出版物。"""
|
||||
d = _clean_isbn(isbn)
|
||||
return d.startswith("9787") or d.startswith("7")
|
||||
|
||||
|
||||
# ── 豆瓣刮削 ─────────────────────────────────────────────────────────
|
||||
|
||||
def _parse_douban_html(html):
|
||||
"""解析豆瓣图书详情页 HTML,返回 dict。需要 beautifulsoup4。"""
|
||||
try:
|
||||
from bs4 import BeautifulSoup
|
||||
except ImportError:
|
||||
raise RuntimeError("请先安装 beautifulsoup4:pip install beautifulsoup4")
|
||||
|
||||
soup = BeautifulSoup(html, "html.parser")
|
||||
|
||||
result = {}
|
||||
|
||||
# 书名
|
||||
title_tag = soup.find("h1")
|
||||
if title_tag:
|
||||
span = title_tag.find("span")
|
||||
result["title"] = (span or title_tag).get_text(strip=True)
|
||||
|
||||
# 封面
|
||||
cover_tag = soup.find("a", attrs={"class": "nbg"})
|
||||
if cover_tag:
|
||||
img = cover_tag.find("img")
|
||||
if img:
|
||||
result["cover_url"] = img.get("src", "")
|
||||
|
||||
# 豆瓣评分
|
||||
rating_tag = soup.find("strong", attrs={"class": "ll rating_num"})
|
||||
if rating_tag:
|
||||
txt = rating_tag.get_text(strip=True)
|
||||
try:
|
||||
result["douban_score"] = float(txt)
|
||||
except ValueError:
|
||||
pass
|
||||
|
||||
# 作者、出版社等在 #info 块
|
||||
info = soup.find("div", id="info")
|
||||
if info:
|
||||
# 方法1:通过 <span class="pl"> 标签精确提取
|
||||
for span in info.find_all("span", class_="pl"):
|
||||
label = span.get_text(strip=True).rstrip(":: ")
|
||||
# 获取同级后续文本和链接
|
||||
values = []
|
||||
for sib in span.next_siblings:
|
||||
if hasattr(sib, 'name'):
|
||||
if sib.name == 'br':
|
||||
break
|
||||
if sib.name == 'span' and 'pl' in (sib.get('class') or []):
|
||||
break
|
||||
if sib.name == 'a':
|
||||
values.append(sib.get_text(strip=True))
|
||||
elif sib.name == 'span':
|
||||
values.append(sib.get_text(strip=True))
|
||||
else:
|
||||
t = str(sib).strip().strip("::/·, ")
|
||||
if t:
|
||||
values.append(t)
|
||||
value = " ".join(v for v in values if v).strip(" /::")
|
||||
|
||||
if not value:
|
||||
continue
|
||||
|
||||
if "作者" in label:
|
||||
# 去掉 [国籍] 标注
|
||||
author_clean = re.sub(r"\[.*?\]", "", value).strip(" /")
|
||||
# 清理多余空格
|
||||
author_clean = re.sub(r"\s+", " ", author_clean).strip()
|
||||
if author_clean:
|
||||
result["author"] = author_clean
|
||||
elif "译者" in label:
|
||||
result["translator"] = re.sub(r"\s+", " ", value).strip(" /")
|
||||
elif "出版社" in label:
|
||||
result["publisher"] = value
|
||||
elif "出版年" in label:
|
||||
result["pub_date"] = value
|
||||
elif "ISBN" in label.upper():
|
||||
result["isbn"] = value
|
||||
|
||||
# 方法2:兜底 — 如果 span.pl 没有拿到,用纯文本
|
||||
if not result.get("author"):
|
||||
text = info.get_text(separator="\n")
|
||||
for line in text.splitlines():
|
||||
line = line.strip()
|
||||
if "作者" in line:
|
||||
parts = re.split(r"[::]", line, 1)
|
||||
if len(parts) == 2 and parts[1].strip():
|
||||
author = re.sub(r"\[.*?\]", "", parts[1]).strip(" /")
|
||||
if author:
|
||||
result["author"] = author
|
||||
break
|
||||
|
||||
# 标签
|
||||
tag_links = soup.select("a.tag")
|
||||
if tag_links:
|
||||
tags = [a.get_text(strip=True) for a in tag_links[:6]]
|
||||
result["tags"] = ",".join(tags)
|
||||
|
||||
return result if result.get("title") else None
|
||||
|
||||
|
||||
def fetch_douban(identifier: str):
|
||||
"""
|
||||
从豆瓣刮削图书信息。
|
||||
identifier 可以是:
|
||||
- ISBN(10位或13位)
|
||||
- 豆瓣图书链接(含 subject/id)
|
||||
- 豆瓣 subject id(纯数字)
|
||||
"""
|
||||
identifier = identifier.strip()
|
||||
|
||||
if _is_isbn(identifier):
|
||||
url = DOUBAN_ISBN_URL.format(isbn=_clean_isbn(identifier))
|
||||
elif "douban.com/subject/" in identifier:
|
||||
url = identifier.split("?")[0].rstrip("/") + "/"
|
||||
elif re.fullmatch(r"\d+", identifier):
|
||||
url = DOUBAN_SUBJECT_URL.format(sid=identifier)
|
||||
else:
|
||||
return None
|
||||
|
||||
_sleep()
|
||||
html = _get_with_playwright(url)
|
||||
if not html or "豆瓣" not in html:
|
||||
# Playwright 失败时尝试普通 requests(偶尔可能成功)
|
||||
html = _get(url)
|
||||
if not html or "豆瓣" not in html:
|
||||
return None
|
||||
|
||||
data = _parse_douban_html(html)
|
||||
return data
|
||||
|
||||
|
||||
# ── Goodreads 刮削 ────────────────────────────────────────────────────
|
||||
|
||||
def _parse_goodreads_html(html):
|
||||
"""解析 Goodreads 图书详情页 HTML,返回 dict。"""
|
||||
try:
|
||||
from bs4 import BeautifulSoup
|
||||
except ImportError:
|
||||
raise RuntimeError("请先安装 beautifulsoup4:pip install beautifulsoup4")
|
||||
|
||||
soup = BeautifulSoup(html, "html.parser")
|
||||
result = {}
|
||||
|
||||
# 书名 — <h1 class="Text Text__title1" ...>
|
||||
title_tag = soup.find("h1", attrs={"data-testid": "bookTitle"})
|
||||
if not title_tag:
|
||||
title_tag = soup.find("h1")
|
||||
if title_tag:
|
||||
result["title"] = title_tag.get_text(strip=True)
|
||||
|
||||
# 作者 — <span class="ContributorLink__name" ...>
|
||||
author_tags = soup.select("span.ContributorLink__name")
|
||||
if author_tags:
|
||||
authors = []
|
||||
for a in author_tags:
|
||||
name = a.get_text(strip=True)
|
||||
# 排除角色标注如 "(Primary Contributor)"
|
||||
if name and "Contributor" not in name:
|
||||
authors.append(name)
|
||||
if authors:
|
||||
result["author"] = " / ".join(dict.fromkeys(authors)) # 去重保序
|
||||
|
||||
# 评分 — 在页面文本中匹配 "X.XX" 紧跟 ratings 数字
|
||||
# 格式通常是 "3.8015,305 ratings" 或类似
|
||||
rating_match = re.search(
|
||||
r'"ratingValue"\s*:\s*([\d.]+)', html
|
||||
)
|
||||
if rating_match:
|
||||
try:
|
||||
result["goodreads_score"] = float(rating_match.group(1))
|
||||
except ValueError:
|
||||
pass
|
||||
|
||||
if not result.get("goodreads_score"):
|
||||
# 备选:从可见文本中提取
|
||||
rating_div = soup.find("div", attrs={"class": re.compile(r"RatingStatistics__rating")})
|
||||
if rating_div:
|
||||
txt = rating_div.get_text(strip=True)
|
||||
try:
|
||||
result["goodreads_score"] = float(txt)
|
||||
except ValueError:
|
||||
pass
|
||||
|
||||
if not result.get("goodreads_score"):
|
||||
# 再备选:页面全文正则("4.18" 后跟 "ratings")
|
||||
fallback = re.search(r'(\d\.\d{1,2})\s*[\d,]+\s*rating', html)
|
||||
if fallback:
|
||||
try:
|
||||
result["goodreads_score"] = float(fallback.group(1))
|
||||
except ValueError:
|
||||
pass
|
||||
|
||||
# 分类标签 — genre links
|
||||
genre_links = soup.select("span.BookPageMetadataSection__genreButton a")
|
||||
if not genre_links:
|
||||
genre_links = soup.select("a[href*='/genres/']")
|
||||
if genre_links:
|
||||
tags = []
|
||||
for a in genre_links:
|
||||
tag = a.get_text(strip=True)
|
||||
if tag and tag not in tags and len(tags) < 6:
|
||||
tags.append(tag)
|
||||
if tags:
|
||||
result["tags"] = ",".join(tags)
|
||||
|
||||
# 出版信息 — 在页面文本中查找
|
||||
pub_match = re.search(
|
||||
r'(?:First published|Published)\s+(\w+\s+\d{1,2},\s*\d{4}|\w+\s+\d{4}|\d{4})',
|
||||
html
|
||||
)
|
||||
if pub_match:
|
||||
result["pub_date"] = pub_match.group(1)
|
||||
|
||||
# 封面
|
||||
cover_tag = soup.find("img", attrs={"class": re.compile(r"ResponsiveImage")})
|
||||
if cover_tag:
|
||||
src = cover_tag.get("src", "")
|
||||
if src and "nophoto" not in src:
|
||||
result["cover_url"] = src
|
||||
|
||||
return result if result.get("title") else None
|
||||
|
||||
|
||||
def fetch_goodreads(identifier: str):
|
||||
"""
|
||||
从 Goodreads 刮削图书信息。
|
||||
identifier 可以是:
|
||||
- Goodreads 链接(含 /book/show/)
|
||||
- ISBN(会用搜索页查找)
|
||||
"""
|
||||
identifier = identifier.strip()
|
||||
|
||||
if "goodreads.com/book/show/" in identifier:
|
||||
url = identifier.split("?")[0].split("&")[0]
|
||||
elif _is_isbn(identifier):
|
||||
# 先通过搜索页找到书籍链接
|
||||
search_url = GOODREADS_SEARCH_URL.format(query=_clean_isbn(identifier))
|
||||
_sleep()
|
||||
search_html = _get(search_url)
|
||||
if not search_html:
|
||||
return None
|
||||
# 从搜索结果中提取第一个 /book/show/ 链接
|
||||
match = re.search(r'/book/show/(\d+[^"\s\']*)', search_html)
|
||||
if not match:
|
||||
return None
|
||||
url = f"https://www.goodreads.com/book/show/{match.group(1)}"
|
||||
else:
|
||||
return None
|
||||
|
||||
_sleep()
|
||||
html = _get(url)
|
||||
if not html or "goodreads" not in html.lower():
|
||||
return None
|
||||
|
||||
return _parse_goodreads_html(html)
|
||||
|
||||
|
||||
def fetch_goodreads_score(isbn: str):
|
||||
"""只获取 Goodreads 评分(用于补全其他来源的数据)。"""
|
||||
data = fetch_goodreads(isbn)
|
||||
if data and data.get("goodreads_score"):
|
||||
return data["goodreads_score"]
|
||||
return None
|
||||
|
||||
|
||||
# ── Google Books API ──────────────────────────────────────────────────
|
||||
|
||||
def fetch_google_books(isbn: str):
|
||||
"""通过 Google Books API 获取图书信息(无需 key)。"""
|
||||
isbn = _clean_isbn(isbn)
|
||||
url = GOOGLE_BOOKS_URL.format(isbn=isbn)
|
||||
html = _get(url)
|
||||
if not html:
|
||||
return None
|
||||
|
||||
try:
|
||||
data = json.loads(html)
|
||||
except json.JSONDecodeError:
|
||||
return None
|
||||
|
||||
if data.get("totalItems", 0) == 0:
|
||||
return None
|
||||
|
||||
item = data["items"][0]
|
||||
vol = item.get("volumeInfo", {})
|
||||
|
||||
result = {"title": vol.get("title", "")}
|
||||
|
||||
authors = vol.get("authors", [])
|
||||
if authors:
|
||||
result["author"] = " / ".join(authors)
|
||||
|
||||
if vol.get("publisher"):
|
||||
result["publisher"] = vol["publisher"]
|
||||
|
||||
if vol.get("publishedDate"):
|
||||
result["pub_date"] = vol["publishedDate"][:7] # YYYY-MM
|
||||
|
||||
# 封面
|
||||
images = vol.get("imageLinks", {})
|
||||
cover = images.get("thumbnail") or images.get("smallThumbnail")
|
||||
if cover:
|
||||
# 换成更大图
|
||||
result["cover_url"] = cover.replace("zoom=1", "zoom=3")
|
||||
|
||||
# 分类标签
|
||||
cats = vol.get("categories", [])
|
||||
if cats:
|
||||
result["tags"] = ",".join(cats[:4])
|
||||
|
||||
# 评分(Google Books 有 averageRating 字段)
|
||||
if vol.get("averageRating"):
|
||||
result["goodreads_score"] = float(vol["averageRating"])
|
||||
|
||||
return result if result.get("title") else None
|
||||
|
||||
|
||||
# ── Open Library 兜底 ─────────────────────────────────────────────────
|
||||
|
||||
def fetch_open_library(isbn: str):
|
||||
"""Open Library ISBN API 兜底。"""
|
||||
isbn = _clean_isbn(isbn)
|
||||
url = OPEN_LIBRARY_URL.format(isbn=isbn)
|
||||
html = _get(url)
|
||||
if not html:
|
||||
return None
|
||||
|
||||
try:
|
||||
data = json.loads(html)
|
||||
except json.JSONDecodeError:
|
||||
return None
|
||||
|
||||
key = f"ISBN:{isbn}"
|
||||
if key not in data:
|
||||
return None
|
||||
|
||||
book = data[key]
|
||||
result = {"title": book.get("title", "")}
|
||||
|
||||
authors = book.get("authors", [])
|
||||
if authors:
|
||||
result["author"] = " / ".join(a.get("name", "") for a in authors)
|
||||
|
||||
pubs = book.get("publishers", [])
|
||||
if pubs:
|
||||
result["publisher"] = pubs[0].get("name", "")
|
||||
|
||||
if book.get("publish_date"):
|
||||
result["pub_date"] = book["publish_date"]
|
||||
|
||||
# 封面:直接用 covers API(即使信息里没有封面也能试)
|
||||
cover_url = f"https://covers.openlibrary.org/b/isbn/{isbn}-L.jpg"
|
||||
result["cover_url"] = cover_url
|
||||
|
||||
subjects = book.get("subjects", [])
|
||||
if subjects:
|
||||
tags = [s.get("name", "") for s in subjects[:4]]
|
||||
result["tags"] = ",".join(t for t in tags if t)
|
||||
|
||||
return result if result.get("title") else None
|
||||
|
||||
|
||||
# ── 主入口 ────────────────────────────────────────────────────────────
|
||||
|
||||
def fetch_by_isbn(isbn: str):
|
||||
"""
|
||||
根据 ISBN 自动选择数据源。
|
||||
中文 ISBN → 豆瓣 + Goodreads 补 GR 评分。
|
||||
英文 ISBN → Goodreads 为主力 → Google Books 兜底。
|
||||
"""
|
||||
if _is_chinese_isbn(isbn):
|
||||
result = fetch_douban(isbn)
|
||||
if result:
|
||||
# 尝试补全 Goodreads 评分
|
||||
if not result.get("goodreads_score"):
|
||||
gr_score = fetch_goodreads_score(isbn)
|
||||
if gr_score:
|
||||
result["goodreads_score"] = gr_score
|
||||
return result, "豆瓣"
|
||||
result = fetch_open_library(isbn)
|
||||
if result:
|
||||
return result, "Open Library"
|
||||
else:
|
||||
# 英文书:Goodreads 为主力
|
||||
result = fetch_goodreads(isbn)
|
||||
if result:
|
||||
return result, "Goodreads"
|
||||
# 兜底
|
||||
result = fetch_google_books(isbn)
|
||||
if result:
|
||||
return result, "Google Books"
|
||||
result = fetch_open_library(isbn)
|
||||
if result:
|
||||
return result, "Open Library"
|
||||
|
||||
return None, None
|
||||
|
||||
|
||||
def preview(data: dict, source: str):
|
||||
"""格式化打印刮削结果预览。"""
|
||||
lines = [f"\n📖 《{data.get('title', '未知')}》 [来源: {source}]"]
|
||||
if data.get("author"):
|
||||
lines.append(f" 作者:{data['author']}")
|
||||
if data.get("translator"):
|
||||
lines.append(f" 译者:{data['translator']}")
|
||||
if data.get("publisher") or data.get("pub_date"):
|
||||
pub = " · ".join(filter(None, [data.get("publisher"), data.get("pub_date")]))
|
||||
lines.append(f" 出版:{pub}")
|
||||
if data.get("douban_score"):
|
||||
lines.append(f" 豆瓣:{data['douban_score']}")
|
||||
if data.get("goodreads_score"):
|
||||
lines.append(f" GR:{data['goodreads_score']}")
|
||||
if data.get("tags"):
|
||||
lines.append(f" 标签:{data['tags']}")
|
||||
if data.get("cover_url"):
|
||||
lines.append(f" 封面:{data['cover_url']}")
|
||||
return "\n".join(lines)
|
||||
Reference in New Issue
Block a user