diff --git a/.gitignore b/.gitignore index c18dd8d..db75bd5 100644 --- a/.gitignore +++ b/.gitignore @@ -1 +1,4 @@ __pycache__/ +.venv/ +books.db +output/ diff --git a/bookshelf.py b/bookshelf.py index 61f38d3..35092e5 100644 --- a/bookshelf.py +++ b/bookshelf.py @@ -5,6 +5,7 @@ import argparse import sys import db import web +import scrape STATUS_CHOICES = ["to-read", "reading", "read"] @@ -78,6 +79,72 @@ def cmd_delete(args): print(f"🗑️ 已删除:《{book['title']}》(ID: {args.id})") +def cmd_fetch(args): + """刮削图书元数据,可选择直接写入数据库。""" + identifier = args.identifier + + # 选择刮削策略 + if "douban.com" in identifier: + print(f"🔍 正在从豆瓣刮削:{identifier}") + data = scrape.fetch_douban(identifier) + source = "豆瓣" + elif "goodreads.com" in identifier: + print(f"🔍 正在从 Goodreads 刮削:{identifier}") + data = scrape.fetch_goodreads(identifier) + source = "Goodreads" + elif scrape._is_isbn(identifier): + print(f"🔍 正在刮削 ISBN:{identifier}") + data, source = scrape.fetch_by_isbn(identifier) + else: + print("❌ 请提供有效的 ISBN(10/13位)、豆瓣链接或 Goodreads 链接。") + sys.exit(1) + + if not data: + print("❌ 刮削失败,未能获取到图书信息。") + print(" 提示:豆瓣偶尔会拒绝请求(418/403),请稍后重试。") + sys.exit(1) + + # 打印预览 + print(scrape.preview(data, source)) + print() + + if args.update_id: + # 更新已有记录 + db.init_db() + book = db.get_book(args.update_id) + if not book: + print(f"❌ 未找到 ID 为 {args.update_id} 的书目。") + sys.exit(1) + # 只覆盖刮削到的字段,保留用户已设置的字段 + updates = {k: v for k, v in data.items() if v} + if args.format: + updates["format"] = args.format + if args.status: + updates["status"] = args.status + db.update_book(args.update_id, **updates) + print(f"✅ 已更新:《{book['title']}》→《{data.get('title', book['title'])}》(ID: {args.update_id})") + + elif args.add: + # 写入新记录 + db.init_db() + book_id = db.add_book( + data.get("title", ""), + author=data.get("author"), + translator=data.get("translator"), + publisher=data.get("publisher"), + pub_date=data.get("pub_date"), + cover_url=data.get("cover_url"), + format=args.format or "paper", + status=args.status or "to-read", + douban_score=data.get("douban_score"), + goodreads_score=data.get("goodreads_score"), + tags=data.get("tags", ""), + ) + print(f"✅ 已添加:《{data.get('title')}》(ID: {book_id})") + else: + print("💡 加 --add 选项可直接写入数据库,或用 --update-id 更新已有记录。") + + def cmd_build(args): db.init_db() output_path = web.build() @@ -142,6 +209,15 @@ def main(): p_del.add_argument("id", type=int, help="书目 ID") p_del.set_defaults(func=cmd_delete) + # --- fetch --- + p_fetch = sub.add_parser("fetch", help="从豆瓣/Google Books 刮削书目信息") + p_fetch.add_argument("identifier", help="ISBN(10/13位)或豆瓣图书链接") + p_fetch.add_argument("--add", action="store_true", help="刮削后直接写入数据库") + p_fetch.add_argument("--update-id", type=int, metavar="ID", help="刮削后更新指定 ID 的书目") + p_fetch.add_argument("--format", choices=FORMAT_CHOICES, help="书籍格式(入库时使用)") + p_fetch.add_argument("--status", choices=STATUS_CHOICES, help="阅读状态(入库时使用)") + p_fetch.set_defaults(func=cmd_fetch) + # --- build --- p_build = sub.add_parser("build", help="生成展示网页") p_build.set_defaults(func=cmd_build) diff --git a/requirements.txt b/requirements.txt new file mode 100644 index 0000000..52aefce --- /dev/null +++ b/requirements.txt @@ -0,0 +1,3 @@ +requests>=2.31.0 +beautifulsoup4>=4.12.0 +playwright>=1.40.0 diff --git a/scrape.py b/scrape.py new file mode 100644 index 0000000..06e568b --- /dev/null +++ b/scrape.py @@ -0,0 +1,553 @@ +"""图书元数据刮削模块。 + +数据源策略: +- 中文书:豆瓣网页爬取 + Goodreads 补评分 +- 英文书:Goodreads 为主力 +- 兜底:Google Books API / Open Library ISBN API +""" + +import json +import re +import time +import random + +try: + import requests as _requests + _HAS_REQUESTS = True +except ImportError: + import urllib.request + _HAS_REQUESTS = False + +# ── 常量 ──────────────────────────────────────────────────────────── + +_UA = ( + "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) " + "AppleWebKit/537.36 (KHTML, like Gecko) " + "Chrome/123.0.0.0 Safari/537.36" +) + +DOUBAN_ISBN_URL = "https://book.douban.com/isbn/{isbn}" +DOUBAN_SUBJECT_URL = "https://book.douban.com/subject/{sid}/" +GOODREADS_BOOK_URL = "https://www.goodreads.com/book/show/{book_id}" +GOODREADS_SEARCH_URL = "https://www.goodreads.com/search?q={query}" +GOOGLE_BOOKS_URL = "https://www.googleapis.com/books/v1/volumes?q=isbn:{isbn}" +OPEN_LIBRARY_URL = ( + "https://openlibrary.org/api/books?bibkeys=ISBN:{isbn}&format=json&jscmd=data" +) +OPEN_LIBRARY_COVER = "https://covers.openlibrary.org/b/isbn/{isbn}-L.jpg" + +# ── 工具函数 ───────────────────────────────────────────────────────── + +def _get(url, timeout=15): + """带 UA 的 HTTP GET,返回响应文本或 None。优先使用 requests 库。""" + headers = {"User-Agent": _UA} + if _HAS_REQUESTS: + try: + resp = _requests.get(url, headers=headers, timeout=timeout, + allow_redirects=True) + if resp.status_code == 200: + return resp.text + return None + except Exception: + return None + else: + import urllib.request + req = urllib.request.Request(url, headers=headers) + try: + with urllib.request.urlopen(req, timeout=timeout) as resp: + return resp.read().decode("utf-8", errors="replace") + except Exception: + return None + + +def _get_with_playwright(url, timeout=30000): + """用 Playwright 无头浏览器获取页面 HTML(用于豆瓣等反爬严格的站点)。""" + try: + from playwright.sync_api import sync_playwright + except ImportError: + print("⚠️ 需要安装 Playwright:pip install playwright && playwright install") + return None + + html = None + # 多种启动策略:系统 Chrome → Playwright Chromium → 不同 headless 模式 + launch_options = [ + {"channel": "chrome", "headless": True}, # 系统 Chrome + {"channel": "chromium", "headless": True}, # 系统 Chromium + {"headless": True, # Playwright 自带 + "args": ["--disable-blink-features=AutomationControlled"]}, + {"headless": False, # 非 headless(最后手段) + "args": ["--headless=new"]}, # Chrome 新 headless 模式 + ] + + try: + with sync_playwright() as p: + browser = None + for opts in launch_options: + try: + browser = p.chromium.launch(**opts) + break + except Exception: + continue + + if not browser: + print("⚠️ 无法启动浏览器。请确保已安装 Chrome 或运行:playwright install") + return None + + context = browser.new_context( + user_agent=_UA, + viewport={"width": 1280, "height": 800}, + locale="zh-CN", + ) + page = context.new_page() + page.goto(url, wait_until="domcontentloaded", timeout=timeout) + # 等待核心内容出现 + try: + page.wait_for_selector("#info", timeout=8000) + except Exception: + pass # 即使超时也继续,页面可能已部分加载 + html = page.content() + browser.close() + except Exception as e: + print(f"⚠️ Playwright 访问失败:{e}") + return None + + return html + + +def _sleep(): + """礼貌延时,避免被封。""" + time.sleep(random.uniform(1.0, 2.5)) + + +def _is_isbn(s): + digits = re.sub(r"[-\s]", "", s) + return re.fullmatch(r"\d{10}|\d{13}", digits) is not None + + +def _clean_isbn(s): + return re.sub(r"[-\s]", "", s) + + +def _is_chinese_isbn(isbn): + """以 978-7 / 7 开头的 ISBN 通常是中文出版物。""" + d = _clean_isbn(isbn) + return d.startswith("9787") or d.startswith("7") + + +# ── 豆瓣刮削 ───────────────────────────────────────────────────────── + +def _parse_douban_html(html): + """解析豆瓣图书详情页 HTML,返回 dict。需要 beautifulsoup4。""" + try: + from bs4 import BeautifulSoup + except ImportError: + raise RuntimeError("请先安装 beautifulsoup4:pip install beautifulsoup4") + + soup = BeautifulSoup(html, "html.parser") + + result = {} + + # 书名 + title_tag = soup.find("h1") + if title_tag: + span = title_tag.find("span") + result["title"] = (span or title_tag).get_text(strip=True) + + # 封面 + cover_tag = soup.find("a", attrs={"class": "nbg"}) + if cover_tag: + img = cover_tag.find("img") + if img: + result["cover_url"] = img.get("src", "") + + # 豆瓣评分 + rating_tag = soup.find("strong", attrs={"class": "ll rating_num"}) + if rating_tag: + txt = rating_tag.get_text(strip=True) + try: + result["douban_score"] = float(txt) + except ValueError: + pass + + # 作者、出版社等在 #info 块 + info = soup.find("div", id="info") + if info: + # 方法1:通过 标签精确提取 + for span in info.find_all("span", class_="pl"): + label = span.get_text(strip=True).rstrip(":: ") + # 获取同级后续文本和链接 + values = [] + for sib in span.next_siblings: + if hasattr(sib, 'name'): + if sib.name == 'br': + break + if sib.name == 'span' and 'pl' in (sib.get('class') or []): + break + if sib.name == 'a': + values.append(sib.get_text(strip=True)) + elif sib.name == 'span': + values.append(sib.get_text(strip=True)) + else: + t = str(sib).strip().strip("::/·, ") + if t: + values.append(t) + value = " ".join(v for v in values if v).strip(" /::") + + if not value: + continue + + if "作者" in label: + # 去掉 [国籍] 标注 + author_clean = re.sub(r"\[.*?\]", "", value).strip(" /") + # 清理多余空格 + author_clean = re.sub(r"\s+", " ", author_clean).strip() + if author_clean: + result["author"] = author_clean + elif "译者" in label: + result["translator"] = re.sub(r"\s+", " ", value).strip(" /") + elif "出版社" in label: + result["publisher"] = value + elif "出版年" in label: + result["pub_date"] = value + elif "ISBN" in label.upper(): + result["isbn"] = value + + # 方法2:兜底 — 如果 span.pl 没有拿到,用纯文本 + if not result.get("author"): + text = info.get_text(separator="\n") + for line in text.splitlines(): + line = line.strip() + if "作者" in line: + parts = re.split(r"[::]", line, 1) + if len(parts) == 2 and parts[1].strip(): + author = re.sub(r"\[.*?\]", "", parts[1]).strip(" /") + if author: + result["author"] = author + break + + # 标签 + tag_links = soup.select("a.tag") + if tag_links: + tags = [a.get_text(strip=True) for a in tag_links[:6]] + result["tags"] = ",".join(tags) + + return result if result.get("title") else None + + +def fetch_douban(identifier: str): + """ + 从豆瓣刮削图书信息。 + identifier 可以是: + - ISBN(10位或13位) + - 豆瓣图书链接(含 subject/id) + - 豆瓣 subject id(纯数字) + """ + identifier = identifier.strip() + + if _is_isbn(identifier): + url = DOUBAN_ISBN_URL.format(isbn=_clean_isbn(identifier)) + elif "douban.com/subject/" in identifier: + url = identifier.split("?")[0].rstrip("/") + "/" + elif re.fullmatch(r"\d+", identifier): + url = DOUBAN_SUBJECT_URL.format(sid=identifier) + else: + return None + + _sleep() + html = _get_with_playwright(url) + if not html or "豆瓣" not in html: + # Playwright 失败时尝试普通 requests(偶尔可能成功) + html = _get(url) + if not html or "豆瓣" not in html: + return None + + data = _parse_douban_html(html) + return data + + +# ── Goodreads 刮削 ──────────────────────────────────────────────────── + +def _parse_goodreads_html(html): + """解析 Goodreads 图书详情页 HTML,返回 dict。""" + try: + from bs4 import BeautifulSoup + except ImportError: + raise RuntimeError("请先安装 beautifulsoup4:pip install beautifulsoup4") + + soup = BeautifulSoup(html, "html.parser") + result = {} + + # 书名 —

+ title_tag = soup.find("h1", attrs={"data-testid": "bookTitle"}) + if not title_tag: + title_tag = soup.find("h1") + if title_tag: + result["title"] = title_tag.get_text(strip=True) + + # 作者 — + author_tags = soup.select("span.ContributorLink__name") + if author_tags: + authors = [] + for a in author_tags: + name = a.get_text(strip=True) + # 排除角色标注如 "(Primary Contributor)" + if name and "Contributor" not in name: + authors.append(name) + if authors: + result["author"] = " / ".join(dict.fromkeys(authors)) # 去重保序 + + # 评分 — 在页面文本中匹配 "X.XX" 紧跟 ratings 数字 + # 格式通常是 "3.8015,305 ratings" 或类似 + rating_match = re.search( + r'"ratingValue"\s*:\s*([\d.]+)', html + ) + if rating_match: + try: + result["goodreads_score"] = float(rating_match.group(1)) + except ValueError: + pass + + if not result.get("goodreads_score"): + # 备选:从可见文本中提取 + rating_div = soup.find("div", attrs={"class": re.compile(r"RatingStatistics__rating")}) + if rating_div: + txt = rating_div.get_text(strip=True) + try: + result["goodreads_score"] = float(txt) + except ValueError: + pass + + if not result.get("goodreads_score"): + # 再备选:页面全文正则("4.18" 后跟 "ratings") + fallback = re.search(r'(\d\.\d{1,2})\s*[\d,]+\s*rating', html) + if fallback: + try: + result["goodreads_score"] = float(fallback.group(1)) + except ValueError: + pass + + # 分类标签 — genre links + genre_links = soup.select("span.BookPageMetadataSection__genreButton a") + if not genre_links: + genre_links = soup.select("a[href*='/genres/']") + if genre_links: + tags = [] + for a in genre_links: + tag = a.get_text(strip=True) + if tag and tag not in tags and len(tags) < 6: + tags.append(tag) + if tags: + result["tags"] = ",".join(tags) + + # 出版信息 — 在页面文本中查找 + pub_match = re.search( + r'(?:First published|Published)\s+(\w+\s+\d{1,2},\s*\d{4}|\w+\s+\d{4}|\d{4})', + html + ) + if pub_match: + result["pub_date"] = pub_match.group(1) + + # 封面 + cover_tag = soup.find("img", attrs={"class": re.compile(r"ResponsiveImage")}) + if cover_tag: + src = cover_tag.get("src", "") + if src and "nophoto" not in src: + result["cover_url"] = src + + return result if result.get("title") else None + + +def fetch_goodreads(identifier: str): + """ + 从 Goodreads 刮削图书信息。 + identifier 可以是: + - Goodreads 链接(含 /book/show/) + - ISBN(会用搜索页查找) + """ + identifier = identifier.strip() + + if "goodreads.com/book/show/" in identifier: + url = identifier.split("?")[0].split("&")[0] + elif _is_isbn(identifier): + # 先通过搜索页找到书籍链接 + search_url = GOODREADS_SEARCH_URL.format(query=_clean_isbn(identifier)) + _sleep() + search_html = _get(search_url) + if not search_html: + return None + # 从搜索结果中提取第一个 /book/show/ 链接 + match = re.search(r'/book/show/(\d+[^"\s\']*)', search_html) + if not match: + return None + url = f"https://www.goodreads.com/book/show/{match.group(1)}" + else: + return None + + _sleep() + html = _get(url) + if not html or "goodreads" not in html.lower(): + return None + + return _parse_goodreads_html(html) + + +def fetch_goodreads_score(isbn: str): + """只获取 Goodreads 评分(用于补全其他来源的数据)。""" + data = fetch_goodreads(isbn) + if data and data.get("goodreads_score"): + return data["goodreads_score"] + return None + + +# ── Google Books API ────────────────────────────────────────────────── + +def fetch_google_books(isbn: str): + """通过 Google Books API 获取图书信息(无需 key)。""" + isbn = _clean_isbn(isbn) + url = GOOGLE_BOOKS_URL.format(isbn=isbn) + html = _get(url) + if not html: + return None + + try: + data = json.loads(html) + except json.JSONDecodeError: + return None + + if data.get("totalItems", 0) == 0: + return None + + item = data["items"][0] + vol = item.get("volumeInfo", {}) + + result = {"title": vol.get("title", "")} + + authors = vol.get("authors", []) + if authors: + result["author"] = " / ".join(authors) + + if vol.get("publisher"): + result["publisher"] = vol["publisher"] + + if vol.get("publishedDate"): + result["pub_date"] = vol["publishedDate"][:7] # YYYY-MM + + # 封面 + images = vol.get("imageLinks", {}) + cover = images.get("thumbnail") or images.get("smallThumbnail") + if cover: + # 换成更大图 + result["cover_url"] = cover.replace("zoom=1", "zoom=3") + + # 分类标签 + cats = vol.get("categories", []) + if cats: + result["tags"] = ",".join(cats[:4]) + + # 评分(Google Books 有 averageRating 字段) + if vol.get("averageRating"): + result["goodreads_score"] = float(vol["averageRating"]) + + return result if result.get("title") else None + + +# ── Open Library 兜底 ───────────────────────────────────────────────── + +def fetch_open_library(isbn: str): + """Open Library ISBN API 兜底。""" + isbn = _clean_isbn(isbn) + url = OPEN_LIBRARY_URL.format(isbn=isbn) + html = _get(url) + if not html: + return None + + try: + data = json.loads(html) + except json.JSONDecodeError: + return None + + key = f"ISBN:{isbn}" + if key not in data: + return None + + book = data[key] + result = {"title": book.get("title", "")} + + authors = book.get("authors", []) + if authors: + result["author"] = " / ".join(a.get("name", "") for a in authors) + + pubs = book.get("publishers", []) + if pubs: + result["publisher"] = pubs[0].get("name", "") + + if book.get("publish_date"): + result["pub_date"] = book["publish_date"] + + # 封面:直接用 covers API(即使信息里没有封面也能试) + cover_url = f"https://covers.openlibrary.org/b/isbn/{isbn}-L.jpg" + result["cover_url"] = cover_url + + subjects = book.get("subjects", []) + if subjects: + tags = [s.get("name", "") for s in subjects[:4]] + result["tags"] = ",".join(t for t in tags if t) + + return result if result.get("title") else None + + +# ── 主入口 ──────────────────────────────────────────────────────────── + +def fetch_by_isbn(isbn: str): + """ + 根据 ISBN 自动选择数据源。 + 中文 ISBN → 豆瓣 + Goodreads 补 GR 评分。 + 英文 ISBN → Goodreads 为主力 → Google Books 兜底。 + """ + if _is_chinese_isbn(isbn): + result = fetch_douban(isbn) + if result: + # 尝试补全 Goodreads 评分 + if not result.get("goodreads_score"): + gr_score = fetch_goodreads_score(isbn) + if gr_score: + result["goodreads_score"] = gr_score + return result, "豆瓣" + result = fetch_open_library(isbn) + if result: + return result, "Open Library" + else: + # 英文书:Goodreads 为主力 + result = fetch_goodreads(isbn) + if result: + return result, "Goodreads" + # 兜底 + result = fetch_google_books(isbn) + if result: + return result, "Google Books" + result = fetch_open_library(isbn) + if result: + return result, "Open Library" + + return None, None + + +def preview(data: dict, source: str): + """格式化打印刮削结果预览。""" + lines = [f"\n📖 《{data.get('title', '未知')}》 [来源: {source}]"] + if data.get("author"): + lines.append(f" 作者:{data['author']}") + if data.get("translator"): + lines.append(f" 译者:{data['translator']}") + if data.get("publisher") or data.get("pub_date"): + pub = " · ".join(filter(None, [data.get("publisher"), data.get("pub_date")])) + lines.append(f" 出版:{pub}") + if data.get("douban_score"): + lines.append(f" 豆瓣:{data['douban_score']}") + if data.get("goodreads_score"): + lines.append(f" GR:{data['goodreads_score']}") + if data.get("tags"): + lines.append(f" 标签:{data['tags']}") + if data.get("cover_url"): + lines.append(f" 封面:{data['cover_url']}") + return "\n".join(lines)