feat: 重构图书墙交互及多源高阶信息抓取注入系统

1. UI 层:
- 抛弃网格卡片结构,改版为沉浸式的“图书墙”(1:1.43 原比例包含模式)。
- 新增悬浮深色遮罩系统(hover-details),解绑评分点击阻塞。
- 过滤系统默认进入“在读”状态页,彻底移除冗余无效的全选按钮。
2. 抓取与刮削层 (scrape.py):
- 抛弃对 ISBN 查询 Google Books 的严重强依赖。
- 重构抓取链路,首选 Goodreads 与特供的 Books.com.tw(博客来) 高精爬取港台原版资料。
- 建立 Amazon/Goodreads 图片去码净化正则,告别低分辨率与畸形图片。
- 修复 bs4 (.select_first) 解析选择器在旧版环境的兼容性异常。
3. Agent AI 数据管线设计 (bookshelf.py):
- 开放基于 --json 与 --cover 的强力直接注入功能。
- 升级 add 命令作为最优先级的录入手段,免去繁琐的 enrich 阶段,直接连带高清图和满载元数据强插。
- 建立入库反重名智能拦截 (title + author 交叉校验),形成完美的防呆与自动化壁垒。
This commit is contained in:
kai
2026-03-26 11:42:43 +08:00
parent 69c34d4faf
commit 0801f7c7df
9 changed files with 684 additions and 133 deletions
+277 -22
View File
@@ -3,6 +3,7 @@
import argparse
import sys
import sqlite3
import db
import web
import scrape
@@ -11,27 +12,53 @@ import scrape
STATUS_CHOICES = ["to-read", "reading", "read"]
FORMAT_CHOICES = ["paper", "ebook"]
def cmd_add(args):
db.init_db()
book_id = db.add_book(
args.title,
author=args.author,
translator=args.translator,
publisher=args.publisher,
pub_date=args.pub_date,
cover_url=args.cover_url,
format=args.format,
status=args.status,
rating=args.rating,
douban_score=args.douban_score,
goodreads_score=args.goodreads_score,
tags=args.tags or "",
notes=args.notes,
start_date=args.start_date,
finish_date=args.finish_date,
)
print(f"✅ 已添加:《{args.title}》(ID: {book_id})")
kwargs = {
"author": args.author, "translator": args.translator,
"publisher": args.publisher, "pub_date": args.pub_date,
"cover_url": args.cover_url, "format": args.format,
"status": args.status, "rating": args.rating,
"douban_score": args.douban_score, "goodreads_score": args.goodreads_score,
"tags": args.tags or "", "notes": args.notes,
"start_date": args.start_date, "finish_date": args.finish_date,
"title_en": args.title_en or "", "author_en": args.author_en or "",
"douban_url": args.douban_url or "", "goodreads_url": args.goodreads_url or ""
}
if getattr(args, "json", None):
import json
try:
extra = json.loads(args.json)
kwargs.update(extra)
except Exception as e:
print(f"❌ JSON 解析失败: {e}")
return
title = kwargs.pop("title", args.title)
if not title:
print("❌ 必须提供书名 (title)。")
return
all_books = db.list_books()
for b in all_books:
db_title = b['title'].lower()
t_lower = title.lower()
if t_lower == db_title or (len(t_lower) > 3 and (t_lower in db_title or db_title in t_lower)):
a_in = (kwargs.get("author") or "").lower()
a_db = (b.get("author") or "").lower()
if not a_in or not a_db or a_in in a_db or a_db in a_in:
print(f"⚠️ 库中可能已存在同名记录:[ID: {b['id']}] 《{b['title']}》 (作者: {b.get('author', '未知')})。")
print(" 已取消添加。若确认并非同一本请修改书名/作者,若需更新请使用 `update` 命令。")
return
book_id = db.add_book(title)
updates = {k: v for k, v in kwargs.items() if v}
if updates:
db.update_book(book_id, **updates)
print(f"✅ 已添加并录入数据:《{title}》(ID: {book_id})")
def cmd_list(args):
@@ -48,27 +75,55 @@ def cmd_list(args):
print(f" {status_icon} [{b['id']:>3}] 《{b['title']}》— {b['author'] or '未知'}"
f" {fmt_icon}{rating_str}{tags_str}")
def cmd_update(args):
db.init_db()
book = db.get_book(args.id)
if not book:
print(f"❌ 未找到 ID 为 {args.id} 的书目。")
sys.exit(1)
updates = {}
# 获取命令行常规参数
for field in ["title", "author", "translator", "publisher", "pub_date",
"cover_url", "format", "status", "rating", "douban_score",
"goodreads_score", "tags", "notes", "start_date", "finish_date"]:
"goodreads_score", "tags", "notes", "start_date", "finish_date",
"title_en", "author_en", "douban_url", "goodreads_url"]:
val = getattr(args, field.replace("-", "_"), None)
if val is not None:
updates[field] = val
# 从 JSON 批量注入参数(AI优先)
if getattr(args, "json", None):
import json
try:
updates.update(json.loads(args.json))
except Exception as e:
print(f"❌ JSON 解析失败: {e}")
return
# 从本地图片注入封面(AI优先)
if getattr(args, "cover", None):
import os
if not os.path.exists(args.cover):
print(f"❌ 找不到指定的本封面: {args.cover}")
return
try:
with open(args.cover, "rb") as f:
updates["cover_blob"] = f.read()
updates["cover_local"] = 1
print(f"🖼️ 已应用本地封面: {args.cover}")
except Exception as e:
print(f"❌ 读取封面文件失败: {e}")
return
if not updates:
print("⚠️ 未指定任何要更新的字段。")
return
db.update_book(args.id, **updates)
print(f"✅ 已更新:《{book['title']}》(ID: {args.id})")
def cmd_delete(args):
db.init_db()
book = db.get_book(args.id)
@@ -139,12 +194,187 @@ def cmd_fetch(args):
douban_score=data.get("douban_score"),
goodreads_score=data.get("goodreads_score"),
tags=data.get("tags", ""),
title_en=data.get("title_en", ""),
author_en=data.get("author_en", ""),
douban_url=data.get("douban_url", ""),
goodreads_url=data.get("goodreads_url", ""),
)
print(f"✅ 已添加:《{data.get('title')}》(ID: {book_id})")
else:
print("💡 加 --add 选项可直接写入数据库,或用 --update-id <ID> 更新已有记录。")
def _is_match(b_title, b_author, s_title, s_title_en, s_author):
import re
def canon(s):
return re.sub(r'[^\w\u4e00-\u9fff]', '', str(s).lower()) if s else ""
bt = canon(b_title)
st = canon(s_title)
ste = canon(s_title_en)
title_match = (bt in st or st in bt) if (bt and st) else False
if not title_match and ste:
title_match = (bt in ste or ste in bt) if bt else False
ba = canon(b_author)
sa = canon(s_author)
author_match = (ba in sa or sa in ba) if (ba and sa) else False
# 书名匹配或作者匹配之一满足即可认为是同一个记录
return title_match or author_match
def cmd_enrich(args):
"""自动补全数据库中缺失元数据的书目。"""
db.init_db()
books = db.list_books()
to_enrich = []
for b in books:
# 只要没有豆瓣或 GR 的链接,或者没有封面,就尝试 enrich
if not b.get("cover_url") or not b.get("douban_url") or not b.get("goodreads_url"):
to_enrich.append(b)
if args.id:
to_enrich = [b for b in to_enrich if b["id"] == args.id]
if not to_enrich:
print("🎉 所有书目数据已相对完整,无需 enrich。")
return
print(f"🔍 共有 {len(to_enrich)} 本书需补全或验证信息...")
import urllib.parse
for b in to_enrich:
print(f"\n=> 正在处理 [{b['id']}]: 《{b['title']}》 作者: {b.get('author','')} ...")
scraped = None
src = ""
search_term = b['title']
if b.get('author'): search_term += f" {b['author']}"
# 策略 1: 直接向 Goodreads 询问(它对原版图书封面最标准、非促销封套)
print(" 🔍 尝试通过 Goodreads 直搜...")
gr_data = scrape.fetch_goodreads(search_term)
if gr_data and _is_match(b['title'], b.get('author'), gr_data.get('title'), gr_data.get('title_en'), gr_data.get('author')):
scraped = gr_data
src = "Goodreads"
# 策略 2: 如果 GR 失败或缺信息,对于繁体中文,尝试博客来
if not scraped or not scraped.get("cover_url"):
print(" 🔍 尝试通过博客来(Books.com.tw)直搜...")
tw_data = scrape.fetch_books_tw(search_term)
if tw_data and _is_match(b['title'], b.get('author'), tw_data.get('title'), None, tw_data.get('author')):
if not scraped: scraped = tw_data
else:
for k, v in tw_data.items():
if not scraped.get(k): scraped[k] = v
scraped["format"] = "paper" # 博客来主要是实体书
src = "博客来" + ((" / " + src) if src else "")
# 策略 3: Google Books 仅作为属性补充,**决不拿它的 ISBN 去瞎关联豆瓣**
query = f"intitle:{b['title']}"
if b.get('author'): query += f"+inauthor:{b['author']}"
q = urllib.parse.quote(query)
url = f"https://www.googleapis.com/books/v1/volumes?q={q}"
try:
html = scrape._get(url)
if html:
import json
data = json.loads(html)
if data.get("totalItems", 0) > 0:
vol = data["items"][0].get("volumeInfo", {})
# 只要匹配度过关,就拿它的 publisher, pub_date, ISBN
if _is_match(b['title'], b.get('author'), vol.get("title"), None, ",".join(vol.get("authors", []))):
isbn = next((ids.get("identifier") for ids in vol.get("industryIdentifiers", []) if ids.get("type") in ("ISBN_13", "ISBN_10")), None)
if not scraped: scraped = {"title": vol.get("title")}
if vol.get("publisher") and not scraped.get("publisher"): scraped["publisher"] = vol["publisher"]
if vol.get("publishedDate") and not scraped.get("pub_date"): scraped["pub_date"] = vol["publishedDate"][:7]
if isbn and not scraped.get("isbn"): scraped["isbn"] = isbn
src = src or "Google Books"
except Exception:
pass
# 策略 4: 如果有 ISBN,去豆瓣白嫖评分和出版信息(但为了防止封面被覆盖为促销版,不轻易覆写封面)
if scraped and scraped.get("isbn"):
douban_data = scrape.fetch_douban(scraped["isbn"])
if douban_data and _is_match(b['title'], b.get('author'), douban_data.get('title'), douban_data.get('title_en'), douban_data.get('author')):
for k, v in douban_data.items():
if k == "cover_url" and scraped.get("cover_url"): continue # 优先信任前面外站的原版封面
if not scraped.get(k): scraped[k] = v
src += " / 豆瓣"
if scraped:
updates = {}
# 新刮削的标题如果是外文但不等于原书名,放到 title_en 保护原中文书名
if scraped.get("title") and not _is_match(b["title"], None, scraped["title"], None, None):
if not b.get("title_en"):
updates["title_en"] = scraped["title"]
if scraped.get("author") and not _is_match(b["author"], None, scraped["author"], None, None):
if not b.get("author_en"):
updates["author_en"] = scraped["author"]
for k in ["author", "translator", "publisher", "pub_date", "cover_url", "tags",
"douban_score", "goodreads_score", "isbn", "title_en", "author_en", "douban_url", "goodreads_url"]:
if scraped.get(k) and not b.get(k):
updates[k] = scraped[k]
if updates:
db.update_book(b["id"], **updates)
print(f" ✅ 已补全字段: {', '.join(updates.keys())} (来源: {src})")
else:
print(f" 👍 已匹配到信息,但无需补充新字段 (来源: {src})")
else:
print(" 🚫 彻底失败,保留原样。")
# 延时避免被反爬
scrape._sleep()
def cmd_covers(args):
"""批量下载封面存入数据库 BLOB。"""
db.init_db()
books = db.list_books()
to_download = [b for b in books if b.get("cover_url") and not b.get("cover_blob")]
if args.id:
to_download = [b for b in to_download if b["id"] == args.id]
print(f"🖼️ 共有 {len(to_download)} 本书需要下载封面到数据库...")
conn = db._connect()
try:
import requests
except ImportError:
import urllib.request
count = 0
for b in to_download:
url = b["cover_url"]
print(f"下载 [{b['id']}] 《{b['title']}》封面: {url}")
try:
# 豆瓣防盗链需要 Referer,而且可能需要 Playwright 那个 UA
headers = {"User-Agent": scrape._UA, "Referer": "https://book.douban.com/"}
if scrape._HAS_REQUESTS:
resp = requests.get(url, headers=headers, timeout=15)
if resp.status_code == 200:
blob = resp.content
else:
blob = None
else:
req = urllib.request.Request(url, headers=headers)
with urllib.request.urlopen(req, timeout=15) as resp:
blob = resp.read()
if blob:
conn.execute("UPDATE books SET cover_blob = ? WHERE id = ?", (sqlite3.Binary(blob), b["id"]))
conn.commit()
print(" ✅ 成功")
count += 1
else:
print(" ❌ 获取失败")
scrape._sleep()
except Exception as e:
print(f" ❌ 网络错误: {e}")
conn.close()
print(f"🎉 封面下载完成,成功 {count} 本。")
def cmd_build(args):
db.init_db()
output_path = web.build()
@@ -175,6 +405,10 @@ def main():
p_add.add_argument("--notes", help="笔记/简评")
p_add.add_argument("--start-date", help="开始阅读日期")
p_add.add_argument("--finish-date", help="完成阅读日期")
p_add.add_argument("--title-en", help="外文原名")
p_add.add_argument("--author-en", help="外文作者名")
p_add.add_argument("--douban-url", help="豆瓣链接")
p_add.add_argument("--goodreads-url", help="Goodreads 链接")
p_add.set_defaults(func=cmd_add)
# --- list ---
@@ -202,6 +436,10 @@ def main():
p_upd.add_argument("--notes", help="笔记/简评")
p_upd.add_argument("--start-date", help="开始阅读日期")
p_upd.add_argument("--finish-date", help="完成阅读日期")
p_upd.add_argument("--title-en", help="外文原名")
p_upd.add_argument("--author-en", help="外文作者名")
p_upd.add_argument("--douban-url", help="豆瓣链接")
p_upd.add_argument("--goodreads-url", help="Goodreads 链接")
p_upd.set_defaults(func=cmd_update)
# --- delete ---
@@ -218,10 +456,27 @@ def main():
p_fetch.add_argument("--status", choices=STATUS_CHOICES, help="阅读状态(入库时使用)")
p_fetch.set_defaults(func=cmd_fetch)
# --- enrich ---
p_enrich = sub.add_parser("enrich", help="批量刮削补全缺失信息")
p_enrich.add_argument("--id", type=int, help="只补全特定 ID 的书")
p_enrich.set_defaults(func=cmd_enrich)
# --- covers ---
p_covers = sub.add_parser("covers", help="批量下载并保存封面到数据库")
p_covers.add_argument("--id", type=int, help="只下载特定 ID 的书")
p_covers.set_defaults(func=cmd_covers)
# --- build ---
p_build = sub.add_parser("build", help="生成展示网页")
p_build.set_defaults(func=cmd_build)
# --- update ---
p_update = sub.add_parser("update", help="(高级/AI专用)通过 JSON 直接更新元数据或本地封面")
p_update.add_argument("id", type=int, help="书籍记录的数字 ID")
p_update.add_argument("--json", type=str, help="包含需更新字段的 JSON 字符串 (例: '{\"douban_score\": 9.5}')")
p_update.add_argument("--cover", type=str, help="本地图片文件路径,直接注入 cover_blob")
p_update.set_defaults(func=cmd_update)
args = parser.parse_args()
args.func(args)