feat: 重构图书墙交互及多源高阶信息抓取注入系统
1. UI 层: - 抛弃网格卡片结构,改版为沉浸式的“图书墙”(1:1.43 原比例包含模式)。 - 新增悬浮深色遮罩系统(hover-details),解绑评分点击阻塞。 - 过滤系统默认进入“在读”状态页,彻底移除冗余无效的全选按钮。 2. 抓取与刮削层 (scrape.py): - 抛弃对 ISBN 查询 Google Books 的严重强依赖。 - 重构抓取链路,首选 Goodreads 与特供的 Books.com.tw(博客来) 高精爬取港台原版资料。 - 建立 Amazon/Goodreads 图片去码净化正则,告别低分辨率与畸形图片。 - 修复 bs4 (.select_first) 解析选择器在旧版环境的兼容性异常。 3. Agent AI 数据管线设计 (bookshelf.py): - 开放基于 --json 与 --cover 的强力直接注入功能。 - 升级 add 命令作为最优先级的录入手段,免去繁琐的 enrich 阶段,直接连带高清图和满载元数据强插。 - 建立入库反重名智能拦截 (title + author 交叉校验),形成完美的防呆与自动化壁垒。
This commit is contained in:
+277
-22
@@ -3,6 +3,7 @@
|
||||
|
||||
import argparse
|
||||
import sys
|
||||
import sqlite3
|
||||
import db
|
||||
import web
|
||||
import scrape
|
||||
@@ -11,27 +12,53 @@ import scrape
|
||||
STATUS_CHOICES = ["to-read", "reading", "read"]
|
||||
FORMAT_CHOICES = ["paper", "ebook"]
|
||||
|
||||
|
||||
def cmd_add(args):
|
||||
db.init_db()
|
||||
book_id = db.add_book(
|
||||
args.title,
|
||||
author=args.author,
|
||||
translator=args.translator,
|
||||
publisher=args.publisher,
|
||||
pub_date=args.pub_date,
|
||||
cover_url=args.cover_url,
|
||||
format=args.format,
|
||||
status=args.status,
|
||||
rating=args.rating,
|
||||
douban_score=args.douban_score,
|
||||
goodreads_score=args.goodreads_score,
|
||||
tags=args.tags or "",
|
||||
notes=args.notes,
|
||||
start_date=args.start_date,
|
||||
finish_date=args.finish_date,
|
||||
)
|
||||
print(f"✅ 已添加:《{args.title}》(ID: {book_id})")
|
||||
|
||||
kwargs = {
|
||||
"author": args.author, "translator": args.translator,
|
||||
"publisher": args.publisher, "pub_date": args.pub_date,
|
||||
"cover_url": args.cover_url, "format": args.format,
|
||||
"status": args.status, "rating": args.rating,
|
||||
"douban_score": args.douban_score, "goodreads_score": args.goodreads_score,
|
||||
"tags": args.tags or "", "notes": args.notes,
|
||||
"start_date": args.start_date, "finish_date": args.finish_date,
|
||||
"title_en": args.title_en or "", "author_en": args.author_en or "",
|
||||
"douban_url": args.douban_url or "", "goodreads_url": args.goodreads_url or ""
|
||||
}
|
||||
|
||||
if getattr(args, "json", None):
|
||||
import json
|
||||
try:
|
||||
extra = json.loads(args.json)
|
||||
kwargs.update(extra)
|
||||
except Exception as e:
|
||||
print(f"❌ JSON 解析失败: {e}")
|
||||
return
|
||||
|
||||
title = kwargs.pop("title", args.title)
|
||||
if not title:
|
||||
print("❌ 必须提供书名 (title)。")
|
||||
return
|
||||
|
||||
all_books = db.list_books()
|
||||
for b in all_books:
|
||||
db_title = b['title'].lower()
|
||||
t_lower = title.lower()
|
||||
if t_lower == db_title or (len(t_lower) > 3 and (t_lower in db_title or db_title in t_lower)):
|
||||
a_in = (kwargs.get("author") or "").lower()
|
||||
a_db = (b.get("author") or "").lower()
|
||||
if not a_in or not a_db or a_in in a_db or a_db in a_in:
|
||||
print(f"⚠️ 库中可能已存在同名记录:[ID: {b['id']}] 《{b['title']}》 (作者: {b.get('author', '未知')})。")
|
||||
print(" 已取消添加。若确认并非同一本请修改书名/作者,若需更新请使用 `update` 命令。")
|
||||
return
|
||||
|
||||
book_id = db.add_book(title)
|
||||
updates = {k: v for k, v in kwargs.items() if v}
|
||||
if updates:
|
||||
db.update_book(book_id, **updates)
|
||||
|
||||
print(f"✅ 已添加并录入数据:《{title}》(ID: {book_id})")
|
||||
|
||||
|
||||
def cmd_list(args):
|
||||
@@ -48,27 +75,55 @@ def cmd_list(args):
|
||||
print(f" {status_icon} [{b['id']:>3}] 《{b['title']}》— {b['author'] or '未知'}"
|
||||
f" {fmt_icon}{rating_str}{tags_str}")
|
||||
|
||||
|
||||
def cmd_update(args):
|
||||
db.init_db()
|
||||
book = db.get_book(args.id)
|
||||
if not book:
|
||||
print(f"❌ 未找到 ID 为 {args.id} 的书目。")
|
||||
sys.exit(1)
|
||||
|
||||
updates = {}
|
||||
|
||||
# 获取命令行常规参数
|
||||
for field in ["title", "author", "translator", "publisher", "pub_date",
|
||||
"cover_url", "format", "status", "rating", "douban_score",
|
||||
"goodreads_score", "tags", "notes", "start_date", "finish_date"]:
|
||||
"goodreads_score", "tags", "notes", "start_date", "finish_date",
|
||||
"title_en", "author_en", "douban_url", "goodreads_url"]:
|
||||
val = getattr(args, field.replace("-", "_"), None)
|
||||
if val is not None:
|
||||
updates[field] = val
|
||||
|
||||
# 从 JSON 批量注入参数(AI优先)
|
||||
if getattr(args, "json", None):
|
||||
import json
|
||||
try:
|
||||
updates.update(json.loads(args.json))
|
||||
except Exception as e:
|
||||
print(f"❌ JSON 解析失败: {e}")
|
||||
return
|
||||
|
||||
# 从本地图片注入封面(AI优先)
|
||||
if getattr(args, "cover", None):
|
||||
import os
|
||||
if not os.path.exists(args.cover):
|
||||
print(f"❌ 找不到指定的本封面: {args.cover}")
|
||||
return
|
||||
try:
|
||||
with open(args.cover, "rb") as f:
|
||||
updates["cover_blob"] = f.read()
|
||||
updates["cover_local"] = 1
|
||||
print(f"🖼️ 已应用本地封面: {args.cover}")
|
||||
except Exception as e:
|
||||
print(f"❌ 读取封面文件失败: {e}")
|
||||
return
|
||||
|
||||
if not updates:
|
||||
print("⚠️ 未指定任何要更新的字段。")
|
||||
return
|
||||
|
||||
db.update_book(args.id, **updates)
|
||||
print(f"✅ 已更新:《{book['title']}》(ID: {args.id})")
|
||||
|
||||
|
||||
def cmd_delete(args):
|
||||
db.init_db()
|
||||
book = db.get_book(args.id)
|
||||
@@ -139,12 +194,187 @@ def cmd_fetch(args):
|
||||
douban_score=data.get("douban_score"),
|
||||
goodreads_score=data.get("goodreads_score"),
|
||||
tags=data.get("tags", ""),
|
||||
title_en=data.get("title_en", ""),
|
||||
author_en=data.get("author_en", ""),
|
||||
douban_url=data.get("douban_url", ""),
|
||||
goodreads_url=data.get("goodreads_url", ""),
|
||||
)
|
||||
print(f"✅ 已添加:《{data.get('title')}》(ID: {book_id})")
|
||||
else:
|
||||
print("💡 加 --add 选项可直接写入数据库,或用 --update-id <ID> 更新已有记录。")
|
||||
|
||||
|
||||
def _is_match(b_title, b_author, s_title, s_title_en, s_author):
|
||||
import re
|
||||
def canon(s):
|
||||
return re.sub(r'[^\w\u4e00-\u9fff]', '', str(s).lower()) if s else ""
|
||||
bt = canon(b_title)
|
||||
st = canon(s_title)
|
||||
ste = canon(s_title_en)
|
||||
title_match = (bt in st or st in bt) if (bt and st) else False
|
||||
if not title_match and ste:
|
||||
title_match = (bt in ste or ste in bt) if bt else False
|
||||
ba = canon(b_author)
|
||||
sa = canon(s_author)
|
||||
author_match = (ba in sa or sa in ba) if (ba and sa) else False
|
||||
# 书名匹配或作者匹配之一满足即可认为是同一个记录
|
||||
return title_match or author_match
|
||||
|
||||
|
||||
def cmd_enrich(args):
|
||||
"""自动补全数据库中缺失元数据的书目。"""
|
||||
db.init_db()
|
||||
books = db.list_books()
|
||||
to_enrich = []
|
||||
for b in books:
|
||||
# 只要没有豆瓣或 GR 的链接,或者没有封面,就尝试 enrich
|
||||
if not b.get("cover_url") or not b.get("douban_url") or not b.get("goodreads_url"):
|
||||
to_enrich.append(b)
|
||||
|
||||
if args.id:
|
||||
to_enrich = [b for b in to_enrich if b["id"] == args.id]
|
||||
|
||||
if not to_enrich:
|
||||
print("🎉 所有书目数据已相对完整,无需 enrich。")
|
||||
return
|
||||
|
||||
print(f"🔍 共有 {len(to_enrich)} 本书需补全或验证信息...")
|
||||
import urllib.parse
|
||||
|
||||
for b in to_enrich:
|
||||
print(f"\n=> 正在处理 [{b['id']}]: 《{b['title']}》 作者: {b.get('author','')} ...")
|
||||
|
||||
scraped = None
|
||||
src = ""
|
||||
|
||||
search_term = b['title']
|
||||
if b.get('author'): search_term += f" {b['author']}"
|
||||
|
||||
# 策略 1: 直接向 Goodreads 询问(它对原版图书封面最标准、非促销封套)
|
||||
print(" 🔍 尝试通过 Goodreads 直搜...")
|
||||
gr_data = scrape.fetch_goodreads(search_term)
|
||||
if gr_data and _is_match(b['title'], b.get('author'), gr_data.get('title'), gr_data.get('title_en'), gr_data.get('author')):
|
||||
scraped = gr_data
|
||||
src = "Goodreads"
|
||||
|
||||
# 策略 2: 如果 GR 失败或缺信息,对于繁体中文,尝试博客来
|
||||
if not scraped or not scraped.get("cover_url"):
|
||||
print(" 🔍 尝试通过博客来(Books.com.tw)直搜...")
|
||||
tw_data = scrape.fetch_books_tw(search_term)
|
||||
if tw_data and _is_match(b['title'], b.get('author'), tw_data.get('title'), None, tw_data.get('author')):
|
||||
if not scraped: scraped = tw_data
|
||||
else:
|
||||
for k, v in tw_data.items():
|
||||
if not scraped.get(k): scraped[k] = v
|
||||
scraped["format"] = "paper" # 博客来主要是实体书
|
||||
src = "博客来" + ((" / " + src) if src else "")
|
||||
|
||||
# 策略 3: Google Books 仅作为属性补充,**决不拿它的 ISBN 去瞎关联豆瓣**
|
||||
query = f"intitle:{b['title']}"
|
||||
if b.get('author'): query += f"+inauthor:{b['author']}"
|
||||
q = urllib.parse.quote(query)
|
||||
url = f"https://www.googleapis.com/books/v1/volumes?q={q}"
|
||||
try:
|
||||
html = scrape._get(url)
|
||||
if html:
|
||||
import json
|
||||
data = json.loads(html)
|
||||
if data.get("totalItems", 0) > 0:
|
||||
vol = data["items"][0].get("volumeInfo", {})
|
||||
# 只要匹配度过关,就拿它的 publisher, pub_date, ISBN
|
||||
if _is_match(b['title'], b.get('author'), vol.get("title"), None, ",".join(vol.get("authors", []))):
|
||||
isbn = next((ids.get("identifier") for ids in vol.get("industryIdentifiers", []) if ids.get("type") in ("ISBN_13", "ISBN_10")), None)
|
||||
if not scraped: scraped = {"title": vol.get("title")}
|
||||
if vol.get("publisher") and not scraped.get("publisher"): scraped["publisher"] = vol["publisher"]
|
||||
if vol.get("publishedDate") and not scraped.get("pub_date"): scraped["pub_date"] = vol["publishedDate"][:7]
|
||||
if isbn and not scraped.get("isbn"): scraped["isbn"] = isbn
|
||||
src = src or "Google Books"
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
# 策略 4: 如果有 ISBN,去豆瓣白嫖评分和出版信息(但为了防止封面被覆盖为促销版,不轻易覆写封面)
|
||||
if scraped and scraped.get("isbn"):
|
||||
douban_data = scrape.fetch_douban(scraped["isbn"])
|
||||
if douban_data and _is_match(b['title'], b.get('author'), douban_data.get('title'), douban_data.get('title_en'), douban_data.get('author')):
|
||||
for k, v in douban_data.items():
|
||||
if k == "cover_url" and scraped.get("cover_url"): continue # 优先信任前面外站的原版封面
|
||||
if not scraped.get(k): scraped[k] = v
|
||||
src += " / 豆瓣"
|
||||
|
||||
if scraped:
|
||||
updates = {}
|
||||
# 新刮削的标题如果是外文但不等于原书名,放到 title_en 保护原中文书名
|
||||
if scraped.get("title") and not _is_match(b["title"], None, scraped["title"], None, None):
|
||||
if not b.get("title_en"):
|
||||
updates["title_en"] = scraped["title"]
|
||||
if scraped.get("author") and not _is_match(b["author"], None, scraped["author"], None, None):
|
||||
if not b.get("author_en"):
|
||||
updates["author_en"] = scraped["author"]
|
||||
|
||||
for k in ["author", "translator", "publisher", "pub_date", "cover_url", "tags",
|
||||
"douban_score", "goodreads_score", "isbn", "title_en", "author_en", "douban_url", "goodreads_url"]:
|
||||
if scraped.get(k) and not b.get(k):
|
||||
updates[k] = scraped[k]
|
||||
if updates:
|
||||
db.update_book(b["id"], **updates)
|
||||
print(f" ✅ 已补全字段: {', '.join(updates.keys())} (来源: {src})")
|
||||
else:
|
||||
print(f" 👍 已匹配到信息,但无需补充新字段 (来源: {src})")
|
||||
else:
|
||||
print(" 🚫 彻底失败,保留原样。")
|
||||
|
||||
# 延时避免被反爬
|
||||
scrape._sleep()
|
||||
|
||||
|
||||
def cmd_covers(args):
|
||||
"""批量下载封面存入数据库 BLOB。"""
|
||||
db.init_db()
|
||||
books = db.list_books()
|
||||
to_download = [b for b in books if b.get("cover_url") and not b.get("cover_blob")]
|
||||
|
||||
if args.id:
|
||||
to_download = [b for b in to_download if b["id"] == args.id]
|
||||
|
||||
print(f"🖼️ 共有 {len(to_download)} 本书需要下载封面到数据库...")
|
||||
|
||||
conn = db._connect()
|
||||
try:
|
||||
import requests
|
||||
except ImportError:
|
||||
import urllib.request
|
||||
|
||||
count = 0
|
||||
for b in to_download:
|
||||
url = b["cover_url"]
|
||||
print(f"下载 [{b['id']}] 《{b['title']}》封面: {url}")
|
||||
try:
|
||||
# 豆瓣防盗链需要 Referer,而且可能需要 Playwright 那个 UA
|
||||
headers = {"User-Agent": scrape._UA, "Referer": "https://book.douban.com/"}
|
||||
if scrape._HAS_REQUESTS:
|
||||
resp = requests.get(url, headers=headers, timeout=15)
|
||||
if resp.status_code == 200:
|
||||
blob = resp.content
|
||||
else:
|
||||
blob = None
|
||||
else:
|
||||
req = urllib.request.Request(url, headers=headers)
|
||||
with urllib.request.urlopen(req, timeout=15) as resp:
|
||||
blob = resp.read()
|
||||
|
||||
if blob:
|
||||
conn.execute("UPDATE books SET cover_blob = ? WHERE id = ?", (sqlite3.Binary(blob), b["id"]))
|
||||
conn.commit()
|
||||
print(" ✅ 成功")
|
||||
count += 1
|
||||
else:
|
||||
print(" ❌ 获取失败")
|
||||
scrape._sleep()
|
||||
except Exception as e:
|
||||
print(f" ❌ 网络错误: {e}")
|
||||
|
||||
conn.close()
|
||||
print(f"🎉 封面下载完成,成功 {count} 本。")
|
||||
def cmd_build(args):
|
||||
db.init_db()
|
||||
output_path = web.build()
|
||||
@@ -175,6 +405,10 @@ def main():
|
||||
p_add.add_argument("--notes", help="笔记/简评")
|
||||
p_add.add_argument("--start-date", help="开始阅读日期")
|
||||
p_add.add_argument("--finish-date", help="完成阅读日期")
|
||||
p_add.add_argument("--title-en", help="外文原名")
|
||||
p_add.add_argument("--author-en", help="外文作者名")
|
||||
p_add.add_argument("--douban-url", help="豆瓣链接")
|
||||
p_add.add_argument("--goodreads-url", help="Goodreads 链接")
|
||||
p_add.set_defaults(func=cmd_add)
|
||||
|
||||
# --- list ---
|
||||
@@ -202,6 +436,10 @@ def main():
|
||||
p_upd.add_argument("--notes", help="笔记/简评")
|
||||
p_upd.add_argument("--start-date", help="开始阅读日期")
|
||||
p_upd.add_argument("--finish-date", help="完成阅读日期")
|
||||
p_upd.add_argument("--title-en", help="外文原名")
|
||||
p_upd.add_argument("--author-en", help="外文作者名")
|
||||
p_upd.add_argument("--douban-url", help="豆瓣链接")
|
||||
p_upd.add_argument("--goodreads-url", help="Goodreads 链接")
|
||||
p_upd.set_defaults(func=cmd_update)
|
||||
|
||||
# --- delete ---
|
||||
@@ -218,10 +456,27 @@ def main():
|
||||
p_fetch.add_argument("--status", choices=STATUS_CHOICES, help="阅读状态(入库时使用)")
|
||||
p_fetch.set_defaults(func=cmd_fetch)
|
||||
|
||||
# --- enrich ---
|
||||
p_enrich = sub.add_parser("enrich", help="批量刮削补全缺失信息")
|
||||
p_enrich.add_argument("--id", type=int, help="只补全特定 ID 的书")
|
||||
p_enrich.set_defaults(func=cmd_enrich)
|
||||
|
||||
# --- covers ---
|
||||
p_covers = sub.add_parser("covers", help="批量下载并保存封面到数据库")
|
||||
p_covers.add_argument("--id", type=int, help="只下载特定 ID 的书")
|
||||
p_covers.set_defaults(func=cmd_covers)
|
||||
|
||||
# --- build ---
|
||||
p_build = sub.add_parser("build", help="生成展示网页")
|
||||
p_build.set_defaults(func=cmd_build)
|
||||
|
||||
# --- update ---
|
||||
p_update = sub.add_parser("update", help="(高级/AI专用)通过 JSON 直接更新元数据或本地封面")
|
||||
p_update.add_argument("id", type=int, help="书籍记录的数字 ID")
|
||||
p_update.add_argument("--json", type=str, help="包含需更新字段的 JSON 字符串 (例: '{\"douban_score\": 9.5}')")
|
||||
p_update.add_argument("--cover", type=str, help="本地图片文件路径,直接注入 cover_blob")
|
||||
p_update.set_defaults(func=cmd_update)
|
||||
|
||||
args = parser.parse_args()
|
||||
args.func(args)
|
||||
|
||||
|
||||
Reference in New Issue
Block a user