- hexi core: message handling, rate limiting, cooldown, plugin manager - Custom plugins: BF stats, daily check-in, quotes, persona cards, etc. - Community plugins vendored under hexi/plugins with local fixes - Web admin frontends (learning-chat, persona-admin), unified hexi/web - Tests for rate_limit/cooldown/memes/persona; poetry.lock Co-Authored-By: Claude <noreply@anthropic.com>
270 lines
9.4 KiB
Python
270 lines
9.4 KiB
Python
"""B站动态/图文/文章内容解析 — 基于 bilibili-api-python
|
||
|
||
借鉴 nonebot-plugin-parser 的 BilibiliParser:
|
||
- bilibili.com/opus/{id}、bilibili.com/dynamic/{id}、t.bilibili.com/{id}
|
||
→ 动态(图文 / 视频 / 纯文字)
|
||
- bilibili.com/read/cv{id} → 专栏(转为图文动态解析)
|
||
- b23.tv / bili2233.cn 短链 → 先重定向
|
||
|
||
返回 (title, 文件):
|
||
- 图文动态/专栏 → (title, [图片路径列表])
|
||
- 视频动态 → (title, 视频文件 Path,复用 yt-dlp 链路)
|
||
- 纯文字动态 → (title, [])
|
||
- 解析失败 → (None, None)
|
||
"""
|
||
|
||
import re
|
||
import tempfile
|
||
from datetime import datetime
|
||
from pathlib import Path
|
||
from typing import Optional, Union
|
||
|
||
from nonebot import logger
|
||
|
||
from ..models import ContentFetchError
|
||
from ..utils import parse_netscape_cookies, slugify
|
||
|
||
DATA_DIR = Path(__file__).resolve().parent.parent / "data"
|
||
|
||
BILI_UA = (
|
||
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 "
|
||
"(KHTML, like Gecko) Chrome/122.0.0.0 Safari/537.36"
|
||
)
|
||
BILI_REFERER = "https://www.bilibili.com/"
|
||
|
||
OPUS_RE = re.compile(r"bilibili\.com/(?:opus|dynamic)/(\d+)")
|
||
T_BILI_RE = re.compile(r"t\.bilibili\.com/(\d+)")
|
||
READ_RE = re.compile(r"bilibili\.com/read/cv(\d+)")
|
||
BILI_SHORT_RE = re.compile(r"(?:b23\.tv|bili2233\.cn)/[0-9a-zA-Z._?%&+=/#]+")
|
||
|
||
|
||
async def fetch_bilibili_content(
|
||
url: str,
|
||
) -> tuple[Optional[str], Optional[Union[Path, list[Path]]]]:
|
||
"""解析 B站动态/图文/文章链接"""
|
||
try:
|
||
return await _parse(url)
|
||
except Exception:
|
||
logger.exception(f"B站内容解析失败: {url}")
|
||
return None, None
|
||
|
||
|
||
async def _parse(url: str):
|
||
# 1. 短链重定向
|
||
if "b23.tv" in url or "bili2233.cn" in url:
|
||
resolved = await resolve_short_link(url)
|
||
if resolved:
|
||
logger.info(f"b23 短链重定向: {url} -> {resolved}")
|
||
url = resolved
|
||
|
||
# 2. 类型与 ID 提取
|
||
if m := READ_RE.search(url):
|
||
return await _parse_article(int(m.group(1)))
|
||
elif m := OPUS_RE.search(url):
|
||
dynamic_id = int(m.group(1))
|
||
elif m := T_BILI_RE.search(url):
|
||
dynamic_id = int(m.group(1))
|
||
else:
|
||
logger.warning(f"无法识别的 B站链接: {url}")
|
||
return None, None
|
||
|
||
# 3. 延迟导入 bilibili_api(未安装时不阻塞插件启动)
|
||
from bilibili_api import request_settings, select_client
|
||
|
||
# 模拟浏览器指纹,避免被 B站风控限流(-509)
|
||
select_client("curl_cffi")
|
||
request_settings.set("impersonate", "chrome131")
|
||
|
||
from bilibili_api.dynamic import Dynamic
|
||
|
||
dynamic = Dynamic(dynamic_id, _build_credential())
|
||
|
||
# 4. 文章动态 → 转 opus
|
||
if await dynamic.is_article():
|
||
return await _parse_opus(dynamic.turn_to_opus(), "文章")
|
||
|
||
info = await dynamic.get_info()
|
||
return await _parse_dynamic_info(info)
|
||
|
||
|
||
async def _parse_article(read_id: int) -> tuple[str, Union[Path, list[Path]]]:
|
||
"""专栏 cv{id} → 转为图文动态"""
|
||
from bilibili_api.article import Article
|
||
|
||
# 文章接口对匿名请求风控更严(-509),必须带凭证
|
||
article = Article(read_id, _build_credential())
|
||
opus = await article.turn_to_opus()
|
||
return await _parse_opus(opus, "文章")
|
||
|
||
|
||
async def _parse_opus(opus, kind: str) -> tuple[str, Union[Path, list[Path]]]:
|
||
"""图文动态/专栏解析(opus 接口返回 dict,直接访问)"""
|
||
info = await opus.get_info()
|
||
item = info.get("item") or {}
|
||
basic = item.get("basic") or {}
|
||
title = basic.get("title") or ""
|
||
|
||
images: list[str] = []
|
||
texts: list[str] = []
|
||
author = ""
|
||
for module in item.get("modules") or []:
|
||
if module.get("module_type") == "MODULE_TYPE_AUTHOR":
|
||
author_info = module.get("module_author") or {}
|
||
author = author_info.get("name", "")
|
||
elif module.get("module_type") == "MODULE_TYPE_CONTENT":
|
||
content = module.get("module_content") or {}
|
||
for para in content.get("paragraphs") or []:
|
||
if pic := (para.get("pic") or {}).get("pics"):
|
||
images.extend(p.get("url", "") for p in pic)
|
||
elif nodes := ((para.get("text") or {}).get("nodes")):
|
||
if text := _extract_text(nodes):
|
||
texts.append(text)
|
||
|
||
images = [u for u in images if u]
|
||
text = title or (texts[0] if texts else "")
|
||
logger.info(f"B站{kind}解析: 作者={author}, 标题={text[:40]}, 图片={len(images)} 张")
|
||
|
||
if not images:
|
||
return text or f"B站{kind}", []
|
||
|
||
file_name = _build_file_name(author, text or f"B站{kind}", kind)
|
||
file_paths = await _download_images(images, file_name)
|
||
return text, file_paths
|
||
|
||
|
||
async def _parse_dynamic_info(info: dict) -> tuple[str, Union[Path, list[Path]]]:
|
||
"""动态解析(图文 / 视频 / 纯文字)"""
|
||
item = info.get("item") or {}
|
||
modules = item.get("modules") or {}
|
||
author = ((modules.get("module_author") or {}).get("name")) or "B站用户"
|
||
module_dynamic = modules.get("module_dynamic") or {}
|
||
major = module_dynamic.get("major") or {}
|
||
major_type = major.get("type", "")
|
||
desc = ((module_dynamic.get("desc") or {}).get("text")) or ""
|
||
|
||
# 1. 视频动态 → 提取 bvid 走 yt-dlp
|
||
if major_type == "MAJOR_TYPE_ARCHIVE":
|
||
archive = major.get("archive") or {}
|
||
bvid = archive.get("bvid")
|
||
if bvid:
|
||
from .video_downloader import download_video
|
||
|
||
title = archive.get("title") or desc or "B站视频动态"
|
||
logger.info(f"B站视频动态: bvid={bvid} 标题={title[:40]}")
|
||
video_path = await download_video(f"https://www.bilibili.com/video/{bvid}")
|
||
if video_path:
|
||
return title, video_path
|
||
raise ContentFetchError(f"视频动态下载失败: {bvid}")
|
||
return desc or "B站视频动态", []
|
||
|
||
# 2. 图文动态(新版 opus / 老版 draw)
|
||
images: list[str] = []
|
||
title = author
|
||
if major_type == "MAJOR_TYPE_OPUS":
|
||
opus = major.get("opus") or {}
|
||
images = [pic.get("url", "") for pic in opus.get("pics") or []]
|
||
desc = desc or ((opus.get("summary") or {}).get("text")) or ""
|
||
title = opus.get("title") or title
|
||
elif major_type == "MAJOR_TYPE_DRAW":
|
||
images = [
|
||
item.get("src", "")
|
||
for item in (major.get("draw") or {}).get("items") or []
|
||
]
|
||
|
||
images = [u for u in images if u]
|
||
if images:
|
||
file_name = _build_file_name(author, title, "动态")
|
||
file_paths = await _download_images(images, file_name)
|
||
logger.info(f"B站图文动态: 作者={author}, 标题={title[:40]}, 图片={len(images)} 张")
|
||
return title, file_paths
|
||
|
||
# 3. 纯文字动态
|
||
text = desc or "B站动态"
|
||
logger.info(f"B站文字动态: {text[:30]}")
|
||
return text, []
|
||
|
||
|
||
def _extract_text(nodes: list) -> str:
|
||
"""从文章段落节点提取文字"""
|
||
parts = []
|
||
for node in nodes or []:
|
||
word = node.get("word") or {}
|
||
if node.get("type") in (
|
||
"TEXT_NODE_TYPE_WORD",
|
||
"TEXT_NODE_TYPE_RICH",
|
||
) and word.get("words"):
|
||
parts.append(word["words"])
|
||
return "".join(parts)
|
||
|
||
|
||
def _build_file_name(author: str, title: str, kind: str) -> str:
|
||
"""构建文件名 stem: {作者}_{标题}_{类型}_{时间}"""
|
||
slug_author = slugify(author)
|
||
slug_title = slugify(title or "", max_length=15)
|
||
if not slug_title:
|
||
slug_title = datetime.now().strftime("%H%M%S")
|
||
time_suffix = datetime.now().strftime("%H%M%S")
|
||
return f"{slug_author}_{slug_title}_{kind}_{time_suffix}"
|
||
|
||
|
||
def _build_credential():
|
||
"""从 cookies.txt 构建 B站凭证(无 SESSDATA 时返回 None 匿名访问)"""
|
||
from bilibili_api import Credential
|
||
|
||
cookies_path = DATA_DIR / "cookies.txt"
|
||
if not cookies_path.exists():
|
||
return None
|
||
cookies = parse_netscape_cookies(str(cookies_path))
|
||
ck = {c["name"]: c["value"] for c in cookies}
|
||
sessdata = ck.get("SESSDATA")
|
||
if not sessdata:
|
||
logger.warning("cookies.txt 中无 SESSDATA,B站将以匿名身份访问")
|
||
return None
|
||
return Credential(
|
||
sessdata=sessdata,
|
||
bili_jct=ck.get("bili_jct", ""),
|
||
buvid3=ck.get("buvid3", ""),
|
||
)
|
||
|
||
|
||
async def _download_images(image_urls: list[str], file_name: str) -> list[Path]:
|
||
"""并发下载图片(复用抖音图文的下载流程)"""
|
||
import httpx
|
||
|
||
from .douyin_api import _process_note_with_parsed
|
||
|
||
tmp_root = Path(tempfile.gettempdir()) / "bilibili"
|
||
tmp_root.mkdir(parents=True, exist_ok=True)
|
||
headers = {
|
||
"Referer": BILI_REFERER,
|
||
"User-Agent": BILI_UA,
|
||
}
|
||
return await _process_note_with_parsed(
|
||
[[u] for u in image_urls], None, tmp_root, file_name, headers
|
||
)
|
||
|
||
|
||
async def resolve_short_link(url: str) -> Optional[str]:
|
||
"""b23 短链重定向(httpx 取 Location,最多 3 跳)"""
|
||
import httpx
|
||
|
||
try:
|
||
async with httpx.AsyncClient(
|
||
headers={"User-Agent": BILI_UA},
|
||
follow_redirects=False,
|
||
timeout=10,
|
||
) as client:
|
||
current = url
|
||
for _ in range(3):
|
||
resp = await client.get(current)
|
||
if resp.status_code >= 400:
|
||
return None
|
||
location = resp.headers.get("Location")
|
||
if not location:
|
||
return str(resp.url)
|
||
current = location
|
||
return current
|
||
except Exception:
|
||
logger.warning(f"b23 短链重定向失败: {url}")
|
||
return None
|