Files
HeXi/hexi/plugins/nonebot_plugin_video_analysis/fetchers/bilibili_content.py
T

270 lines
9.4 KiB
Python
Raw Normal View History

"""B站动态/图文/文章内容解析 — 基于 bilibili-api-python
借鉴 nonebot-plugin-parser 的 BilibiliParser:
- bilibili.com/opus/{id}、bilibili.com/dynamic/{id}、t.bilibili.com/{id}
→ 动态(图文 / 视频 / 纯文字)
- bilibili.com/read/cv{id} → 专栏(转为图文动态解析)
- b23.tv / bili2233.cn 短链 → 先重定向
返回 (title, 文件):
- 图文动态/专栏 → (title, [图片路径列表])
- 视频动态 → (title, 视频文件 Path,复用 yt-dlp 链路)
- 纯文字动态 → (title, [])
- 解析失败 → (None, None)
"""
import re
import tempfile
from datetime import datetime
from pathlib import Path
from typing import Optional, Union
from nonebot import logger
from ..models import ContentFetchError
from ..utils import parse_netscape_cookies, slugify
DATA_DIR = Path(__file__).resolve().parent.parent / "data"
BILI_UA = (
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 "
"(KHTML, like Gecko) Chrome/122.0.0.0 Safari/537.36"
)
BILI_REFERER = "https://www.bilibili.com/"
OPUS_RE = re.compile(r"bilibili\.com/(?:opus|dynamic)/(\d+)")
T_BILI_RE = re.compile(r"t\.bilibili\.com/(\d+)")
READ_RE = re.compile(r"bilibili\.com/read/cv(\d+)")
BILI_SHORT_RE = re.compile(r"(?:b23\.tv|bili2233\.cn)/[0-9a-zA-Z._?%&+=/#]+")
async def fetch_bilibili_content(
url: str,
) -> tuple[Optional[str], Optional[Union[Path, list[Path]]]]:
"""解析 B站动态/图文/文章链接"""
try:
return await _parse(url)
except Exception:
logger.exception(f"B站内容解析失败: {url}")
return None, None
async def _parse(url: str):
# 1. 短链重定向
if "b23.tv" in url or "bili2233.cn" in url:
resolved = await resolve_short_link(url)
if resolved:
logger.info(f"b23 短链重定向: {url} -> {resolved}")
url = resolved
# 2. 类型与 ID 提取
if m := READ_RE.search(url):
return await _parse_article(int(m.group(1)))
elif m := OPUS_RE.search(url):
dynamic_id = int(m.group(1))
elif m := T_BILI_RE.search(url):
dynamic_id = int(m.group(1))
else:
logger.warning(f"无法识别的 B站链接: {url}")
return None, None
# 3. 延迟导入 bilibili_api(未安装时不阻塞插件启动)
from bilibili_api import request_settings, select_client
# 模拟浏览器指纹,避免被 B站风控限流(-509)
select_client("curl_cffi")
request_settings.set("impersonate", "chrome131")
from bilibili_api.dynamic import Dynamic
dynamic = Dynamic(dynamic_id, _build_credential())
# 4. 文章动态 → 转 opus
if await dynamic.is_article():
return await _parse_opus(dynamic.turn_to_opus(), "文章")
info = await dynamic.get_info()
return await _parse_dynamic_info(info)
async def _parse_article(read_id: int) -> tuple[str, Union[Path, list[Path]]]:
"""专栏 cv{id} → 转为图文动态"""
from bilibili_api.article import Article
# 文章接口对匿名请求风控更严(-509),必须带凭证
article = Article(read_id, _build_credential())
opus = await article.turn_to_opus()
return await _parse_opus(opus, "文章")
async def _parse_opus(opus, kind: str) -> tuple[str, Union[Path, list[Path]]]:
"""图文动态/专栏解析(opus 接口返回 dict,直接访问)"""
info = await opus.get_info()
item = info.get("item") or {}
basic = item.get("basic") or {}
title = basic.get("title") or ""
images: list[str] = []
texts: list[str] = []
author = ""
for module in item.get("modules") or []:
if module.get("module_type") == "MODULE_TYPE_AUTHOR":
author_info = module.get("module_author") or {}
author = author_info.get("name", "")
elif module.get("module_type") == "MODULE_TYPE_CONTENT":
content = module.get("module_content") or {}
for para in content.get("paragraphs") or []:
if pic := (para.get("pic") or {}).get("pics"):
images.extend(p.get("url", "") for p in pic)
elif nodes := ((para.get("text") or {}).get("nodes")):
if text := _extract_text(nodes):
texts.append(text)
images = [u for u in images if u]
text = title or (texts[0] if texts else "")
logger.info(f"B站{kind}解析: 作者={author}, 标题={text[:40]}, 图片={len(images)} 张")
if not images:
return text or f"B站{kind}", []
file_name = _build_file_name(author, text or f"B站{kind}", kind)
file_paths = await _download_images(images, file_name)
return text, file_paths
async def _parse_dynamic_info(info: dict) -> tuple[str, Union[Path, list[Path]]]:
"""动态解析(图文 / 视频 / 纯文字)"""
item = info.get("item") or {}
modules = item.get("modules") or {}
author = ((modules.get("module_author") or {}).get("name")) or "B站用户"
module_dynamic = modules.get("module_dynamic") or {}
major = module_dynamic.get("major") or {}
major_type = major.get("type", "")
desc = ((module_dynamic.get("desc") or {}).get("text")) or ""
# 1. 视频动态 → 提取 bvid 走 yt-dlp
if major_type == "MAJOR_TYPE_ARCHIVE":
archive = major.get("archive") or {}
bvid = archive.get("bvid")
if bvid:
from .video_downloader import download_video
title = archive.get("title") or desc or "B站视频动态"
logger.info(f"B站视频动态: bvid={bvid} 标题={title[:40]}")
video_path = await download_video(f"https://www.bilibili.com/video/{bvid}")
if video_path:
return title, video_path
raise ContentFetchError(f"视频动态下载失败: {bvid}")
return desc or "B站视频动态", []
# 2. 图文动态(新版 opus / 老版 draw)
images: list[str] = []
title = author
if major_type == "MAJOR_TYPE_OPUS":
opus = major.get("opus") or {}
images = [pic.get("url", "") for pic in opus.get("pics") or []]
desc = desc or ((opus.get("summary") or {}).get("text")) or ""
title = opus.get("title") or title
elif major_type == "MAJOR_TYPE_DRAW":
images = [
item.get("src", "")
for item in (major.get("draw") or {}).get("items") or []
]
images = [u for u in images if u]
if images:
file_name = _build_file_name(author, title, "动态")
file_paths = await _download_images(images, file_name)
logger.info(f"B站图文动态: 作者={author}, 标题={title[:40]}, 图片={len(images)} 张")
return title, file_paths
# 3. 纯文字动态
text = desc or "B站动态"
logger.info(f"B站文字动态: {text[:30]}")
return text, []
def _extract_text(nodes: list) -> str:
"""从文章段落节点提取文字"""
parts = []
for node in nodes or []:
word = node.get("word") or {}
if node.get("type") in (
"TEXT_NODE_TYPE_WORD",
"TEXT_NODE_TYPE_RICH",
) and word.get("words"):
parts.append(word["words"])
return "".join(parts)
def _build_file_name(author: str, title: str, kind: str) -> str:
"""构建文件名 stem: {作者}_{标题}_{类型}_{时间}"""
slug_author = slugify(author)
slug_title = slugify(title or "", max_length=15)
if not slug_title:
slug_title = datetime.now().strftime("%H%M%S")
time_suffix = datetime.now().strftime("%H%M%S")
return f"{slug_author}_{slug_title}_{kind}_{time_suffix}"
def _build_credential():
"""从 cookies.txt 构建 B站凭证(无 SESSDATA 时返回 None 匿名访问)"""
from bilibili_api import Credential
cookies_path = DATA_DIR / "cookies.txt"
if not cookies_path.exists():
return None
cookies = parse_netscape_cookies(str(cookies_path))
ck = {c["name"]: c["value"] for c in cookies}
sessdata = ck.get("SESSDATA")
if not sessdata:
logger.warning("cookies.txt 中无 SESSDATA,B站将以匿名身份访问")
return None
return Credential(
sessdata=sessdata,
bili_jct=ck.get("bili_jct", ""),
buvid3=ck.get("buvid3", ""),
)
async def _download_images(image_urls: list[str], file_name: str) -> list[Path]:
"""并发下载图片(复用抖音图文的下载流程)"""
import httpx
from .douyin_api import _process_note_with_parsed
tmp_root = Path(tempfile.gettempdir()) / "bilibili"
tmp_root.mkdir(parents=True, exist_ok=True)
headers = {
"Referer": BILI_REFERER,
"User-Agent": BILI_UA,
}
return await _process_note_with_parsed(
[[u] for u in image_urls], None, tmp_root, file_name, headers
)
async def resolve_short_link(url: str) -> Optional[str]:
"""b23 短链重定向(httpx 取 Location,最多 3 跳)"""
import httpx
try:
async with httpx.AsyncClient(
headers={"User-Agent": BILI_UA},
follow_redirects=False,
timeout=10,
) as client:
current = url
for _ in range(3):
resp = await client.get(current)
if resp.status_code >= 400:
return None
location = resp.headers.get("Location")
if not location:
return str(resp.url)
current = location
return current
except Exception:
logger.warning(f"b23 短链重定向失败: {url}")
return None