Files
HeXi/hexi/plugins/nonebot_plugin_video_analysis/fetchers/bilibili_content.py
T
sansenhoshiandClaude b61d09f09f Add HeXi bot codebase: custom plugins, web frontends, tests
- hexi core: message handling, rate limiting, cooldown, plugin manager
- Custom plugins: BF stats, daily check-in, quotes, persona cards, etc.
- Community plugins vendored under hexi/plugins with local fixes
- Web admin frontends (learning-chat, persona-admin), unified hexi/web
- Tests for rate_limit/cooldown/memes/persona; poetry.lock

Co-Authored-By: Claude <noreply@anthropic.com>
2026-09-01 13:13:40 +08:00

270 lines
9.4 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""B站动态/图文/文章内容解析 — 基于 bilibili-api-python
借鉴 nonebot-plugin-parser 的 BilibiliParser:
- bilibili.com/opus/{id}、bilibili.com/dynamic/{id}、t.bilibili.com/{id}
→ 动态(图文 / 视频 / 纯文字)
- bilibili.com/read/cv{id} → 专栏(转为图文动态解析)
- b23.tv / bili2233.cn 短链 → 先重定向
返回 (title, 文件):
- 图文动态/专栏 → (title, [图片路径列表])
- 视频动态 → (title, 视频文件 Path,复用 yt-dlp 链路)
- 纯文字动态 → (title, [])
- 解析失败 → (None, None)
"""
import re
import tempfile
from datetime import datetime
from pathlib import Path
from typing import Optional, Union
from nonebot import logger
from ..models import ContentFetchError
from ..utils import parse_netscape_cookies, slugify
DATA_DIR = Path(__file__).resolve().parent.parent / "data"
BILI_UA = (
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 "
"(KHTML, like Gecko) Chrome/122.0.0.0 Safari/537.36"
)
BILI_REFERER = "https://www.bilibili.com/"
OPUS_RE = re.compile(r"bilibili\.com/(?:opus|dynamic)/(\d+)")
T_BILI_RE = re.compile(r"t\.bilibili\.com/(\d+)")
READ_RE = re.compile(r"bilibili\.com/read/cv(\d+)")
BILI_SHORT_RE = re.compile(r"(?:b23\.tv|bili2233\.cn)/[0-9a-zA-Z._?%&+=/#]+")
async def fetch_bilibili_content(
url: str,
) -> tuple[Optional[str], Optional[Union[Path, list[Path]]]]:
"""解析 B站动态/图文/文章链接"""
try:
return await _parse(url)
except Exception:
logger.exception(f"B站内容解析失败: {url}")
return None, None
async def _parse(url: str):
# 1. 短链重定向
if "b23.tv" in url or "bili2233.cn" in url:
resolved = await resolve_short_link(url)
if resolved:
logger.info(f"b23 短链重定向: {url} -> {resolved}")
url = resolved
# 2. 类型与 ID 提取
if m := READ_RE.search(url):
return await _parse_article(int(m.group(1)))
elif m := OPUS_RE.search(url):
dynamic_id = int(m.group(1))
elif m := T_BILI_RE.search(url):
dynamic_id = int(m.group(1))
else:
logger.warning(f"无法识别的 B站链接: {url}")
return None, None
# 3. 延迟导入 bilibili_api(未安装时不阻塞插件启动)
from bilibili_api import request_settings, select_client
# 模拟浏览器指纹,避免被 B站风控限流(-509)
select_client("curl_cffi")
request_settings.set("impersonate", "chrome131")
from bilibili_api.dynamic import Dynamic
dynamic = Dynamic(dynamic_id, _build_credential())
# 4. 文章动态 → 转 opus
if await dynamic.is_article():
return await _parse_opus(dynamic.turn_to_opus(), "文章")
info = await dynamic.get_info()
return await _parse_dynamic_info(info)
async def _parse_article(read_id: int) -> tuple[str, Union[Path, list[Path]]]:
"""专栏 cv{id} → 转为图文动态"""
from bilibili_api.article import Article
# 文章接口对匿名请求风控更严(-509),必须带凭证
article = Article(read_id, _build_credential())
opus = await article.turn_to_opus()
return await _parse_opus(opus, "文章")
async def _parse_opus(opus, kind: str) -> tuple[str, Union[Path, list[Path]]]:
"""图文动态/专栏解析(opus 接口返回 dict,直接访问)"""
info = await opus.get_info()
item = info.get("item") or {}
basic = item.get("basic") or {}
title = basic.get("title") or ""
images: list[str] = []
texts: list[str] = []
author = ""
for module in item.get("modules") or []:
if module.get("module_type") == "MODULE_TYPE_AUTHOR":
author_info = module.get("module_author") or {}
author = author_info.get("name", "")
elif module.get("module_type") == "MODULE_TYPE_CONTENT":
content = module.get("module_content") or {}
for para in content.get("paragraphs") or []:
if pic := (para.get("pic") or {}).get("pics"):
images.extend(p.get("url", "") for p in pic)
elif nodes := ((para.get("text") or {}).get("nodes")):
if text := _extract_text(nodes):
texts.append(text)
images = [u for u in images if u]
text = title or (texts[0] if texts else "")
logger.info(f"B站{kind}解析: 作者={author}, 标题={text[:40]}, 图片={len(images)} 张")
if not images:
return text or f"B站{kind}", []
file_name = _build_file_name(author, text or f"B站{kind}", kind)
file_paths = await _download_images(images, file_name)
return text, file_paths
async def _parse_dynamic_info(info: dict) -> tuple[str, Union[Path, list[Path]]]:
"""动态解析(图文 / 视频 / 纯文字)"""
item = info.get("item") or {}
modules = item.get("modules") or {}
author = ((modules.get("module_author") or {}).get("name")) or "B站用户"
module_dynamic = modules.get("module_dynamic") or {}
major = module_dynamic.get("major") or {}
major_type = major.get("type", "")
desc = ((module_dynamic.get("desc") or {}).get("text")) or ""
# 1. 视频动态 → 提取 bvid 走 yt-dlp
if major_type == "MAJOR_TYPE_ARCHIVE":
archive = major.get("archive") or {}
bvid = archive.get("bvid")
if bvid:
from .video_downloader import download_video
title = archive.get("title") or desc or "B站视频动态"
logger.info(f"B站视频动态: bvid={bvid} 标题={title[:40]}")
video_path = await download_video(f"https://www.bilibili.com/video/{bvid}")
if video_path:
return title, video_path
raise ContentFetchError(f"视频动态下载失败: {bvid}")
return desc or "B站视频动态", []
# 2. 图文动态(新版 opus / 老版 draw)
images: list[str] = []
title = author
if major_type == "MAJOR_TYPE_OPUS":
opus = major.get("opus") or {}
images = [pic.get("url", "") for pic in opus.get("pics") or []]
desc = desc or ((opus.get("summary") or {}).get("text")) or ""
title = opus.get("title") or title
elif major_type == "MAJOR_TYPE_DRAW":
images = [
item.get("src", "")
for item in (major.get("draw") or {}).get("items") or []
]
images = [u for u in images if u]
if images:
file_name = _build_file_name(author, title, "动态")
file_paths = await _download_images(images, file_name)
logger.info(f"B站图文动态: 作者={author}, 标题={title[:40]}, 图片={len(images)} 张")
return title, file_paths
# 3. 纯文字动态
text = desc or "B站动态"
logger.info(f"B站文字动态: {text[:30]}")
return text, []
def _extract_text(nodes: list) -> str:
"""从文章段落节点提取文字"""
parts = []
for node in nodes or []:
word = node.get("word") or {}
if node.get("type") in (
"TEXT_NODE_TYPE_WORD",
"TEXT_NODE_TYPE_RICH",
) and word.get("words"):
parts.append(word["words"])
return "".join(parts)
def _build_file_name(author: str, title: str, kind: str) -> str:
"""构建文件名 stem: {作者}_{标题}_{类型}_{时间}"""
slug_author = slugify(author)
slug_title = slugify(title or "", max_length=15)
if not slug_title:
slug_title = datetime.now().strftime("%H%M%S")
time_suffix = datetime.now().strftime("%H%M%S")
return f"{slug_author}_{slug_title}_{kind}_{time_suffix}"
def _build_credential():
"""从 cookies.txt 构建 B站凭证(无 SESSDATA 时返回 None 匿名访问)"""
from bilibili_api import Credential
cookies_path = DATA_DIR / "cookies.txt"
if not cookies_path.exists():
return None
cookies = parse_netscape_cookies(str(cookies_path))
ck = {c["name"]: c["value"] for c in cookies}
sessdata = ck.get("SESSDATA")
if not sessdata:
logger.warning("cookies.txt 中无 SESSDATA,B站将以匿名身份访问")
return None
return Credential(
sessdata=sessdata,
bili_jct=ck.get("bili_jct", ""),
buvid3=ck.get("buvid3", ""),
)
async def _download_images(image_urls: list[str], file_name: str) -> list[Path]:
"""并发下载图片(复用抖音图文的下载流程)"""
import httpx
from .douyin_api import _process_note_with_parsed
tmp_root = Path(tempfile.gettempdir()) / "bilibili"
tmp_root.mkdir(parents=True, exist_ok=True)
headers = {
"Referer": BILI_REFERER,
"User-Agent": BILI_UA,
}
return await _process_note_with_parsed(
[[u] for u in image_urls], None, tmp_root, file_name, headers
)
async def resolve_short_link(url: str) -> Optional[str]:
"""b23 短链重定向(httpx 取 Location,最多 3 跳)"""
import httpx
try:
async with httpx.AsyncClient(
headers={"User-Agent": BILI_UA},
follow_redirects=False,
timeout=10,
) as client:
current = url
for _ in range(3):
resp = await client.get(current)
if resp.status_code >= 400:
return None
location = resp.headers.get("Location")
if not location:
return str(resp.url)
current = location
return current
except Exception:
logger.warning(f"b23 短链重定向失败: {url}")
return None