"""B站动态/图文/文章内容解析 — 基于 bilibili-api-python 借鉴 nonebot-plugin-parser 的 BilibiliParser: - bilibili.com/opus/{id}、bilibili.com/dynamic/{id}、t.bilibili.com/{id} → 动态(图文 / 视频 / 纯文字) - bilibili.com/read/cv{id} → 专栏(转为图文动态解析) - b23.tv / bili2233.cn 短链 → 先重定向 返回 (title, 文件): - 图文动态/专栏 → (title, [图片路径列表]) - 视频动态 → (title, 视频文件 Path,复用 yt-dlp 链路) - 纯文字动态 → (title, []) - 解析失败 → (None, None) """ import re import tempfile from datetime import datetime from pathlib import Path from typing import Optional, Union from nonebot import logger from ..models import ContentFetchError from ..utils import parse_netscape_cookies, slugify DATA_DIR = Path(__file__).resolve().parent.parent / "data" BILI_UA = ( "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 " "(KHTML, like Gecko) Chrome/122.0.0.0 Safari/537.36" ) BILI_REFERER = "https://www.bilibili.com/" OPUS_RE = re.compile(r"bilibili\.com/(?:opus|dynamic)/(\d+)") T_BILI_RE = re.compile(r"t\.bilibili\.com/(\d+)") READ_RE = re.compile(r"bilibili\.com/read/cv(\d+)") BILI_SHORT_RE = re.compile(r"(?:b23\.tv|bili2233\.cn)/[0-9a-zA-Z._?%&+=/#]+") async def fetch_bilibili_content( url: str, ) -> tuple[Optional[str], Optional[Union[Path, list[Path]]]]: """解析 B站动态/图文/文章链接""" try: return await _parse(url) except Exception: logger.exception(f"B站内容解析失败: {url}") return None, None async def _parse(url: str): # 1. 短链重定向 if "b23.tv" in url or "bili2233.cn" in url: resolved = await resolve_short_link(url) if resolved: logger.info(f"b23 短链重定向: {url} -> {resolved}") url = resolved # 2. 类型与 ID 提取 if m := READ_RE.search(url): return await _parse_article(int(m.group(1))) elif m := OPUS_RE.search(url): dynamic_id = int(m.group(1)) elif m := T_BILI_RE.search(url): dynamic_id = int(m.group(1)) else: logger.warning(f"无法识别的 B站链接: {url}") return None, None # 3. 延迟导入 bilibili_api(未安装时不阻塞插件启动) from bilibili_api import request_settings, select_client # 模拟浏览器指纹,避免被 B站风控限流(-509) select_client("curl_cffi") request_settings.set("impersonate", "chrome131") from bilibili_api.dynamic import Dynamic dynamic = Dynamic(dynamic_id, _build_credential()) # 4. 文章动态 → 转 opus if await dynamic.is_article(): return await _parse_opus(dynamic.turn_to_opus(), "文章") info = await dynamic.get_info() return await _parse_dynamic_info(info) async def _parse_article(read_id: int) -> tuple[str, Union[Path, list[Path]]]: """专栏 cv{id} → 转为图文动态""" from bilibili_api.article import Article # 文章接口对匿名请求风控更严(-509),必须带凭证 article = Article(read_id, _build_credential()) opus = await article.turn_to_opus() return await _parse_opus(opus, "文章") async def _parse_opus(opus, kind: str) -> tuple[str, Union[Path, list[Path]]]: """图文动态/专栏解析(opus 接口返回 dict,直接访问)""" info = await opus.get_info() item = info.get("item") or {} basic = item.get("basic") or {} title = basic.get("title") or "" images: list[str] = [] texts: list[str] = [] author = "" for module in item.get("modules") or []: if module.get("module_type") == "MODULE_TYPE_AUTHOR": author_info = module.get("module_author") or {} author = author_info.get("name", "") elif module.get("module_type") == "MODULE_TYPE_CONTENT": content = module.get("module_content") or {} for para in content.get("paragraphs") or []: if pic := (para.get("pic") or {}).get("pics"): images.extend(p.get("url", "") for p in pic) elif nodes := ((para.get("text") or {}).get("nodes")): if text := _extract_text(nodes): texts.append(text) images = [u for u in images if u] text = title or (texts[0] if texts else "") logger.info(f"B站{kind}解析: 作者={author}, 标题={text[:40]}, 图片={len(images)} 张") if not images: return text or f"B站{kind}", [] file_name = _build_file_name(author, text or f"B站{kind}", kind) file_paths = await _download_images(images, file_name) return text, file_paths async def _parse_dynamic_info(info: dict) -> tuple[str, Union[Path, list[Path]]]: """动态解析(图文 / 视频 / 纯文字)""" item = info.get("item") or {} modules = item.get("modules") or {} author = ((modules.get("module_author") or {}).get("name")) or "B站用户" module_dynamic = modules.get("module_dynamic") or {} major = module_dynamic.get("major") or {} major_type = major.get("type", "") desc = ((module_dynamic.get("desc") or {}).get("text")) or "" # 1. 视频动态 → 提取 bvid 走 yt-dlp if major_type == "MAJOR_TYPE_ARCHIVE": archive = major.get("archive") or {} bvid = archive.get("bvid") if bvid: from .video_downloader import download_video title = archive.get("title") or desc or "B站视频动态" logger.info(f"B站视频动态: bvid={bvid} 标题={title[:40]}") video_path = await download_video(f"https://www.bilibili.com/video/{bvid}") if video_path: return title, video_path raise ContentFetchError(f"视频动态下载失败: {bvid}") return desc or "B站视频动态", [] # 2. 图文动态(新版 opus / 老版 draw) images: list[str] = [] title = author if major_type == "MAJOR_TYPE_OPUS": opus = major.get("opus") or {} images = [pic.get("url", "") for pic in opus.get("pics") or []] desc = desc or ((opus.get("summary") or {}).get("text")) or "" title = opus.get("title") or title elif major_type == "MAJOR_TYPE_DRAW": images = [ item.get("src", "") for item in (major.get("draw") or {}).get("items") or [] ] images = [u for u in images if u] if images: file_name = _build_file_name(author, title, "动态") file_paths = await _download_images(images, file_name) logger.info(f"B站图文动态: 作者={author}, 标题={title[:40]}, 图片={len(images)} 张") return title, file_paths # 3. 纯文字动态 text = desc or "B站动态" logger.info(f"B站文字动态: {text[:30]}") return text, [] def _extract_text(nodes: list) -> str: """从文章段落节点提取文字""" parts = [] for node in nodes or []: word = node.get("word") or {} if node.get("type") in ( "TEXT_NODE_TYPE_WORD", "TEXT_NODE_TYPE_RICH", ) and word.get("words"): parts.append(word["words"]) return "".join(parts) def _build_file_name(author: str, title: str, kind: str) -> str: """构建文件名 stem: {作者}_{标题}_{类型}_{时间}""" slug_author = slugify(author) slug_title = slugify(title or "", max_length=15) if not slug_title: slug_title = datetime.now().strftime("%H%M%S") time_suffix = datetime.now().strftime("%H%M%S") return f"{slug_author}_{slug_title}_{kind}_{time_suffix}" def _build_credential(): """从 cookies.txt 构建 B站凭证(无 SESSDATA 时返回 None 匿名访问)""" from bilibili_api import Credential cookies_path = DATA_DIR / "cookies.txt" if not cookies_path.exists(): return None cookies = parse_netscape_cookies(str(cookies_path)) ck = {c["name"]: c["value"] for c in cookies} sessdata = ck.get("SESSDATA") if not sessdata: logger.warning("cookies.txt 中无 SESSDATA,B站将以匿名身份访问") return None return Credential( sessdata=sessdata, bili_jct=ck.get("bili_jct", ""), buvid3=ck.get("buvid3", ""), ) async def _download_images(image_urls: list[str], file_name: str) -> list[Path]: """并发下载图片(复用抖音图文的下载流程)""" import httpx from .douyin_api import _process_note_with_parsed tmp_root = Path(tempfile.gettempdir()) / "bilibili" tmp_root.mkdir(parents=True, exist_ok=True) headers = { "Referer": BILI_REFERER, "User-Agent": BILI_UA, } return await _process_note_with_parsed( [[u] for u in image_urls], None, tmp_root, file_name, headers ) async def resolve_short_link(url: str) -> Optional[str]: """b23 短链重定向(httpx 取 Location,最多 3 跳)""" import httpx try: async with httpx.AsyncClient( headers={"User-Agent": BILI_UA}, follow_redirects=False, timeout=10, ) as client: current = url for _ in range(3): resp = await client.get(current) if resp.status_code >= 400: return None location = resp.headers.get("Location") if not location: return str(resp.url) current = location return current except Exception: logger.warning(f"b23 短链重定向失败: {url}") return None