Files
HeXi/hexi/plugins/nonebot_plugin_video_analysis/fetchers/rednote_content.py
T

367 lines
14 KiB
Python
Raw Normal View History

"""小红书笔记解析 — window.__INITIAL_STATE__ 静态提取
借鉴 nonebot-plugin-parser 的 XiaoHongShuParser:
- xhslink.com/cn 短链 → 重定向(保留 xsec_token 参数)
- xiaohongshu.com/explore/{id}?xsec_token=... → 桌面端点
- xiaohongshu.com/discovery/item/{id}?xsec_token=... → 移动端点(fallback)
- 页面 HTML 提取 window.__INITIAL_STATE__ → noteDetailMap[id].note
请求端点依次尝试 国际站 rednote.com(含无水印原片 originVideoKey)
→ 国内站 xiaohongshu.com(国内笔记国际站无数据,需 cookies.txt 登录态)。
返回 (title, 文件):
- 图文笔记 → (title, [图片路径列表])
- 视频笔记 → (title, 视频文件 Path,h265 无水印优先)
- 纯文字笔记 → (title, [])
- 解析失败 → (None, None)
"""
import asyncio
import json
import re
import tempfile
import urllib.parse
from datetime import datetime
from pathlib import Path
from typing import Optional, Union
from nonebot import logger
from ..models import ContentFetchError
from ..utils import get_temp_root, parse_netscape_cookies, slugify
DATA_DIR = Path(__file__).resolve().parent.parent / "data"
REDNOTE_UA = (
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 "
"(KHTML, like Gecko) Chrome/122.0.0.0 Safari/537.36"
)
REDNOTE_REFERER = "https://www.xiaohongshu.com/"
REDNOTE_SHORT_RE = re.compile(r"xhslink\.(?:com|cn)/[A-Za-z0-9._?%&+=/#@-]+")
EXPLORE_RE = re.compile(
r"xiaohongshu\.com/explore/([0-9a-zA-Z]+)(\?[^\s\"'<>]+)?"
)
DISCOVERY_RE = re.compile(
r"xiaohongshu\.com/discovery/item/([0-9a-zA-Z]+)(\?[^\s\"'<>]+)?"
)
INITIAL_STATE_RE = re.compile(r"window\.__INITIAL_STATE__=(.*?)</script>", re.S)
# 请求 host 顺序:国际站优先(含无水印原片),失败回退国内站(国内笔记)
REDNOTE_HOSTS = ("www.rednote.com", "www.xiaohongshu.com")
async def fetch_rednote_content(
url: str,
) -> tuple[Optional[str], Optional[Union[Path, list[Path]]]]:
"""解析小红书笔记链接"""
try:
return await _parse(url)
except Exception:
logger.exception(f"小红书解析失败: {url}")
return None, None
async def _parse(url: str):
# 1. 短链 → 重定向(重定向 URL 带 xsec_token,必须保留)
if "xhslink" in url:
resolved = await _resolve_short_link(url)
if resolved:
logger.info(f"xhslink 重定向: {url} -> {resolved}")
url = resolved
# 2. explore 优先(国际站才有 originVideoKey 无水印原片),
# 依次尝试 国际站 → 国内站;失败再用同一 id+query 回退 discovery 端点
if m := EXPLORE_RE.search(url):
note_id, query = m.group(1), (m.group(2) or "").lstrip("?")
return await _fetch_with_retry(note_id, query, ("explore", "discovery"))
if m := DISCOVERY_RE.search(url):
note_id, query = m.group(1), (m.group(2) or "").lstrip("?")
return await _fetch_with_retry(note_id, query, ("discovery",))
# 3. 短链重定向终态: xiaohongshu.com/explore?target_note_id={id}&xsec_token=...
# (discovery/item 会 302 到该形态,note id 在 query 里),
# 只保留访问必需的 xsec_token / xsec_source
if m := re.search(r"xiaohongshu\.com/explore\?([^\"'<>]+)", url):
qs = urllib.parse.parse_qs(m.group(1))
note_id = (qs.get("target_note_id") or [""])[0]
if note_id:
keep = {
k: v
for k, v in qs.items()
if k in ("xsec_token", "xsec_source")
}
query = urllib.parse.urlencode(keep, doseq=True)
return await _fetch_with_retry(
note_id, query, ("explore", "discovery")
)
logger.warning(f"无法识别的小红书链接: {url}")
return None, None
async def _fetch_with_retry(
note_id: str, query: str, paths: tuple[str, ...]
) -> tuple[str, Union[Path, list[Path]]] | None:
"""依次尝试 国际站 → 国内站 × 端点;全部失败后稍候重试一轮
小红书对突发请求会节流(页面 200 但 noteDetailMap 为空/无数据),
重试一轮可绕过大部分瞬时风控;重试仍失败返回 (None, None)。
"""
for attempt in range(2):
for host in REDNOTE_HOSTS:
for path in paths:
fetch = _parse_explore if path == "explore" else _parse_discovery
try:
return await fetch(host, note_id, query)
except Exception as e:
logger.warning(
f"小红书 {host}/{path} 解析失败(第 {attempt + 1} 轮): {e}"
)
if attempt == 0:
await asyncio.sleep(2)
logger.warning(f"小红书解析失败(所有端点): {note_id}")
return None, None
async def _parse_explore(host: str, note_id: str, query: str):
# 国际站 rednote.com 的页面数据含 video.consumer.originVideoKey(无水印原片)
url = f"https://{host}/explore/{note_id}?{query}"
logger.info(f"小红书 explore: {url}")
html = await _fetch_page(
url,
headers={
"User-Agent": REDNOTE_UA,
"Referer": f"https://{host}/explore/{note_id}",
"origin": f"https://{host}",
"accept": (
"text/html,application/xhtml+xml,application/xml;q=0.9,"
"image/avif,image/webp,image/apng,*/*;q=0.8,"
"application/signed-exchange;v=b3;q=0.7"
),
"cookie": _build_cookie_header(),
},
)
note = _extract_note(html, note_id)
return await _build_result(note)
async def _parse_discovery(host: str, note_id: str, query: str):
url = f"https://{host}/discovery/item/{note_id}?{query}"
logger.info(f"小红书 discovery: {url}")
html = await _fetch_page(
url,
headers={
"User-Agent": REDNOTE_UA,
"Referer": f"https://{host}/discovery/item/{note_id}",
"origin": f"https://{host}",
"x-requested-with": "XMLHttpRequest",
"sec-fetch-site": "same-origin",
"sec-fetch-mode": "cors",
"sec-fetch-dest": "empty",
"cookie": _build_cookie_header(),
},
)
note = _extract_note(html, note_id)
return await _build_result(note)
def _build_cookie_header() -> str:
"""从 cookies.txt 构建小红书登录 cookie 头(无登录态时返回空串)"""
cookies_path = DATA_DIR / "cookies.txt"
if not cookies_path.exists():
return ""
cookies = parse_netscape_cookies(str(cookies_path))
rednote = [c for c in cookies if "xiaohongshu" in c.get("domain", "")]
if not rednote:
logger.warning("cookies.txt 中无小红书登录态,匿名访问(依赖链接 xsec_token)")
return ""
logger.info(f"小红书登录态: {len(rednote)} 条 cookie")
return "; ".join(f"{c['name']}={c['value']}" for c in rednote)
def _extract_note(html: str, note_id: str) -> dict:
"""从 __INITIAL_STATE__ 提取笔记详情"""
m = INITIAL_STATE_RE.search(html)
if not m:
raise ContentFetchError("小红书页面无 __INITIAL_STATE__(可能已删除或风控)")
raw = m.group(1)
# JS 语法清理:__INITIAL_STATE__ 不是纯 JSON
# - undefined → null(老问题)
# - new Map([]) / new Set([])(2026-08-24 实测:
# "noteDetailMap":new Map([]) 不做处理 json.loads 必挂)
raw = raw.replace("undefined", "null")
raw = re.sub(r"new Map\([^)]*\)", "{}", raw)
raw = re.sub(r"new Set\([^)]*\)", "[]", raw)
try:
data = json.loads(raw)
except json.JSONDecodeError:
raise ContentFetchError("小红书 __INITIAL_STATE__ JSON 解析失败")
note = ((data.get("note") or {}).get("noteDetailMap") or {}).get(note_id)
if not note:
raise ContentFetchError(f"页面数据中未找到笔记 {note_id}")
return note.get("note") or {}
async def _build_result(note: dict) -> tuple[str, Union[Path, list[Path]]]:
# 空壳 note(短链重定向带 undertake_note_error=该内容暂时无法查看)
# → 笔记已删除/私密,直接报错而不是误判为纯文字笔记
if not any(
note.get(k) for k in ("title", "desc", "type", "imageList", "video")
):
raise ContentFetchError("笔记内容不可见(可能已删除/私密),无法解析")
title = note.get("title") or ""
desc = note.get("desc") or ""
nickname = ((note.get("user") or {}).get("nickname")) or "小红书用户"
text = title or desc or "小红书笔记"
# 1. 视频笔记 → 无水印原片优先
if note.get("type") == "video" and note.get("video"):
# 1a. 无水印原片(国际站数据 video.consumer.originVideoKey)
consumer = (note["video"].get("consumer") or {})
okey = consumer.get("originVideoKey")
if okey:
video_url = f"https://sns-video-bd.xhscdn.com/{okey}"
logger.info(f"小红书视频: 无水印原片 originVideoKey={okey[:30]}...")
file_name = _build_file_name(nickname, text, "视频")
video_path = await _download_video(video_url, file_name)
return text, video_path
# 1b. 无 originVideoKey(国内站数据)→ 从 stream 分组选无水印原片
# 国内站 masterUrl 同样是 sns-video-v6 原片 CDN(无水印),
# 但不同抓取批次返回的清晰度集合不同(同组多条/分组顺序不定),
# 因此跨全部编码分组收集候选,取 size 最大(质量最高)的流。
stream = ((note["video"].get("media") or {}).get("stream")) or {}
candidates = [
it
for items in stream.values()
if isinstance(items, list)
for it in items
if isinstance(it, dict) and it.get("masterUrl")
]
if candidates:
best = max(
candidates,
key=lambda it: (it.get("size") or 0, it.get("avgBitrate") or 0),
)
video_url = best["masterUrl"]
duration = best.get("duration", 0)
logger.info(
f"小红书视频: 无水印流 {best.get('qualityType')} "
f"{best.get('width')}x{best.get('height')} {best.get('fps')}fps "
f"size={best.get('size')} duration={duration}ms"
)
file_name = _build_file_name(nickname, text, "视频")
video_path = await _download_video(video_url, file_name)
return text, video_path
raise ContentFetchError("小红书视频流解析失败")
# 2. 图文笔记
images = [
img.get("urlDefault") or img.get("url")
for img in note.get("imageList") or []
]
images = [u for u in images if u]
if not images:
logger.info(f"小红书文字笔记: {text[:30]}")
return text, []
file_name = _build_file_name(nickname, text, "笔记")
file_paths = await _download_images(images, file_name)
logger.info(f"小红书图文笔记: 作者={nickname}, 图片={len(images)} 张")
return text, file_paths
def _build_file_name(nickname: str, title: str, kind: str) -> str:
"""构建文件名 stem: {作者}_{标题}_{类型}_{时间}"""
slug_nickname = slugify(nickname)
slug_title = slugify(title or "", max_length=15)
if not slug_title:
slug_title = datetime.now().strftime("%H%M%S")
time_suffix = datetime.now().strftime("%H%M%S")
return f"{slug_nickname}_{slug_title}_{kind}_{time_suffix}"
async def _download_images(image_urls: list[str], file_name: str) -> list[Path]:
"""并发下载图片(复用抖音图文的下载流程)"""
import httpx
from .douyin_api import _process_note_with_parsed
tmp_root = get_temp_root("xiaohongshu")
headers = {
"Referer": REDNOTE_REFERER,
"User-Agent": REDNOTE_UA,
}
return await _process_note_with_parsed(
[[u] for u in image_urls], None, tmp_root, file_name, headers
)
async def _download_video(video_url: str, file_name: str) -> Path:
"""流式下载视频
注意:sns-video-bd(无水印原片)不带 Referer 或带 xiaohongshu.com
均可,但带 rednote.com Referer 会 403,因此不设 Referer。
"""
import httpx
tmp_root = get_temp_root("xiaohongshu")
output_path = tmp_root / f"{file_name}.mp4"
headers = {"User-Agent": REDNOTE_UA}
async with httpx.AsyncClient(headers=headers, timeout=300) as client:
async with client.stream("GET", video_url) as resp:
resp.raise_for_status()
with open(output_path, "wb") as f:
async for chunk in resp.aiter_bytes(8192):
f.write(chunk)
logger.info(f"小红书视频下载完成: {output_path}")
return output_path
async def _fetch_page(url: str, headers: dict) -> str:
import httpx
async with httpx.AsyncClient(
headers=headers, timeout=20, follow_redirects=True
) as client:
resp = await client.get(url)
if resp.status_code >= 400:
raise ContentFetchError(f"小红书页面请求失败: status={resp.status_code}")
return resp.text
async def _resolve_short_link(url: str) -> Optional[str]:
"""xhslink 短链重定向(最多 3 跳取最终 URL,重定向 URL 带 xsec_token)
借鉴 nonebot-plugin-parser:xhslink 用移动端 headers 请求
(origin / x-requested-with 等),避免被当作非 App 来源拒绝。
xhslink.cn 可能先跳到 xhslink.com 再跳小红书,需循环取跳。
"""
import httpx
headers = {
"User-Agent": REDNOTE_UA,
"Referer": REDNOTE_REFERER,
"origin": "https://www.xiaohongshu.com",
"x-requested-with": "XMLHttpRequest",
}
try:
async with httpx.AsyncClient(
headers=headers, follow_redirects=False, timeout=10
) as client:
current = url
for _ in range(3):
resp = await client.get(current)
if resp.status_code >= 400:
return None
location = resp.headers.get("Location")
if not location:
return str(resp.url)
# Location 可能是相对路径(如 /explore?...),需拼上当前 URL
current = urllib.parse.urljoin(current, location)
return current
except Exception:
logger.warning(f"xhslink 重定向失败: {url}")
return None