- hexi core: message handling, rate limiting, cooldown, plugin manager - Custom plugins: BF stats, daily check-in, quotes, persona cards, etc. - Community plugins vendored under hexi/plugins with local fixes - Web admin frontends (learning-chat, persona-admin), unified hexi/web - Tests for rate_limit/cooldown/memes/persona; poetry.lock Co-Authored-By: Claude <noreply@anthropic.com>
367 lines
14 KiB
Python
367 lines
14 KiB
Python
"""小红书笔记解析 — window.__INITIAL_STATE__ 静态提取
|
||
|
||
借鉴 nonebot-plugin-parser 的 XiaoHongShuParser:
|
||
- xhslink.com/cn 短链 → 重定向(保留 xsec_token 参数)
|
||
- xiaohongshu.com/explore/{id}?xsec_token=... → 桌面端点
|
||
- xiaohongshu.com/discovery/item/{id}?xsec_token=... → 移动端点(fallback)
|
||
- 页面 HTML 提取 window.__INITIAL_STATE__ → noteDetailMap[id].note
|
||
|
||
请求端点依次尝试 国际站 rednote.com(含无水印原片 originVideoKey)
|
||
→ 国内站 xiaohongshu.com(国内笔记国际站无数据,需 cookies.txt 登录态)。
|
||
|
||
返回 (title, 文件):
|
||
- 图文笔记 → (title, [图片路径列表])
|
||
- 视频笔记 → (title, 视频文件 Path,h265 无水印优先)
|
||
- 纯文字笔记 → (title, [])
|
||
- 解析失败 → (None, None)
|
||
"""
|
||
|
||
import asyncio
|
||
import json
|
||
import re
|
||
import tempfile
|
||
import urllib.parse
|
||
from datetime import datetime
|
||
from pathlib import Path
|
||
from typing import Optional, Union
|
||
|
||
from nonebot import logger
|
||
|
||
from ..models import ContentFetchError
|
||
from ..utils import get_temp_root, parse_netscape_cookies, slugify
|
||
|
||
DATA_DIR = Path(__file__).resolve().parent.parent / "data"
|
||
|
||
REDNOTE_UA = (
|
||
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 "
|
||
"(KHTML, like Gecko) Chrome/122.0.0.0 Safari/537.36"
|
||
)
|
||
REDNOTE_REFERER = "https://www.xiaohongshu.com/"
|
||
|
||
REDNOTE_SHORT_RE = re.compile(r"xhslink\.(?:com|cn)/[A-Za-z0-9._?%&+=/#@-]+")
|
||
EXPLORE_RE = re.compile(
|
||
r"xiaohongshu\.com/explore/([0-9a-zA-Z]+)(\?[^\s\"'<>]+)?"
|
||
)
|
||
DISCOVERY_RE = re.compile(
|
||
r"xiaohongshu\.com/discovery/item/([0-9a-zA-Z]+)(\?[^\s\"'<>]+)?"
|
||
)
|
||
INITIAL_STATE_RE = re.compile(r"window\.__INITIAL_STATE__=(.*?)</script>", re.S)
|
||
|
||
# 请求 host 顺序:国际站优先(含无水印原片),失败回退国内站(国内笔记)
|
||
REDNOTE_HOSTS = ("www.rednote.com", "www.xiaohongshu.com")
|
||
|
||
|
||
async def fetch_rednote_content(
|
||
url: str,
|
||
) -> tuple[Optional[str], Optional[Union[Path, list[Path]]]]:
|
||
"""解析小红书笔记链接"""
|
||
try:
|
||
return await _parse(url)
|
||
except Exception:
|
||
logger.exception(f"小红书解析失败: {url}")
|
||
return None, None
|
||
|
||
|
||
async def _parse(url: str):
|
||
# 1. 短链 → 重定向(重定向 URL 带 xsec_token,必须保留)
|
||
if "xhslink" in url:
|
||
resolved = await _resolve_short_link(url)
|
||
if resolved:
|
||
logger.info(f"xhslink 重定向: {url} -> {resolved}")
|
||
url = resolved
|
||
|
||
# 2. explore 优先(国际站才有 originVideoKey 无水印原片),
|
||
# 依次尝试 国际站 → 国内站;失败再用同一 id+query 回退 discovery 端点
|
||
if m := EXPLORE_RE.search(url):
|
||
note_id, query = m.group(1), (m.group(2) or "").lstrip("?")
|
||
return await _fetch_with_retry(note_id, query, ("explore", "discovery"))
|
||
if m := DISCOVERY_RE.search(url):
|
||
note_id, query = m.group(1), (m.group(2) or "").lstrip("?")
|
||
return await _fetch_with_retry(note_id, query, ("discovery",))
|
||
|
||
# 3. 短链重定向终态: xiaohongshu.com/explore?target_note_id={id}&xsec_token=...
|
||
# (discovery/item 会 302 到该形态,note id 在 query 里),
|
||
# 只保留访问必需的 xsec_token / xsec_source
|
||
if m := re.search(r"xiaohongshu\.com/explore\?([^\"'<>]+)", url):
|
||
qs = urllib.parse.parse_qs(m.group(1))
|
||
note_id = (qs.get("target_note_id") or [""])[0]
|
||
if note_id:
|
||
keep = {
|
||
k: v
|
||
for k, v in qs.items()
|
||
if k in ("xsec_token", "xsec_source")
|
||
}
|
||
query = urllib.parse.urlencode(keep, doseq=True)
|
||
return await _fetch_with_retry(
|
||
note_id, query, ("explore", "discovery")
|
||
)
|
||
|
||
logger.warning(f"无法识别的小红书链接: {url}")
|
||
return None, None
|
||
|
||
|
||
async def _fetch_with_retry(
|
||
note_id: str, query: str, paths: tuple[str, ...]
|
||
) -> tuple[str, Union[Path, list[Path]]] | None:
|
||
"""依次尝试 国际站 → 国内站 × 端点;全部失败后稍候重试一轮
|
||
|
||
小红书对突发请求会节流(页面 200 但 noteDetailMap 为空/无数据),
|
||
重试一轮可绕过大部分瞬时风控;重试仍失败返回 (None, None)。
|
||
"""
|
||
for attempt in range(2):
|
||
for host in REDNOTE_HOSTS:
|
||
for path in paths:
|
||
fetch = _parse_explore if path == "explore" else _parse_discovery
|
||
try:
|
||
return await fetch(host, note_id, query)
|
||
except Exception as e:
|
||
logger.warning(
|
||
f"小红书 {host}/{path} 解析失败(第 {attempt + 1} 轮): {e}"
|
||
)
|
||
if attempt == 0:
|
||
await asyncio.sleep(2)
|
||
logger.warning(f"小红书解析失败(所有端点): {note_id}")
|
||
return None, None
|
||
|
||
|
||
async def _parse_explore(host: str, note_id: str, query: str):
|
||
# 国际站 rednote.com 的页面数据含 video.consumer.originVideoKey(无水印原片)
|
||
url = f"https://{host}/explore/{note_id}?{query}"
|
||
logger.info(f"小红书 explore: {url}")
|
||
html = await _fetch_page(
|
||
url,
|
||
headers={
|
||
"User-Agent": REDNOTE_UA,
|
||
"Referer": f"https://{host}/explore/{note_id}",
|
||
"origin": f"https://{host}",
|
||
"accept": (
|
||
"text/html,application/xhtml+xml,application/xml;q=0.9,"
|
||
"image/avif,image/webp,image/apng,*/*;q=0.8,"
|
||
"application/signed-exchange;v=b3;q=0.7"
|
||
),
|
||
"cookie": _build_cookie_header(),
|
||
},
|
||
)
|
||
note = _extract_note(html, note_id)
|
||
return await _build_result(note)
|
||
|
||
|
||
async def _parse_discovery(host: str, note_id: str, query: str):
|
||
url = f"https://{host}/discovery/item/{note_id}?{query}"
|
||
logger.info(f"小红书 discovery: {url}")
|
||
html = await _fetch_page(
|
||
url,
|
||
headers={
|
||
"User-Agent": REDNOTE_UA,
|
||
"Referer": f"https://{host}/discovery/item/{note_id}",
|
||
"origin": f"https://{host}",
|
||
"x-requested-with": "XMLHttpRequest",
|
||
"sec-fetch-site": "same-origin",
|
||
"sec-fetch-mode": "cors",
|
||
"sec-fetch-dest": "empty",
|
||
"cookie": _build_cookie_header(),
|
||
},
|
||
)
|
||
note = _extract_note(html, note_id)
|
||
return await _build_result(note)
|
||
|
||
|
||
def _build_cookie_header() -> str:
|
||
"""从 cookies.txt 构建小红书登录 cookie 头(无登录态时返回空串)"""
|
||
cookies_path = DATA_DIR / "cookies.txt"
|
||
if not cookies_path.exists():
|
||
return ""
|
||
cookies = parse_netscape_cookies(str(cookies_path))
|
||
rednote = [c for c in cookies if "xiaohongshu" in c.get("domain", "")]
|
||
if not rednote:
|
||
logger.warning("cookies.txt 中无小红书登录态,匿名访问(依赖链接 xsec_token)")
|
||
return ""
|
||
logger.info(f"小红书登录态: {len(rednote)} 条 cookie")
|
||
return "; ".join(f"{c['name']}={c['value']}" for c in rednote)
|
||
|
||
|
||
def _extract_note(html: str, note_id: str) -> dict:
|
||
"""从 __INITIAL_STATE__ 提取笔记详情"""
|
||
m = INITIAL_STATE_RE.search(html)
|
||
if not m:
|
||
raise ContentFetchError("小红书页面无 __INITIAL_STATE__(可能已删除或风控)")
|
||
raw = m.group(1)
|
||
# JS 语法清理:__INITIAL_STATE__ 不是纯 JSON
|
||
# - undefined → null(老问题)
|
||
# - new Map([]) / new Set([])(2026-08-24 实测:
|
||
# "noteDetailMap":new Map([]) 不做处理 json.loads 必挂)
|
||
raw = raw.replace("undefined", "null")
|
||
raw = re.sub(r"new Map\([^)]*\)", "{}", raw)
|
||
raw = re.sub(r"new Set\([^)]*\)", "[]", raw)
|
||
try:
|
||
data = json.loads(raw)
|
||
except json.JSONDecodeError:
|
||
raise ContentFetchError("小红书 __INITIAL_STATE__ JSON 解析失败")
|
||
note = ((data.get("note") or {}).get("noteDetailMap") or {}).get(note_id)
|
||
if not note:
|
||
raise ContentFetchError(f"页面数据中未找到笔记 {note_id}")
|
||
return note.get("note") or {}
|
||
|
||
|
||
async def _build_result(note: dict) -> tuple[str, Union[Path, list[Path]]]:
|
||
# 空壳 note(短链重定向带 undertake_note_error=该内容暂时无法查看)
|
||
# → 笔记已删除/私密,直接报错而不是误判为纯文字笔记
|
||
if not any(
|
||
note.get(k) for k in ("title", "desc", "type", "imageList", "video")
|
||
):
|
||
raise ContentFetchError("笔记内容不可见(可能已删除/私密),无法解析")
|
||
title = note.get("title") or ""
|
||
desc = note.get("desc") or ""
|
||
nickname = ((note.get("user") or {}).get("nickname")) or "小红书用户"
|
||
text = title or desc or "小红书笔记"
|
||
|
||
# 1. 视频笔记 → 无水印原片优先
|
||
if note.get("type") == "video" and note.get("video"):
|
||
# 1a. 无水印原片(国际站数据 video.consumer.originVideoKey)
|
||
consumer = (note["video"].get("consumer") or {})
|
||
okey = consumer.get("originVideoKey")
|
||
if okey:
|
||
video_url = f"https://sns-video-bd.xhscdn.com/{okey}"
|
||
logger.info(f"小红书视频: 无水印原片 originVideoKey={okey[:30]}...")
|
||
file_name = _build_file_name(nickname, text, "视频")
|
||
video_path = await _download_video(video_url, file_name)
|
||
return text, video_path
|
||
|
||
# 1b. 无 originVideoKey(国内站数据)→ 从 stream 分组选无水印原片
|
||
# 国内站 masterUrl 同样是 sns-video-v6 原片 CDN(无水印),
|
||
# 但不同抓取批次返回的清晰度集合不同(同组多条/分组顺序不定),
|
||
# 因此跨全部编码分组收集候选,取 size 最大(质量最高)的流。
|
||
stream = ((note["video"].get("media") or {}).get("stream")) or {}
|
||
candidates = [
|
||
it
|
||
for items in stream.values()
|
||
if isinstance(items, list)
|
||
for it in items
|
||
if isinstance(it, dict) and it.get("masterUrl")
|
||
]
|
||
if candidates:
|
||
best = max(
|
||
candidates,
|
||
key=lambda it: (it.get("size") or 0, it.get("avgBitrate") or 0),
|
||
)
|
||
video_url = best["masterUrl"]
|
||
duration = best.get("duration", 0)
|
||
logger.info(
|
||
f"小红书视频: 无水印流 {best.get('qualityType')} "
|
||
f"{best.get('width')}x{best.get('height')} {best.get('fps')}fps "
|
||
f"size={best.get('size')} duration={duration}ms"
|
||
)
|
||
file_name = _build_file_name(nickname, text, "视频")
|
||
video_path = await _download_video(video_url, file_name)
|
||
return text, video_path
|
||
raise ContentFetchError("小红书视频流解析失败")
|
||
|
||
# 2. 图文笔记
|
||
images = [
|
||
img.get("urlDefault") or img.get("url")
|
||
for img in note.get("imageList") or []
|
||
]
|
||
images = [u for u in images if u]
|
||
if not images:
|
||
logger.info(f"小红书文字笔记: {text[:30]}")
|
||
return text, []
|
||
|
||
file_name = _build_file_name(nickname, text, "笔记")
|
||
file_paths = await _download_images(images, file_name)
|
||
logger.info(f"小红书图文笔记: 作者={nickname}, 图片={len(images)} 张")
|
||
return text, file_paths
|
||
|
||
|
||
def _build_file_name(nickname: str, title: str, kind: str) -> str:
|
||
"""构建文件名 stem: {作者}_{标题}_{类型}_{时间}"""
|
||
slug_nickname = slugify(nickname)
|
||
slug_title = slugify(title or "", max_length=15)
|
||
if not slug_title:
|
||
slug_title = datetime.now().strftime("%H%M%S")
|
||
time_suffix = datetime.now().strftime("%H%M%S")
|
||
return f"{slug_nickname}_{slug_title}_{kind}_{time_suffix}"
|
||
|
||
|
||
async def _download_images(image_urls: list[str], file_name: str) -> list[Path]:
|
||
"""并发下载图片(复用抖音图文的下载流程)"""
|
||
import httpx
|
||
|
||
from .douyin_api import _process_note_with_parsed
|
||
|
||
tmp_root = get_temp_root("xiaohongshu")
|
||
headers = {
|
||
"Referer": REDNOTE_REFERER,
|
||
"User-Agent": REDNOTE_UA,
|
||
}
|
||
return await _process_note_with_parsed(
|
||
[[u] for u in image_urls], None, tmp_root, file_name, headers
|
||
)
|
||
|
||
|
||
async def _download_video(video_url: str, file_name: str) -> Path:
|
||
"""流式下载视频
|
||
|
||
注意:sns-video-bd(无水印原片)不带 Referer 或带 xiaohongshu.com
|
||
均可,但带 rednote.com Referer 会 403,因此不设 Referer。
|
||
"""
|
||
import httpx
|
||
|
||
tmp_root = get_temp_root("xiaohongshu")
|
||
output_path = tmp_root / f"{file_name}.mp4"
|
||
headers = {"User-Agent": REDNOTE_UA}
|
||
async with httpx.AsyncClient(headers=headers, timeout=300) as client:
|
||
async with client.stream("GET", video_url) as resp:
|
||
resp.raise_for_status()
|
||
with open(output_path, "wb") as f:
|
||
async for chunk in resp.aiter_bytes(8192):
|
||
f.write(chunk)
|
||
logger.info(f"小红书视频下载完成: {output_path}")
|
||
return output_path
|
||
|
||
|
||
async def _fetch_page(url: str, headers: dict) -> str:
|
||
import httpx
|
||
|
||
async with httpx.AsyncClient(
|
||
headers=headers, timeout=20, follow_redirects=True
|
||
) as client:
|
||
resp = await client.get(url)
|
||
if resp.status_code >= 400:
|
||
raise ContentFetchError(f"小红书页面请求失败: status={resp.status_code}")
|
||
return resp.text
|
||
|
||
|
||
async def _resolve_short_link(url: str) -> Optional[str]:
|
||
"""xhslink 短链重定向(最多 3 跳取最终 URL,重定向 URL 带 xsec_token)
|
||
|
||
借鉴 nonebot-plugin-parser:xhslink 用移动端 headers 请求
|
||
(origin / x-requested-with 等),避免被当作非 App 来源拒绝。
|
||
xhslink.cn 可能先跳到 xhslink.com 再跳小红书,需循环取跳。
|
||
"""
|
||
import httpx
|
||
|
||
headers = {
|
||
"User-Agent": REDNOTE_UA,
|
||
"Referer": REDNOTE_REFERER,
|
||
"origin": "https://www.xiaohongshu.com",
|
||
"x-requested-with": "XMLHttpRequest",
|
||
}
|
||
try:
|
||
async with httpx.AsyncClient(
|
||
headers=headers, follow_redirects=False, timeout=10
|
||
) as client:
|
||
current = url
|
||
for _ in range(3):
|
||
resp = await client.get(current)
|
||
if resp.status_code >= 400:
|
||
return None
|
||
location = resp.headers.get("Location")
|
||
if not location:
|
||
return str(resp.url)
|
||
# Location 可能是相对路径(如 /explore?...),需拼上当前 URL
|
||
current = urllib.parse.urljoin(current, location)
|
||
return current
|
||
except Exception:
|
||
logger.warning(f"xhslink 重定向失败: {url}")
|
||
return None
|