367 lines
14 KiB
Python
367 lines
14 KiB
Python
"""小红书笔记解析 — window.__INITIAL_STATE__ 静态提取
|
||||
|
|
|
|||
|
|
借鉴 nonebot-plugin-parser 的 XiaoHongShuParser:
|
|||
|
|
- xhslink.com/cn 短链 → 重定向(保留 xsec_token 参数)
|
|||
|
|
- xiaohongshu.com/explore/{id}?xsec_token=... → 桌面端点
|
|||
|
|
- xiaohongshu.com/discovery/item/{id}?xsec_token=... → 移动端点(fallback)
|
|||
|
|
- 页面 HTML 提取 window.__INITIAL_STATE__ → noteDetailMap[id].note
|
|||
|
|
|
|||
|
|
请求端点依次尝试 国际站 rednote.com(含无水印原片 originVideoKey)
|
|||
|
|
→ 国内站 xiaohongshu.com(国内笔记国际站无数据,需 cookies.txt 登录态)。
|
|||
|
|
|
|||
|
|
返回 (title, 文件):
|
|||
|
|
- 图文笔记 → (title, [图片路径列表])
|
|||
|
|
- 视频笔记 → (title, 视频文件 Path,h265 无水印优先)
|
|||
|
|
- 纯文字笔记 → (title, [])
|
|||
|
|
- 解析失败 → (None, None)
|
|||
|
|
"""
|
|||
|
|
|
|||
|
|
import asyncio
|
|||
|
|
import json
|
|||
|
|
import re
|
|||
|
|
import tempfile
|
|||
|
|
import urllib.parse
|
|||
|
|
from datetime import datetime
|
|||
|
|
from pathlib import Path
|
|||
|
|
from typing import Optional, Union
|
|||
|
|
|
|||
|
|
from nonebot import logger
|
|||
|
|
|
|||
|
|
from ..models import ContentFetchError
|
|||
|
|
from ..utils import get_temp_root, parse_netscape_cookies, slugify
|
|||
|
|
|
|||
|
|
DATA_DIR = Path(__file__).resolve().parent.parent / "data"
|
|||
|
|
|
|||
|
|
REDNOTE_UA = (
|
|||
|
|
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 "
|
|||
|
|
"(KHTML, like Gecko) Chrome/122.0.0.0 Safari/537.36"
|
|||
|
|
)
|
|||
|
|
REDNOTE_REFERER = "https://www.xiaohongshu.com/"
|
|||
|
|
|
|||
|
|
REDNOTE_SHORT_RE = re.compile(r"xhslink\.(?:com|cn)/[A-Za-z0-9._?%&+=/#@-]+")
|
|||
|
|
EXPLORE_RE = re.compile(
|
|||
|
|
r"xiaohongshu\.com/explore/([0-9a-zA-Z]+)(\?[^\s\"'<>]+)?"
|
|||
|
|
)
|
|||
|
|
DISCOVERY_RE = re.compile(
|
|||
|
|
r"xiaohongshu\.com/discovery/item/([0-9a-zA-Z]+)(\?[^\s\"'<>]+)?"
|
|||
|
|
)
|
|||
|
|
INITIAL_STATE_RE = re.compile(r"window\.__INITIAL_STATE__=(.*?)</script>", re.S)
|
|||
|
|
|
|||
|
|
# 请求 host 顺序:国际站优先(含无水印原片),失败回退国内站(国内笔记)
|
|||
|
|
REDNOTE_HOSTS = ("www.rednote.com", "www.xiaohongshu.com")
|
|||
|
|
|
|||
|
|
|
|||
|
|
async def fetch_rednote_content(
|
|||
|
|
url: str,
|
|||
|
|
) -> tuple[Optional[str], Optional[Union[Path, list[Path]]]]:
|
|||
|
|
"""解析小红书笔记链接"""
|
|||
|
|
try:
|
|||
|
|
return await _parse(url)
|
|||
|
|
except Exception:
|
|||
|
|
logger.exception(f"小红书解析失败: {url}")
|
|||
|
|
return None, None
|
|||
|
|
|
|||
|
|
|
|||
|
|
async def _parse(url: str):
|
|||
|
|
# 1. 短链 → 重定向(重定向 URL 带 xsec_token,必须保留)
|
|||
|
|
if "xhslink" in url:
|
|||
|
|
resolved = await _resolve_short_link(url)
|
|||
|
|
if resolved:
|
|||
|
|
logger.info(f"xhslink 重定向: {url} -> {resolved}")
|
|||
|
|
url = resolved
|
|||
|
|
|
|||
|
|
# 2. explore 优先(国际站才有 originVideoKey 无水印原片),
|
|||
|
|
# 依次尝试 国际站 → 国内站;失败再用同一 id+query 回退 discovery 端点
|
|||
|
|
if m := EXPLORE_RE.search(url):
|
|||
|
|
note_id, query = m.group(1), (m.group(2) or "").lstrip("?")
|
|||
|
|
return await _fetch_with_retry(note_id, query, ("explore", "discovery"))
|
|||
|
|
if m := DISCOVERY_RE.search(url):
|
|||
|
|
note_id, query = m.group(1), (m.group(2) or "").lstrip("?")
|
|||
|
|
return await _fetch_with_retry(note_id, query, ("discovery",))
|
|||
|
|
|
|||
|
|
# 3. 短链重定向终态: xiaohongshu.com/explore?target_note_id={id}&xsec_token=...
|
|||
|
|
# (discovery/item 会 302 到该形态,note id 在 query 里),
|
|||
|
|
# 只保留访问必需的 xsec_token / xsec_source
|
|||
|
|
if m := re.search(r"xiaohongshu\.com/explore\?([^\"'<>]+)", url):
|
|||
|
|
qs = urllib.parse.parse_qs(m.group(1))
|
|||
|
|
note_id = (qs.get("target_note_id") or [""])[0]
|
|||
|
|
if note_id:
|
|||
|
|
keep = {
|
|||
|
|
k: v
|
|||
|
|
for k, v in qs.items()
|
|||
|
|
if k in ("xsec_token", "xsec_source")
|
|||
|
|
}
|
|||
|
|
query = urllib.parse.urlencode(keep, doseq=True)
|
|||
|
|
return await _fetch_with_retry(
|
|||
|
|
note_id, query, ("explore", "discovery")
|
|||
|
|
)
|
|||
|
|
|
|||
|
|
logger.warning(f"无法识别的小红书链接: {url}")
|
|||
|
|
return None, None
|
|||
|
|
|
|||
|
|
|
|||
|
|
async def _fetch_with_retry(
|
|||
|
|
note_id: str, query: str, paths: tuple[str, ...]
|
|||
|
|
) -> tuple[str, Union[Path, list[Path]]] | None:
|
|||
|
|
"""依次尝试 国际站 → 国内站 × 端点;全部失败后稍候重试一轮
|
|||
|
|
|
|||
|
|
小红书对突发请求会节流(页面 200 但 noteDetailMap 为空/无数据),
|
|||
|
|
重试一轮可绕过大部分瞬时风控;重试仍失败返回 (None, None)。
|
|||
|
|
"""
|
|||
|
|
for attempt in range(2):
|
|||
|
|
for host in REDNOTE_HOSTS:
|
|||
|
|
for path in paths:
|
|||
|
|
fetch = _parse_explore if path == "explore" else _parse_discovery
|
|||
|
|
try:
|
|||
|
|
return await fetch(host, note_id, query)
|
|||
|
|
except Exception as e:
|
|||
|
|
logger.warning(
|
|||
|
|
f"小红书 {host}/{path} 解析失败(第 {attempt + 1} 轮): {e}"
|
|||
|
|
)
|
|||
|
|
if attempt == 0:
|
|||
|
|
await asyncio.sleep(2)
|
|||
|
|
logger.warning(f"小红书解析失败(所有端点): {note_id}")
|
|||
|
|
return None, None
|
|||
|
|
|
|||
|
|
|
|||
|
|
async def _parse_explore(host: str, note_id: str, query: str):
|
|||
|
|
# 国际站 rednote.com 的页面数据含 video.consumer.originVideoKey(无水印原片)
|
|||
|
|
url = f"https://{host}/explore/{note_id}?{query}"
|
|||
|
|
logger.info(f"小红书 explore: {url}")
|
|||
|
|
html = await _fetch_page(
|
|||
|
|
url,
|
|||
|
|
headers={
|
|||
|
|
"User-Agent": REDNOTE_UA,
|
|||
|
|
"Referer": f"https://{host}/explore/{note_id}",
|
|||
|
|
"origin": f"https://{host}",
|
|||
|
|
"accept": (
|
|||
|
|
"text/html,application/xhtml+xml,application/xml;q=0.9,"
|
|||
|
|
"image/avif,image/webp,image/apng,*/*;q=0.8,"
|
|||
|
|
"application/signed-exchange;v=b3;q=0.7"
|
|||
|
|
),
|
|||
|
|
"cookie": _build_cookie_header(),
|
|||
|
|
},
|
|||
|
|
)
|
|||
|
|
note = _extract_note(html, note_id)
|
|||
|
|
return await _build_result(note)
|
|||
|
|
|
|||
|
|
|
|||
|
|
async def _parse_discovery(host: str, note_id: str, query: str):
|
|||
|
|
url = f"https://{host}/discovery/item/{note_id}?{query}"
|
|||
|
|
logger.info(f"小红书 discovery: {url}")
|
|||
|
|
html = await _fetch_page(
|
|||
|
|
url,
|
|||
|
|
headers={
|
|||
|
|
"User-Agent": REDNOTE_UA,
|
|||
|
|
"Referer": f"https://{host}/discovery/item/{note_id}",
|
|||
|
|
"origin": f"https://{host}",
|
|||
|
|
"x-requested-with": "XMLHttpRequest",
|
|||
|
|
"sec-fetch-site": "same-origin",
|
|||
|
|
"sec-fetch-mode": "cors",
|
|||
|
|
"sec-fetch-dest": "empty",
|
|||
|
|
"cookie": _build_cookie_header(),
|
|||
|
|
},
|
|||
|
|
)
|
|||
|
|
note = _extract_note(html, note_id)
|
|||
|
|
return await _build_result(note)
|
|||
|
|
|
|||
|
|
|
|||
|
|
def _build_cookie_header() -> str:
|
|||
|
|
"""从 cookies.txt 构建小红书登录 cookie 头(无登录态时返回空串)"""
|
|||
|
|
cookies_path = DATA_DIR / "cookies.txt"
|
|||
|
|
if not cookies_path.exists():
|
|||
|
|
return ""
|
|||
|
|
cookies = parse_netscape_cookies(str(cookies_path))
|
|||
|
|
rednote = [c for c in cookies if "xiaohongshu" in c.get("domain", "")]
|
|||
|
|
if not rednote:
|
|||
|
|
logger.warning("cookies.txt 中无小红书登录态,匿名访问(依赖链接 xsec_token)")
|
|||
|
|
return ""
|
|||
|
|
logger.info(f"小红书登录态: {len(rednote)} 条 cookie")
|
|||
|
|
return "; ".join(f"{c['name']}={c['value']}" for c in rednote)
|
|||
|
|
|
|||
|
|
|
|||
|
|
def _extract_note(html: str, note_id: str) -> dict:
|
|||
|
|
"""从 __INITIAL_STATE__ 提取笔记详情"""
|
|||
|
|
m = INITIAL_STATE_RE.search(html)
|
|||
|
|
if not m:
|
|||
|
|
raise ContentFetchError("小红书页面无 __INITIAL_STATE__(可能已删除或风控)")
|
|||
|
|
raw = m.group(1)
|
|||
|
|
# JS 语法清理:__INITIAL_STATE__ 不是纯 JSON
|
|||
|
|
# - undefined → null(老问题)
|
|||
|
|
# - new Map([]) / new Set([])(2026-08-24 实测:
|
|||
|
|
# "noteDetailMap":new Map([]) 不做处理 json.loads 必挂)
|
|||
|
|
raw = raw.replace("undefined", "null")
|
|||
|
|
raw = re.sub(r"new Map\([^)]*\)", "{}", raw)
|
|||
|
|
raw = re.sub(r"new Set\([^)]*\)", "[]", raw)
|
|||
|
|
try:
|
|||
|
|
data = json.loads(raw)
|
|||
|
|
except json.JSONDecodeError:
|
|||
|
|
raise ContentFetchError("小红书 __INITIAL_STATE__ JSON 解析失败")
|
|||
|
|
note = ((data.get("note") or {}).get("noteDetailMap") or {}).get(note_id)
|
|||
|
|
if not note:
|
|||
|
|
raise ContentFetchError(f"页面数据中未找到笔记 {note_id}")
|
|||
|
|
return note.get("note") or {}
|
|||
|
|
|
|||
|
|
|
|||
|
|
async def _build_result(note: dict) -> tuple[str, Union[Path, list[Path]]]:
|
|||
|
|
# 空壳 note(短链重定向带 undertake_note_error=该内容暂时无法查看)
|
|||
|
|
# → 笔记已删除/私密,直接报错而不是误判为纯文字笔记
|
|||
|
|
if not any(
|
|||
|
|
note.get(k) for k in ("title", "desc", "type", "imageList", "video")
|
|||
|
|
):
|
|||
|
|
raise ContentFetchError("笔记内容不可见(可能已删除/私密),无法解析")
|
|||
|
|
title = note.get("title") or ""
|
|||
|
|
desc = note.get("desc") or ""
|
|||
|
|
nickname = ((note.get("user") or {}).get("nickname")) or "小红书用户"
|
|||
|
|
text = title or desc or "小红书笔记"
|
|||
|
|
|
|||
|
|
# 1. 视频笔记 → 无水印原片优先
|
|||
|
|
if note.get("type") == "video" and note.get("video"):
|
|||
|
|
# 1a. 无水印原片(国际站数据 video.consumer.originVideoKey)
|
|||
|
|
consumer = (note["video"].get("consumer") or {})
|
|||
|
|
okey = consumer.get("originVideoKey")
|
|||
|
|
if okey:
|
|||
|
|
video_url = f"https://sns-video-bd.xhscdn.com/{okey}"
|
|||
|
|
logger.info(f"小红书视频: 无水印原片 originVideoKey={okey[:30]}...")
|
|||
|
|
file_name = _build_file_name(nickname, text, "视频")
|
|||
|
|
video_path = await _download_video(video_url, file_name)
|
|||
|
|
return text, video_path
|
|||
|
|
|
|||
|
|
# 1b. 无 originVideoKey(国内站数据)→ 从 stream 分组选无水印原片
|
|||
|
|
# 国内站 masterUrl 同样是 sns-video-v6 原片 CDN(无水印),
|
|||
|
|
# 但不同抓取批次返回的清晰度集合不同(同组多条/分组顺序不定),
|
|||
|
|
# 因此跨全部编码分组收集候选,取 size 最大(质量最高)的流。
|
|||
|
|
stream = ((note["video"].get("media") or {}).get("stream")) or {}
|
|||
|
|
candidates = [
|
|||
|
|
it
|
|||
|
|
for items in stream.values()
|
|||
|
|
if isinstance(items, list)
|
|||
|
|
for it in items
|
|||
|
|
if isinstance(it, dict) and it.get("masterUrl")
|
|||
|
|
]
|
|||
|
|
if candidates:
|
|||
|
|
best = max(
|
|||
|
|
candidates,
|
|||
|
|
key=lambda it: (it.get("size") or 0, it.get("avgBitrate") or 0),
|
|||
|
|
)
|
|||
|
|
video_url = best["masterUrl"]
|
|||
|
|
duration = best.get("duration", 0)
|
|||
|
|
logger.info(
|
|||
|
|
f"小红书视频: 无水印流 {best.get('qualityType')} "
|
|||
|
|
f"{best.get('width')}x{best.get('height')} {best.get('fps')}fps "
|
|||
|
|
f"size={best.get('size')} duration={duration}ms"
|
|||
|
|
)
|
|||
|
|
file_name = _build_file_name(nickname, text, "视频")
|
|||
|
|
video_path = await _download_video(video_url, file_name)
|
|||
|
|
return text, video_path
|
|||
|
|
raise ContentFetchError("小红书视频流解析失败")
|
|||
|
|
|
|||
|
|
# 2. 图文笔记
|
|||
|
|
images = [
|
|||
|
|
img.get("urlDefault") or img.get("url")
|
|||
|
|
for img in note.get("imageList") or []
|
|||
|
|
]
|
|||
|
|
images = [u for u in images if u]
|
|||
|
|
if not images:
|
|||
|
|
logger.info(f"小红书文字笔记: {text[:30]}")
|
|||
|
|
return text, []
|
|||
|
|
|
|||
|
|
file_name = _build_file_name(nickname, text, "笔记")
|
|||
|
|
file_paths = await _download_images(images, file_name)
|
|||
|
|
logger.info(f"小红书图文笔记: 作者={nickname}, 图片={len(images)} 张")
|
|||
|
|
return text, file_paths
|
|||
|
|
|
|||
|
|
|
|||
|
|
def _build_file_name(nickname: str, title: str, kind: str) -> str:
|
|||
|
|
"""构建文件名 stem: {作者}_{标题}_{类型}_{时间}"""
|
|||
|
|
slug_nickname = slugify(nickname)
|
|||
|
|
slug_title = slugify(title or "", max_length=15)
|
|||
|
|
if not slug_title:
|
|||
|
|
slug_title = datetime.now().strftime("%H%M%S")
|
|||
|
|
time_suffix = datetime.now().strftime("%H%M%S")
|
|||
|
|
return f"{slug_nickname}_{slug_title}_{kind}_{time_suffix}"
|
|||
|
|
|
|||
|
|
|
|||
|
|
async def _download_images(image_urls: list[str], file_name: str) -> list[Path]:
|
|||
|
|
"""并发下载图片(复用抖音图文的下载流程)"""
|
|||
|
|
import httpx
|
|||
|
|
|
|||
|
|
from .douyin_api import _process_note_with_parsed
|
|||
|
|
|
|||
|
|
tmp_root = get_temp_root("xiaohongshu")
|
|||
|
|
headers = {
|
|||
|
|
"Referer": REDNOTE_REFERER,
|
|||
|
|
"User-Agent": REDNOTE_UA,
|
|||
|
|
}
|
|||
|
|
return await _process_note_with_parsed(
|
|||
|
|
[[u] for u in image_urls], None, tmp_root, file_name, headers
|
|||
|
|
)
|
|||
|
|
|
|||
|
|
|
|||
|
|
async def _download_video(video_url: str, file_name: str) -> Path:
|
|||
|
|
"""流式下载视频
|
|||
|
|
|
|||
|
|
注意:sns-video-bd(无水印原片)不带 Referer 或带 xiaohongshu.com
|
|||
|
|
均可,但带 rednote.com Referer 会 403,因此不设 Referer。
|
|||
|
|
"""
|
|||
|
|
import httpx
|
|||
|
|
|
|||
|
|
tmp_root = get_temp_root("xiaohongshu")
|
|||
|
|
output_path = tmp_root / f"{file_name}.mp4"
|
|||
|
|
headers = {"User-Agent": REDNOTE_UA}
|
|||
|
|
async with httpx.AsyncClient(headers=headers, timeout=300) as client:
|
|||
|
|
async with client.stream("GET", video_url) as resp:
|
|||
|
|
resp.raise_for_status()
|
|||
|
|
with open(output_path, "wb") as f:
|
|||
|
|
async for chunk in resp.aiter_bytes(8192):
|
|||
|
|
f.write(chunk)
|
|||
|
|
logger.info(f"小红书视频下载完成: {output_path}")
|
|||
|
|
return output_path
|
|||
|
|
|
|||
|
|
|
|||
|
|
async def _fetch_page(url: str, headers: dict) -> str:
|
|||
|
|
import httpx
|
|||
|
|
|
|||
|
|
async with httpx.AsyncClient(
|
|||
|
|
headers=headers, timeout=20, follow_redirects=True
|
|||
|
|
) as client:
|
|||
|
|
resp = await client.get(url)
|
|||
|
|
if resp.status_code >= 400:
|
|||
|
|
raise ContentFetchError(f"小红书页面请求失败: status={resp.status_code}")
|
|||
|
|
return resp.text
|
|||
|
|
|
|||
|
|
|
|||
|
|
async def _resolve_short_link(url: str) -> Optional[str]:
|
|||
|
|
"""xhslink 短链重定向(最多 3 跳取最终 URL,重定向 URL 带 xsec_token)
|
|||
|
|
|
|||
|
|
借鉴 nonebot-plugin-parser:xhslink 用移动端 headers 请求
|
|||
|
|
(origin / x-requested-with 等),避免被当作非 App 来源拒绝。
|
|||
|
|
xhslink.cn 可能先跳到 xhslink.com 再跳小红书,需循环取跳。
|
|||
|
|
"""
|
|||
|
|
import httpx
|
|||
|
|
|
|||
|
|
headers = {
|
|||
|
|
"User-Agent": REDNOTE_UA,
|
|||
|
|
"Referer": REDNOTE_REFERER,
|
|||
|
|
"origin": "https://www.xiaohongshu.com",
|
|||
|
|
"x-requested-with": "XMLHttpRequest",
|
|||
|
|
}
|
|||
|
|
try:
|
|||
|
|
async with httpx.AsyncClient(
|
|||
|
|
headers=headers, follow_redirects=False, timeout=10
|
|||
|
|
) as client:
|
|||
|
|
current = url
|
|||
|
|
for _ in range(3):
|
|||
|
|
resp = await client.get(current)
|
|||
|
|
if resp.status_code >= 400:
|
|||
|
|
return None
|
|||
|
|
location = resp.headers.get("Location")
|
|||
|
|
if not location:
|
|||
|
|
return str(resp.url)
|
|||
|
|
# Location 可能是相对路径(如 /explore?...),需拼上当前 URL
|
|||
|
|
current = urllib.parse.urljoin(current, location)
|
|||
|
|
return current
|
|||
|
|
except Exception:
|
|||
|
|
logger.warning(f"xhslink 重定向失败: {url}")
|
|||
|
|
return None
|