feat(video-analysis): 群策略 v3 / 群文件投递通道 / Web 管理页
- policy.py:per-group 正交策略(自动解析 / 自动策略 / 禁用策略 / 存储 A·B·C /
公网 / 链接 / 群文件 + 平台限定),list.json v1/v2 → v3 自动迁移,
写入统一走 PolicyStore(加锁 + .tmp 原子替换 + 字段归一)
- 群文件并行通道 group_file.py:打包 zip(可选 pyzipper AES-256)后优先走 S3 预签名、
本地直传兜底;设了密码但 pyzipper 不可用就放弃上传,不退化成明文
- list_proc.py 收敛到「视频策略」统一入口,权限判定改走 policy
- Web 管理页 /hub/video_analysis(群策略 + 链接解析面板)与 services/web_jobs.py
(只复用纯函数层,Web 上下文不发消息;内存任务表 + 并发闸门 + 超时)
- 媒体命名统一到 utils.py({作者}_{作者id}/{作品名}[_短码]),cleanup 回收空目录
- 测试:policy / 命名 / 群文件 / web_jobs 四组
顺带 pyproject 的 pytest 加 testpaths=tests(避免收进 debug/ 下的调试脚本)。
Co-Authored-By: Claude Code <noreply@anthropic.com>
This commit is contained in:
@@ -14,15 +14,19 @@
|
||||
"""
|
||||
|
||||
import re
|
||||
import tempfile
|
||||
from datetime import datetime
|
||||
from pathlib import Path
|
||||
from typing import Optional, Union
|
||||
|
||||
from nonebot import logger
|
||||
|
||||
from ...models import ContentFetchError
|
||||
from ...utils import get_data_dir, parse_netscape_cookies, slugify
|
||||
from ...utils import (
|
||||
build_author_dir,
|
||||
build_work_stem,
|
||||
get_data_dir,
|
||||
get_temp_root,
|
||||
parse_netscape_cookies,
|
||||
)
|
||||
|
||||
DATA_DIR = get_data_dir()
|
||||
|
||||
@@ -81,10 +85,10 @@ async def _parse(url: str):
|
||||
|
||||
# 4. 文章动态 → 转 opus
|
||||
if await dynamic.is_article():
|
||||
return await _parse_opus(dynamic.turn_to_opus(), "文章")
|
||||
return await _parse_opus(dynamic.turn_to_opus(), "文章", url)
|
||||
|
||||
info = await dynamic.get_info()
|
||||
return await _parse_dynamic_info(info)
|
||||
return await _parse_dynamic_info(info, url)
|
||||
|
||||
|
||||
async def _parse_article(read_id: int) -> tuple[str, Union[Path, list[Path]]]:
|
||||
@@ -94,10 +98,12 @@ async def _parse_article(read_id: int) -> tuple[str, Union[Path, list[Path]]]:
|
||||
# 文章接口对匿名请求风控更严(-509),必须带凭证
|
||||
article = Article(read_id, _build_credential())
|
||||
opus = await article.turn_to_opus()
|
||||
return await _parse_opus(opus, "文章")
|
||||
return await _parse_opus(opus, "文章", f"cv{read_id}")
|
||||
|
||||
|
||||
async def _parse_opus(opus, kind: str) -> tuple[str, Union[Path, list[Path]]]:
|
||||
async def _parse_opus(
|
||||
opus, kind: str, source: str = ""
|
||||
) -> tuple[str, Union[Path, list[Path]]]:
|
||||
"""图文动态/专栏解析(opus 接口返回 dict,直接访问)"""
|
||||
info = await opus.get_info()
|
||||
item = info.get("item") or {}
|
||||
@@ -107,10 +113,12 @@ async def _parse_opus(opus, kind: str) -> tuple[str, Union[Path, list[Path]]]:
|
||||
images: list[str] = []
|
||||
texts: list[str] = []
|
||||
author = ""
|
||||
author_id = ""
|
||||
for module in item.get("modules") or []:
|
||||
if module.get("module_type") == "MODULE_TYPE_AUTHOR":
|
||||
author_info = module.get("module_author") or {}
|
||||
author = author_info.get("name", "")
|
||||
author_id = str(author_info.get("mid") or "")
|
||||
elif module.get("module_type") == "MODULE_TYPE_CONTENT":
|
||||
content = module.get("module_content") or {}
|
||||
for para in content.get("paragraphs") or []:
|
||||
@@ -127,16 +135,20 @@ async def _parse_opus(opus, kind: str) -> tuple[str, Union[Path, list[Path]]]:
|
||||
if not images:
|
||||
return text or f"B站{kind}", []
|
||||
|
||||
file_name = _build_file_name(author, text or f"B站{kind}", kind)
|
||||
file_paths = await _download_images(images, file_name)
|
||||
rel_stem = _build_rel_stem(author, author_id, text or f"B站{kind}", source)
|
||||
file_paths = await _download_images(images, rel_stem)
|
||||
return text, file_paths
|
||||
|
||||
|
||||
async def _parse_dynamic_info(info: dict) -> tuple[str, Union[Path, list[Path]]]:
|
||||
async def _parse_dynamic_info(
|
||||
info: dict, source: str = ""
|
||||
) -> tuple[str, Union[Path, list[Path]]]:
|
||||
"""动态解析(图文 / 视频 / 纯文字)"""
|
||||
item = info.get("item") or {}
|
||||
modules = item.get("modules") or {}
|
||||
author = ((modules.get("module_author") or {}).get("name")) or "B站用户"
|
||||
module_author = modules.get("module_author") or {}
|
||||
author = module_author.get("name") or "B站用户"
|
||||
author_id = str(module_author.get("mid") or "")
|
||||
module_dynamic = modules.get("module_dynamic") or {}
|
||||
major = module_dynamic.get("major") or {}
|
||||
major_type = major.get("type", "")
|
||||
@@ -151,7 +163,13 @@ async def _parse_dynamic_info(info: dict) -> tuple[str, Union[Path, list[Path]]]
|
||||
|
||||
title = archive.get("title") or desc or "B站视频动态"
|
||||
logger.info(f"B站视频动态: bvid={bvid} 标题={title[:40]}")
|
||||
video_path = await download_video(f"https://www.bilibili.com/video/{bvid}")
|
||||
# 动态数据里已有 up 的昵称/mid,传下去才能和同一位 up 的图文
|
||||
# 落在同一个作者目录(否则要赌 yt-dlp 返回的 id 对得上)
|
||||
video_path, _ = await download_video(
|
||||
f"https://www.bilibili.com/video/{bvid}",
|
||||
author=author,
|
||||
author_id=author_id,
|
||||
)
|
||||
if video_path:
|
||||
return title, video_path
|
||||
raise ContentFetchError(f"视频动态下载失败: {bvid}")
|
||||
@@ -173,8 +191,8 @@ async def _parse_dynamic_info(info: dict) -> tuple[str, Union[Path, list[Path]]]
|
||||
|
||||
images = [u for u in images if u]
|
||||
if images:
|
||||
file_name = _build_file_name(author, title, "动态")
|
||||
file_paths = await _download_images(images, file_name)
|
||||
rel_stem = _build_rel_stem(author, author_id, title, source)
|
||||
file_paths = await _download_images(images, rel_stem)
|
||||
logger.info(f"B站图文动态: 作者={author}, 标题={title[:40]}, 图片={len(images)} 张")
|
||||
return title, file_paths
|
||||
|
||||
@@ -197,14 +215,17 @@ def _extract_text(nodes: list) -> str:
|
||||
return "".join(parts)
|
||||
|
||||
|
||||
def _build_file_name(author: str, title: str, kind: str) -> str:
|
||||
"""构建文件名 stem: {作者}_{标题}_{类型}_{时间}"""
|
||||
slug_author = slugify(author)
|
||||
slug_title = slugify(title or "", max_length=15)
|
||||
if not slug_title:
|
||||
slug_title = datetime.now().strftime("%H%M%S")
|
||||
time_suffix = datetime.now().strftime("%H%M%S")
|
||||
return f"{slug_author}_{slug_title}_{kind}_{time_suffix}"
|
||||
def _build_rel_stem(
|
||||
author: str, author_id: str, title: str, source: str = ""
|
||||
) -> str:
|
||||
"""相对平台根的路径词干:`{作者}_{mid}/{作品名}`
|
||||
|
||||
昵称/mid 都拿不到时用 source(动态链接)当来源码(见 utils.build_author_dir)。
|
||||
"""
|
||||
return (
|
||||
f"{build_author_dir(author, author_id, source=source)}"
|
||||
f"/{build_work_stem(title)}"
|
||||
)
|
||||
|
||||
|
||||
def _build_credential():
|
||||
@@ -227,20 +248,21 @@ def _build_credential():
|
||||
)
|
||||
|
||||
|
||||
async def _download_images(image_urls: list[str], file_name: str) -> list[Path]:
|
||||
"""并发下载图片(复用抖音图文的下载流程)"""
|
||||
import httpx
|
||||
async def _download_images(image_urls: list[str], rel_stem: str) -> list[Path]:
|
||||
"""并发下载图片(复用抖音图文的下载流程)
|
||||
|
||||
落 hexi/data/temp/bilibili(原先落系统 temp,cleanup 扫不到、永不清理)。
|
||||
"""
|
||||
|
||||
from .douyin_api import _process_note_with_parsed
|
||||
|
||||
tmp_root = Path(tempfile.gettempdir()) / "bilibili"
|
||||
tmp_root.mkdir(parents=True, exist_ok=True)
|
||||
tmp_root = get_temp_root("bilibili")
|
||||
headers = {
|
||||
"Referer": BILI_REFERER,
|
||||
"User-Agent": BILI_UA,
|
||||
}
|
||||
return await _process_note_with_parsed(
|
||||
[[u] for u in image_urls], None, tmp_root, file_name, headers
|
||||
[[u] for u in image_urls], None, tmp_root, rel_stem, headers
|
||||
)
|
||||
|
||||
|
||||
|
||||
@@ -10,11 +10,9 @@ from nonebot import logger
|
||||
from playwright.async_api import async_playwright
|
||||
|
||||
from ...models import DouyinFetchError
|
||||
from ...utils import ensure_unique_path, get_temp_root
|
||||
from ...utils import get_temp_root, unique_media_path
|
||||
from .douyin_parser import (
|
||||
ParsedDouyinContent,
|
||||
extract_trailing_digits,
|
||||
is_animated_note,
|
||||
parse_animated_note_videos,
|
||||
parse_douyin_response,
|
||||
parse_note_images,
|
||||
@@ -194,9 +192,9 @@ async def fetch_douyin_content(
|
||||
|
||||
if parsed.media_type == "视频":
|
||||
content = await _process_video(
|
||||
api_response, tmp_root, parsed.file_name, aweme_id, headers
|
||||
api_response, tmp_root, parsed.rel_stem, aweme_id, headers
|
||||
)
|
||||
return parsed.file_name, content
|
||||
return parsed.raw_title, content
|
||||
|
||||
elif parsed.media_type == "图片":
|
||||
# 先解析 images 列表,区分纯动图和图文/图+视频
|
||||
@@ -209,15 +207,15 @@ async def fetch_douyin_content(
|
||||
images_urls,
|
||||
video_url,
|
||||
tmp_root,
|
||||
parsed.file_name,
|
||||
parsed.rel_stem,
|
||||
headers,
|
||||
)
|
||||
else:
|
||||
# 纯动图(所有项都是视频)
|
||||
content = await _process_animated_note(
|
||||
api_response, tmp_root, parsed.file_name, headers
|
||||
api_response, tmp_root, parsed.rel_stem, headers
|
||||
)
|
||||
return parsed.file_name, content
|
||||
return parsed.raw_title, content
|
||||
|
||||
return None, None
|
||||
|
||||
@@ -228,11 +226,14 @@ async def fetch_douyin_content(
|
||||
async def _process_video(
|
||||
api_response: dict,
|
||||
tmp_root: Path,
|
||||
file_name: str,
|
||||
rel_stem: str,
|
||||
aweme_id: str,
|
||||
headers: Dict[str, str],
|
||||
) -> Path:
|
||||
"""处理视频内容,返回本地文件路径"""
|
||||
"""处理视频内容,返回本地文件路径
|
||||
|
||||
rel_stem 是相对平台根的路径词干 `{作者目录}/{作品名}`(见 ParsedDouyinContent.rel_stem)。
|
||||
"""
|
||||
groups = parse_video_urls(api_response)
|
||||
|
||||
best_group = None
|
||||
@@ -251,7 +252,7 @@ async def _process_video(
|
||||
best = max(full, key=lambda x: x["br"])
|
||||
logger.info(f"选择码率: {best['br']} - {best['url'][:60]}...")
|
||||
|
||||
output_path = ensure_unique_path(tmp_root / f"{file_name}.mp4")
|
||||
output_path = unique_media_path(tmp_root / f"{rel_stem}.mp4")
|
||||
async with httpx.AsyncClient(headers=headers) as client:
|
||||
async with client.stream("GET", best["url"]) as resp:
|
||||
resp.raise_for_status()
|
||||
@@ -271,9 +272,10 @@ async def _process_video(
|
||||
logger.info(f"选择视频码率: {video['br']}")
|
||||
logger.info(f"选择音频码率: {audio['br']}")
|
||||
|
||||
video_path = tmp_root / f"{file_name}_v.mp4"
|
||||
audio_path = tmp_root / f"{file_name}_a.mp4"
|
||||
output_path = ensure_unique_path(tmp_root / f"{file_name}.mp4")
|
||||
# 先定下产物名(顺带建好作者目录),分轨中间文件与产物同目录
|
||||
output_path = unique_media_path(tmp_root / f"{rel_stem}.mp4")
|
||||
video_path = output_path.parent / f"{output_path.stem}_v.mp4"
|
||||
audio_path = output_path.parent / f"{output_path.stem}_a.mp4"
|
||||
|
||||
async with httpx.AsyncClient(headers=headers) as client:
|
||||
logger.info("开始下载视频...")
|
||||
@@ -291,7 +293,9 @@ async def _process_video(
|
||||
f.write(chunk)
|
||||
|
||||
logger.info("合并视频和音频...")
|
||||
merge_video_audio(video_path, audio_path, output_path)
|
||||
# ffmpeg 是同步子进程,直接 await 不了:不丢线程池会卡住整个事件循环
|
||||
# (合并期间 Web 轮询、群消息全都停摆)
|
||||
await asyncio.to_thread(merge_video_audio, video_path, audio_path, output_path)
|
||||
video_path.unlink()
|
||||
audio_path.unlink()
|
||||
|
||||
@@ -306,11 +310,14 @@ async def _process_note_with_parsed(
|
||||
images_urls: List[List[str]],
|
||||
video_url: Optional[str],
|
||||
tmp_root: Path,
|
||||
file_name: str,
|
||||
rel_stem: str,
|
||||
headers: Dict[str, str],
|
||||
) -> List[Path]:
|
||||
"""根据已解析的图片/视频 URL 列表,并行下载"""
|
||||
note_dir = ensure_unique_path(tmp_root / file_name)
|
||||
"""根据已解析的图片/视频 URL 列表,并行下载
|
||||
|
||||
一个作品一个目录:`{平台根}/{作者目录}/{作品名}[_{短码}]/001.jpg…`
|
||||
"""
|
||||
note_dir = unique_media_path(tmp_root / rel_stem)
|
||||
note_dir.mkdir(parents=True, exist_ok=True)
|
||||
logger.info(f"图文保存目录: {note_dir}")
|
||||
|
||||
@@ -353,7 +360,7 @@ async def _process_note(
|
||||
api_response: dict,
|
||||
api_response_favorite: dict,
|
||||
tmp_root: Path,
|
||||
file_name: str,
|
||||
rel_stem: str,
|
||||
aweme_id: str,
|
||||
headers: Dict[str, str],
|
||||
) -> List[Path]:
|
||||
@@ -367,7 +374,7 @@ async def _process_note(
|
||||
raise DouyinFetchError("未找到图文链接")
|
||||
|
||||
return await _process_note_with_parsed(
|
||||
images_urls, video_url, tmp_root, file_name, headers
|
||||
images_urls, video_url, tmp_root, rel_stem, headers
|
||||
)
|
||||
|
||||
|
||||
@@ -377,14 +384,14 @@ async def _process_note(
|
||||
async def _process_animated_note(
|
||||
api_response: dict,
|
||||
tmp_root: Path,
|
||||
file_name: str,
|
||||
rel_stem: str,
|
||||
headers: Dict[str, str],
|
||||
) -> List[Path]:
|
||||
"""处理动图内容(media_type=42),并行下载所有无声 mp4 视频"""
|
||||
video_urls = parse_animated_note_videos(api_response)
|
||||
logger.info(f"解析到的动图视频链接: {video_urls}")
|
||||
|
||||
note_dir = ensure_unique_path(tmp_root / file_name)
|
||||
note_dir = unique_media_path(tmp_root / rel_stem)
|
||||
note_dir.mkdir(parents=True, exist_ok=True)
|
||||
logger.info(f"动图保存目录: {note_dir}")
|
||||
|
||||
|
||||
@@ -4,13 +4,12 @@ import json
|
||||
import re
|
||||
import urllib.parse
|
||||
from dataclasses import dataclass
|
||||
from datetime import datetime
|
||||
from typing import Dict, List, Optional
|
||||
|
||||
from nonebot import logger
|
||||
|
||||
from ...models import DouyinFetchError
|
||||
from ...utils import slugify
|
||||
from ...utils import build_author_dir, build_work_stem
|
||||
|
||||
|
||||
# ============================= 数据结构 =============================
|
||||
@@ -22,7 +21,13 @@ class ParsedDouyinContent:
|
||||
raw_title: str
|
||||
raw_nickname: str
|
||||
media_type: str # "视频" | "图片"
|
||||
file_name: str # 构建好的文件名 stem
|
||||
file_name: str # 作品名 stem(落盘文件名/多图子目录名)
|
||||
author_dir: str = "" # 作者目录 `{昵称}_{uid}`(落盘与 S3 key 的首层)
|
||||
|
||||
@property
|
||||
def rel_stem(self) -> str:
|
||||
"""相对平台根目录的路径词干:`{作者目录}/{作品名}`"""
|
||||
return f"{self.author_dir}/{self.file_name}" if self.author_dir else self.file_name
|
||||
|
||||
|
||||
# ============================= URL 工具 =============================
|
||||
@@ -109,6 +114,29 @@ def extract_author_nickname(api_response: dict) -> str:
|
||||
return nickname
|
||||
|
||||
|
||||
def extract_author_id(api_response: dict) -> str:
|
||||
"""从 API 响应提取作者稳定 id,优先级: uid > unique_id(抖音号) > sec_uid
|
||||
|
||||
SSR 路径的 `aweme_list[0].author.uid` 与 API 路径的
|
||||
`aweme_detail.author.uid` 走同一套取值;都拿不到返回空串,
|
||||
作者目录退化成只用昵称(见 utils.build_author_dir)。
|
||||
"""
|
||||
aweme_detail = api_response.get("aweme_detail") or {}
|
||||
author = aweme_detail.get("author") or {}
|
||||
if not author:
|
||||
aweme_list = api_response.get("aweme_list") or []
|
||||
if aweme_list and isinstance(aweme_list, list):
|
||||
author = (aweme_list[0] or {}).get("author") or {}
|
||||
|
||||
for key in ("uid", "unique_id", "sec_uid"):
|
||||
value = str(author.get(key) or "").strip()
|
||||
if value and value != "0":
|
||||
logger.info(f"RAW作者id({key}):{value}")
|
||||
return value
|
||||
logger.info("未获取到作者 id,作者目录只用昵称")
|
||||
return ""
|
||||
|
||||
|
||||
def detect_media_type(referer_url: str | None) -> str | None:
|
||||
"""根据页面 URL 检测媒体类型(视频/图片)"""
|
||||
if referer_url is None:
|
||||
@@ -326,36 +354,6 @@ def parse_ssr_page(html: str) -> Optional[dict]:
|
||||
return {"aweme_list": [item]}
|
||||
|
||||
|
||||
# ============================= 文件名构建 =============================
|
||||
|
||||
|
||||
def build_file_name(
|
||||
raw_title: str,
|
||||
raw_nickname: str,
|
||||
media_type: str,
|
||||
) -> str:
|
||||
"""
|
||||
构建文件名 stem,格式: {作者}_{标题}_{类型}_{时间戳}
|
||||
|
||||
昵称不限长,标题最多 15 字符(slugify 后),末尾 HHMMSS 防覆盖。
|
||||
"""
|
||||
slug_nickname = slugify(raw_nickname)
|
||||
|
||||
if raw_title:
|
||||
slug_title = slugify(raw_title, max_length=15)
|
||||
else:
|
||||
slug_title = ""
|
||||
|
||||
if not slug_title:
|
||||
slug_title = datetime.now().strftime("%H%M%S")
|
||||
logger.info(f"标题为空,使用短时间戳: {slug_title}")
|
||||
|
||||
slug_type = slugify(media_type)
|
||||
time_suffix = datetime.now().strftime("%H%M%S")
|
||||
|
||||
return f"{slug_nickname}_{slug_title}_{slug_type}_{time_suffix}"
|
||||
|
||||
|
||||
# ============================= 动图检测 =============================
|
||||
|
||||
|
||||
@@ -410,11 +408,21 @@ def parse_douyin_response(
|
||||
|
||||
raw_title = extract_title_from_api(api_response)
|
||||
raw_nickname = extract_author_nickname(api_response)
|
||||
file_name = build_file_name(raw_title, raw_nickname, media_type)
|
||||
raw_author_id = extract_author_id(api_response)
|
||||
# 昵称/uid 都拿不到时的来源码:优先作品链接,其次响应里的作品 id
|
||||
aweme_id = str(
|
||||
(api_response.get("aweme_detail") or {}).get("aweme_id")
|
||||
or ((api_response.get("aweme_list") or [{}])[0] or {}).get("aweme_id")
|
||||
or ""
|
||||
)
|
||||
author_dir = build_author_dir(
|
||||
raw_nickname, raw_author_id, source=referer_url or aweme_id
|
||||
)
|
||||
file_name = build_work_stem(raw_title)
|
||||
|
||||
logger.info(
|
||||
f"内容标题: {raw_title}, 作者: {raw_nickname}, "
|
||||
f"类型: {media_type}, 文件名: {file_name}"
|
||||
f"内容标题: {raw_title}, 作者: {raw_nickname}({raw_author_id}), "
|
||||
f"类型: {media_type}, 落盘路径: {author_dir}/{file_name}"
|
||||
)
|
||||
|
||||
return ParsedDouyinContent(
|
||||
@@ -422,4 +430,5 @@ def parse_douyin_response(
|
||||
raw_nickname=raw_nickname,
|
||||
media_type=media_type,
|
||||
file_name=file_name,
|
||||
author_dir=author_dir,
|
||||
)
|
||||
|
||||
@@ -85,13 +85,13 @@ async def fetch_douyin_note_ssr(
|
||||
images_urls, video_url = parse_note_images(api_response, None, vid)
|
||||
if images_urls:
|
||||
file_paths = await _process_note_with_parsed(
|
||||
images_urls, video_url, tmp_root, parsed.file_name, headers
|
||||
images_urls, video_url, tmp_root, parsed.rel_stem, headers
|
||||
)
|
||||
else:
|
||||
file_paths = await _process_animated_note(
|
||||
api_response, tmp_root, parsed.file_name, headers
|
||||
api_response, tmp_root, parsed.rel_stem, headers
|
||||
)
|
||||
return parsed.file_name, file_paths
|
||||
return parsed.raw_title, file_paths
|
||||
|
||||
|
||||
async def _fetch_ssr_api_response(
|
||||
|
||||
@@ -19,16 +19,21 @@
|
||||
import asyncio
|
||||
import json
|
||||
import re
|
||||
import tempfile
|
||||
import urllib.parse
|
||||
from datetime import datetime
|
||||
from pathlib import Path
|
||||
from typing import Optional, Union
|
||||
|
||||
from nonebot import logger
|
||||
|
||||
from ...models import ContentFetchError
|
||||
from ...utils import get_data_dir, get_temp_root, parse_netscape_cookies, slugify
|
||||
from ...utils import (
|
||||
build_author_dir,
|
||||
build_work_stem,
|
||||
get_data_dir,
|
||||
get_temp_root,
|
||||
parse_netscape_cookies,
|
||||
unique_media_path,
|
||||
)
|
||||
|
||||
DATA_DIR = get_data_dir()
|
||||
|
||||
@@ -214,6 +219,12 @@ async def _build_result(note: dict) -> tuple[str, Union[Path, list[Path]]]:
|
||||
desc = note.get("desc") or ""
|
||||
nickname = ((note.get("user") or {}).get("nickname")) or "小红书用户"
|
||||
text = title or desc or "小红书笔记"
|
||||
rel_stem = _build_rel_stem(
|
||||
nickname,
|
||||
_extract_author_id(note),
|
||||
text,
|
||||
source=str(note.get("noteId") or note.get("id") or ""),
|
||||
)
|
||||
|
||||
# 1. 视频笔记 → 无水印原片优先
|
||||
if note.get("type") == "video" and note.get("video"):
|
||||
@@ -223,8 +234,7 @@ async def _build_result(note: dict) -> tuple[str, Union[Path, list[Path]]]:
|
||||
if okey:
|
||||
video_url = f"https://sns-video-bd.xhscdn.com/{okey}"
|
||||
logger.info(f"小红书视频: 无水印原片 originVideoKey={okey[:30]}...")
|
||||
file_name = _build_file_name(nickname, text, "视频")
|
||||
video_path = await _download_video(video_url, file_name)
|
||||
video_path = await _download_video(video_url, rel_stem)
|
||||
return text, video_path
|
||||
|
||||
# 1b. 无 originVideoKey(国内站数据)→ 从 stream 分组选无水印原片
|
||||
@@ -251,8 +261,7 @@ async def _build_result(note: dict) -> tuple[str, Union[Path, list[Path]]]:
|
||||
f"{best.get('width')}x{best.get('height')} {best.get('fps')}fps "
|
||||
f"size={best.get('size')} duration={duration}ms"
|
||||
)
|
||||
file_name = _build_file_name(nickname, text, "视频")
|
||||
video_path = await _download_video(video_url, file_name)
|
||||
video_path = await _download_video(video_url, rel_stem)
|
||||
return text, video_path
|
||||
raise ContentFetchError("小红书视频流解析失败")
|
||||
|
||||
@@ -266,25 +275,36 @@ async def _build_result(note: dict) -> tuple[str, Union[Path, list[Path]]]:
|
||||
logger.info(f"小红书文字笔记: {text[:30]}")
|
||||
return text, []
|
||||
|
||||
file_name = _build_file_name(nickname, text, "笔记")
|
||||
file_paths = await _download_images(images, file_name)
|
||||
file_paths = await _download_images(images, rel_stem)
|
||||
logger.info(f"小红书图文笔记: 作者={nickname}, 图片={len(images)} 张")
|
||||
return text, file_paths
|
||||
|
||||
|
||||
def _build_file_name(nickname: str, title: str, kind: str) -> str:
|
||||
"""构建文件名 stem: {作者}_{标题}_{类型}_{时间}"""
|
||||
slug_nickname = slugify(nickname)
|
||||
slug_title = slugify(title or "", max_length=15)
|
||||
if not slug_title:
|
||||
slug_title = datetime.now().strftime("%H%M%S")
|
||||
time_suffix = datetime.now().strftime("%H%M%S")
|
||||
return f"{slug_nickname}_{slug_title}_{kind}_{time_suffix}"
|
||||
def _extract_author_id(note: dict) -> str:
|
||||
"""小红书作者稳定 id(页面数据字段未实测,逐个兜底;取不到返回空串)"""
|
||||
user = note.get("user") or {}
|
||||
for key in ("userId", "user_id", "id"):
|
||||
value = user.get(key)
|
||||
if isinstance(value, (str, int)) and str(value).strip() not in ("", "0"):
|
||||
return str(value).strip()
|
||||
return ""
|
||||
|
||||
|
||||
async def _download_images(image_urls: list[str], file_name: str) -> list[Path]:
|
||||
def _build_rel_stem(
|
||||
nickname: str, author_id: str, title: str, source: str = ""
|
||||
) -> str:
|
||||
"""相对平台根的路径词干:`{作者}_{userId}/{作品名}`
|
||||
|
||||
昵称/作者 id 都拿不到时用 source(笔记 id)当来源码(见 utils.build_author_dir)。
|
||||
"""
|
||||
return (
|
||||
f"{build_author_dir(nickname, author_id, source=source)}"
|
||||
f"/{build_work_stem(title)}"
|
||||
)
|
||||
|
||||
|
||||
async def _download_images(image_urls: list[str], rel_stem: str) -> list[Path]:
|
||||
"""并发下载图片(复用抖音图文的下载流程)"""
|
||||
import httpx
|
||||
|
||||
from .douyin_api import _process_note_with_parsed
|
||||
|
||||
@@ -294,11 +314,11 @@ async def _download_images(image_urls: list[str], file_name: str) -> list[Path]:
|
||||
"User-Agent": REDNOTE_UA,
|
||||
}
|
||||
return await _process_note_with_parsed(
|
||||
[[u] for u in image_urls], None, tmp_root, file_name, headers
|
||||
[[u] for u in image_urls], None, tmp_root, rel_stem, headers
|
||||
)
|
||||
|
||||
|
||||
async def _download_video(video_url: str, file_name: str) -> Path:
|
||||
async def _download_video(video_url: str, rel_stem: str) -> Path:
|
||||
"""流式下载视频
|
||||
|
||||
注意:sns-video-bd(无水印原片)不带 Referer 或带 xiaohongshu.com
|
||||
@@ -307,7 +327,7 @@ async def _download_video(video_url: str, file_name: str) -> Path:
|
||||
import httpx
|
||||
|
||||
tmp_root = get_temp_root("xiaohongshu")
|
||||
output_path = tmp_root / f"{file_name}.mp4"
|
||||
output_path = unique_media_path(tmp_root / f"{rel_stem}.mp4")
|
||||
headers = {"User-Agent": REDNOTE_UA}
|
||||
async with httpx.AsyncClient(headers=headers, timeout=300) as client:
|
||||
async with client.stream("GET", video_url) as resp:
|
||||
|
||||
@@ -3,9 +3,9 @@
|
||||
import asyncio
|
||||
import os
|
||||
import re
|
||||
import shutil
|
||||
import sys
|
||||
import tempfile
|
||||
from datetime import datetime
|
||||
from pathlib import Path
|
||||
from typing import Optional
|
||||
|
||||
@@ -14,7 +14,14 @@ from nonebot import logger
|
||||
from yt_dlp import YoutubeDL
|
||||
from yt_dlp.utils import DownloadError
|
||||
|
||||
from ...utils import get_data_dir, get_temp_root, slugify, ensure_unique_path
|
||||
from ...utils import (
|
||||
build_author_dir,
|
||||
build_work_stem,
|
||||
get_data_dir,
|
||||
get_temp_root,
|
||||
slugify,
|
||||
unique_media_path,
|
||||
)
|
||||
|
||||
|
||||
def detect_platform(url: str) -> str:
|
||||
@@ -41,6 +48,16 @@ def extract_uploader(info: dict) -> Optional[str]:
|
||||
)
|
||||
|
||||
|
||||
def extract_uploader_id(info: dict) -> str:
|
||||
"""作者稳定 id:channel_id > uploader_id(@handle / B站 mid),拿不到返回空串
|
||||
|
||||
不用 `id`(那是作品 id,会把同一作者的作品拆到不同目录)。
|
||||
"""
|
||||
if not info:
|
||||
return ""
|
||||
return str(info.get("channel_id") or info.get("uploader_id") or "").strip()
|
||||
|
||||
|
||||
def get_ffmpeg_path() -> str:
|
||||
scripts_dir = os.path.dirname(sys.executable)
|
||||
ffmpeg_path = os.path.join(scripts_dir, "ffmpeg.exe")
|
||||
@@ -78,19 +95,36 @@ async def _retry_download(
|
||||
logger.error(
|
||||
f"yt-dlp 重试 {max_retries} 次后仍失败: {str(e)[:120]}"
|
||||
)
|
||||
except Exception as e:
|
||||
except Exception:
|
||||
# 非 DownloadError(如 OSError)不重试,直接抛出
|
||||
raise
|
||||
|
||||
raise last_error # type: ignore[misc]
|
||||
|
||||
|
||||
async def download_video(url: str) -> Optional[Path]:
|
||||
"""下载视频,支持直链和 yt-dlp"""
|
||||
async def download_video(
|
||||
url: str,
|
||||
*,
|
||||
author: Optional[str] = None,
|
||||
author_id: Optional[str] = None,
|
||||
) -> tuple[Optional[Path], str]:
|
||||
"""下载视频,支持直链和 yt-dlp
|
||||
|
||||
author / author_id 可由调用方覆盖(如 B站 视频动态已从动态数据里拿到 mid,
|
||||
传进来才能和同一位 up 的图文落在同一个作者目录)。
|
||||
|
||||
Returns:
|
||||
(本地文件, 作品标题) — 失败时 (None, "")。
|
||||
落盘位置:`temp/{平台}/{作者}_{作者id}/{作品名}[_{短码}].ext`
|
||||
(直链拿不到作者信息,统一进 `未知作者/`)。
|
||||
"""
|
||||
platform = detect_platform(url)
|
||||
temp_root = get_temp_root(platform)
|
||||
|
||||
# ---------- 1. 直链探测 ----------
|
||||
# 不含 m3u8:HLS 播放列表直下只会得到一个文本文件,交给 yt-dlp 处理
|
||||
direct_media_ext = re.search(
|
||||
r"\.(mp4|m3u8|ts|webm|mov|flv)(?:$|\?)", url, re.IGNORECASE
|
||||
r"\.(mp4|ts|webm|mov|flv)(?:$|\?)", url, re.IGNORECASE
|
||||
)
|
||||
is_direct = bool(direct_media_ext)
|
||||
|
||||
@@ -105,39 +139,35 @@ async def download_video(url: str) -> Optional[Path]:
|
||||
is_direct = False
|
||||
|
||||
if is_direct:
|
||||
temp_dir = tempfile.mkdtemp(prefix="direct_ytcache_", dir=get_temp_root("ytcache"))
|
||||
ext = "mp4"
|
||||
m = re.search(r"\.([a-zA-Z0-9]{2,5})(?:$|\?)", url)
|
||||
if m and len(m.group(1)) <= 5:
|
||||
ext = m.group(1)
|
||||
|
||||
url_stem = Path(url.split("?")[0]).stem or "video"
|
||||
slug_stem = slugify(url_stem, max_length=15)
|
||||
if not slug_stem:
|
||||
slug_stem = datetime.now().strftime("%H%M%S")
|
||||
time_suffix = datetime.now().strftime("%H%M%S")
|
||||
new_name = f"{slug_stem}_视频_{time_suffix}.{ext}"
|
||||
filename = os.path.join(temp_dir, new_name)
|
||||
slug_stem = slugify(url_stem, max_length=15) or "视频"
|
||||
# 直链拿不到作者信息 → `未知作者_{来源短码}`(同一链接稳定、不同链接不撞)
|
||||
author_dir = build_author_dir(author, author_id, source=url)
|
||||
final_path = unique_media_path(temp_root / author_dir / f"{slug_stem}.{ext}")
|
||||
|
||||
try:
|
||||
async with AsyncClient(follow_redirects=True, timeout=300) as client:
|
||||
async with client.stream("GET", url) as resp:
|
||||
resp.raise_for_status()
|
||||
with open(filename, "wb") as fh:
|
||||
with open(final_path, "wb") as fh:
|
||||
async for chunk in resp.aiter_bytes(chunk_size=8192):
|
||||
fh.write(chunk)
|
||||
|
||||
final_path = ensure_unique_path(Path(filename))
|
||||
logger.info(f"直接下载完成: {final_path}")
|
||||
return final_path
|
||||
return final_path, ""
|
||||
except Exception:
|
||||
logger.exception("直接下载失败,回退 yt-dlp")
|
||||
if os.path.exists(filename):
|
||||
os.remove(filename)
|
||||
final_path.unlink(missing_ok=True)
|
||||
|
||||
# ---------- 2. yt-dlp 下载 ----------
|
||||
platform = detect_platform(url)
|
||||
temp_dir = tempfile.mkdtemp(prefix="ytcache_", dir=get_temp_root("ytcache"))
|
||||
# 先下到 scratch 目录(outtmpl 必须在拿到 info 之前给定),拿到 info 后再
|
||||
# 按作者归位到 temp/{平台}/{作者}_{作者id}/
|
||||
temp_dir = tempfile.mkdtemp(prefix="_dl_", dir=temp_root)
|
||||
output_path = os.path.join(temp_dir, "%(title).80s.%(ext)s")
|
||||
|
||||
base_opts = {
|
||||
@@ -188,7 +218,10 @@ async def download_video(url: str) -> Optional[Path]:
|
||||
"Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,image/webp,*/*;q=0.8",
|
||||
"Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
|
||||
}
|
||||
base_opts["extractor_args"] = {"twitter": {"api": ["syndication"]}}
|
||||
# 不要强制 twitter:api=syndication:该端点为未登录视角,对敏感/受限推文
|
||||
# 只返回 tombstone(无 mediaDetails),且会无条件覆盖已登录 GraphQL 的结果,
|
||||
# 表现为 "No video could be found in this tweet"。默认走 GraphQL + cookies,
|
||||
# 遇 429 yt-dlp 会自行回退 syndication。
|
||||
elif platform == "youtube":
|
||||
base_opts["http_headers"] = {
|
||||
"User-Agent": ua,
|
||||
@@ -206,54 +239,53 @@ async def download_video(url: str) -> Optional[Path]:
|
||||
info = await _retry_download(loop, url, base_opts)
|
||||
except Exception:
|
||||
logger.exception("yt-dlp 下载失败")
|
||||
# YouTube: cookies 可能触发 bot 检测导致只返回图片无视频格式
|
||||
# 回退无 cookie 模式重试
|
||||
if platform == "youtube" and "cookiefile" in base_opts:
|
||||
# YouTube: cookies 可能触发 bot 检测导致只返回图片无视频格式
|
||||
# 回退无 cookie 模式重试
|
||||
logger.info("YouTube 回退无 cookies 模式重试...")
|
||||
base_opts.pop("cookiefile", None)
|
||||
base_opts.pop("http_headers", None)
|
||||
# 清理失败残留
|
||||
for f in Path(temp_dir).glob("*.*"):
|
||||
try:
|
||||
f.unlink()
|
||||
except Exception:
|
||||
pass
|
||||
try:
|
||||
info = await _retry_download(loop, url, base_opts, max_retries=2)
|
||||
except Exception:
|
||||
logger.exception("yt-dlp 无 cookies 重试也失败")
|
||||
return None
|
||||
elif platform == "twitter":
|
||||
# X 登录态失效(auth_token 过期)时 GraphQL 会直接拒绝请求;
|
||||
# 退回未登录的 syndication 端点,公开推文仍可下载(敏感推文会失败)
|
||||
logger.info("Twitter 回退 syndication 端点重试...")
|
||||
base_opts["extractor_args"] = {"twitter": {"api": ["syndication"]}}
|
||||
else:
|
||||
return None
|
||||
return None, ""
|
||||
|
||||
# 清理失败残留
|
||||
for f in Path(temp_dir).glob("*.*"):
|
||||
try:
|
||||
f.unlink()
|
||||
except Exception:
|
||||
pass
|
||||
try:
|
||||
info = await _retry_download(loop, url, base_opts, max_retries=2)
|
||||
except Exception:
|
||||
logger.exception("yt-dlp 回退重试也失败")
|
||||
return None, ""
|
||||
|
||||
files = list(Path(temp_dir).glob("*.*"))
|
||||
if not files:
|
||||
return None
|
||||
return None, ""
|
||||
|
||||
original_file = files[0]
|
||||
|
||||
# 构建新文件名
|
||||
uploader = extract_uploader(info or {})
|
||||
# 归位到作者目录:{作者}_{作者id}/{作品名}[_{短码}].ext
|
||||
uploader = author or extract_uploader(info or {})
|
||||
uploader_id = author_id or extract_uploader_id(info or {})
|
||||
title = ((info or {}).get("title") or "").strip()
|
||||
|
||||
slug_title = slugify(title, max_length=15) if title else ""
|
||||
if not slug_title:
|
||||
slug_title = datetime.now().strftime("%H%M%S")
|
||||
|
||||
time_suffix = datetime.now().strftime("%H%M%S")
|
||||
if uploader:
|
||||
slug_uploader = slugify(str(uploader))
|
||||
new_stem = f"{slug_uploader}_{slug_title}_视频_{time_suffix}"
|
||||
else:
|
||||
new_stem = f"{slug_title}_视频_{time_suffix}"
|
||||
|
||||
new_path = ensure_unique_path(
|
||||
original_file.with_name(f"{new_stem}{original_file.suffix}")
|
||||
author_dir = build_author_dir(uploader, uploader_id, source=url)
|
||||
new_path = unique_media_path(
|
||||
temp_root / author_dir / f"{build_work_stem(title)}{original_file.suffix}"
|
||||
)
|
||||
original_file.rename(new_path)
|
||||
shutil.move(str(original_file), str(new_path))
|
||||
# 下载用的 scratch 目录已空,顺手收掉(cleanup 不删目录)
|
||||
shutil.rmtree(temp_dir, ignore_errors=True)
|
||||
logger.info(
|
||||
f"yt-dlp 下载完成, 标题: {title}, "
|
||||
f"作者: {uploader}, 重命名: {new_path}"
|
||||
f"作者: {uploader}({uploader_id}), 落盘: {new_path}"
|
||||
)
|
||||
|
||||
return new_path
|
||||
return new_path, title
|
||||
|
||||
Reference in New Issue
Block a user