feat(video-analysis): 群策略 v3 / 群文件投递通道 / Web 管理页

- policy.py:per-group 正交策略(自动解析 / 自动策略 / 禁用策略 / 存储 A·B·C /
  公网 / 链接 / 群文件 + 平台限定),list.json v1/v2 → v3 自动迁移,
  写入统一走 PolicyStore(加锁 + .tmp 原子替换 + 字段归一)
- 群文件并行通道 group_file.py:打包 zip(可选 pyzipper AES-256)后优先走 S3 预签名、
  本地直传兜底;设了密码但 pyzipper 不可用就放弃上传,不退化成明文
- list_proc.py 收敛到「视频策略」统一入口,权限判定改走 policy
- Web 管理页 /hub/video_analysis(群策略 + 链接解析面板)与 services/web_jobs.py
  (只复用纯函数层,Web 上下文不发消息;内存任务表 + 并发闸门 + 超时)
- 媒体命名统一到 utils.py({作者}_{作者id}/{作品名}[_短码]),cleanup 回收空目录
- 测试:policy / 命名 / 群文件 / web_jobs 四组

顺带 pyproject 的 pytest 加 testpaths=tests(避免收进 debug/ 下的调试脚本)。

Co-Authored-By: Claude Code <noreply@anthropic.com>
This commit is contained in:
2026-09-22 14:23:32 +08:00
co-authored by Claude Code
parent 51b08ccb68
commit 4badcfcf32
29 changed files with 4838 additions and 670 deletions
@@ -14,15 +14,19 @@
"""
import re
import tempfile
from datetime import datetime
from pathlib import Path
from typing import Optional, Union
from nonebot import logger
from ...models import ContentFetchError
from ...utils import get_data_dir, parse_netscape_cookies, slugify
from ...utils import (
build_author_dir,
build_work_stem,
get_data_dir,
get_temp_root,
parse_netscape_cookies,
)
DATA_DIR = get_data_dir()
@@ -81,10 +85,10 @@ async def _parse(url: str):
# 4. 文章动态 → 转 opus
if await dynamic.is_article():
return await _parse_opus(dynamic.turn_to_opus(), "文章")
return await _parse_opus(dynamic.turn_to_opus(), "文章", url)
info = await dynamic.get_info()
return await _parse_dynamic_info(info)
return await _parse_dynamic_info(info, url)
async def _parse_article(read_id: int) -> tuple[str, Union[Path, list[Path]]]:
@@ -94,10 +98,12 @@ async def _parse_article(read_id: int) -> tuple[str, Union[Path, list[Path]]]:
# 文章接口对匿名请求风控更严(-509),必须带凭证
article = Article(read_id, _build_credential())
opus = await article.turn_to_opus()
return await _parse_opus(opus, "文章")
return await _parse_opus(opus, "文章", f"cv{read_id}")
async def _parse_opus(opus, kind: str) -> tuple[str, Union[Path, list[Path]]]:
async def _parse_opus(
opus, kind: str, source: str = ""
) -> tuple[str, Union[Path, list[Path]]]:
"""图文动态/专栏解析(opus 接口返回 dict,直接访问)"""
info = await opus.get_info()
item = info.get("item") or {}
@@ -107,10 +113,12 @@ async def _parse_opus(opus, kind: str) -> tuple[str, Union[Path, list[Path]]]:
images: list[str] = []
texts: list[str] = []
author = ""
author_id = ""
for module in item.get("modules") or []:
if module.get("module_type") == "MODULE_TYPE_AUTHOR":
author_info = module.get("module_author") or {}
author = author_info.get("name", "")
author_id = str(author_info.get("mid") or "")
elif module.get("module_type") == "MODULE_TYPE_CONTENT":
content = module.get("module_content") or {}
for para in content.get("paragraphs") or []:
@@ -127,16 +135,20 @@ async def _parse_opus(opus, kind: str) -> tuple[str, Union[Path, list[Path]]]:
if not images:
return text or f"B站{kind}", []
file_name = _build_file_name(author, text or f"B站{kind}", kind)
file_paths = await _download_images(images, file_name)
rel_stem = _build_rel_stem(author, author_id, text or f"B站{kind}", source)
file_paths = await _download_images(images, rel_stem)
return text, file_paths
async def _parse_dynamic_info(info: dict) -> tuple[str, Union[Path, list[Path]]]:
async def _parse_dynamic_info(
info: dict, source: str = ""
) -> tuple[str, Union[Path, list[Path]]]:
"""动态解析(图文 / 视频 / 纯文字)"""
item = info.get("item") or {}
modules = item.get("modules") or {}
author = ((modules.get("module_author") or {}).get("name")) or "B站用户"
module_author = modules.get("module_author") or {}
author = module_author.get("name") or "B站用户"
author_id = str(module_author.get("mid") or "")
module_dynamic = modules.get("module_dynamic") or {}
major = module_dynamic.get("major") or {}
major_type = major.get("type", "")
@@ -151,7 +163,13 @@ async def _parse_dynamic_info(info: dict) -> tuple[str, Union[Path, list[Path]]]
title = archive.get("title") or desc or "B站视频动态"
logger.info(f"B站视频动态: bvid={bvid} 标题={title[:40]}")
video_path = await download_video(f"https://www.bilibili.com/video/{bvid}")
# 动态数据里已有 up 的昵称/mid,传下去才能和同一位 up 的图文
# 落在同一个作者目录(否则要赌 yt-dlp 返回的 id 对得上)
video_path, _ = await download_video(
f"https://www.bilibili.com/video/{bvid}",
author=author,
author_id=author_id,
)
if video_path:
return title, video_path
raise ContentFetchError(f"视频动态下载失败: {bvid}")
@@ -173,8 +191,8 @@ async def _parse_dynamic_info(info: dict) -> tuple[str, Union[Path, list[Path]]]
images = [u for u in images if u]
if images:
file_name = _build_file_name(author, title, "动态")
file_paths = await _download_images(images, file_name)
rel_stem = _build_rel_stem(author, author_id, title, source)
file_paths = await _download_images(images, rel_stem)
logger.info(f"B站图文动态: 作者={author}, 标题={title[:40]}, 图片={len(images)} 张")
return title, file_paths
@@ -197,14 +215,17 @@ def _extract_text(nodes: list) -> str:
return "".join(parts)
def _build_file_name(author: str, title: str, kind: str) -> str:
"""构建文件名 stem: {作者}_{标题}_{类型}_{时间}"""
slug_author = slugify(author)
slug_title = slugify(title or "", max_length=15)
if not slug_title:
slug_title = datetime.now().strftime("%H%M%S")
time_suffix = datetime.now().strftime("%H%M%S")
return f"{slug_author}_{slug_title}_{kind}_{time_suffix}"
def _build_rel_stem(
author: str, author_id: str, title: str, source: str = ""
) -> str:
"""相对平台根的路径词干:`{作者}_{mid}/{作品名}`
昵称/mid 都拿不到时用 source(动态链接)当来源码(见 utils.build_author_dir)。
"""
return (
f"{build_author_dir(author, author_id, source=source)}"
f"/{build_work_stem(title)}"
)
def _build_credential():
@@ -227,20 +248,21 @@ def _build_credential():
)
async def _download_images(image_urls: list[str], file_name: str) -> list[Path]:
"""并发下载图片(复用抖音图文的下载流程)"""
import httpx
async def _download_images(image_urls: list[str], rel_stem: str) -> list[Path]:
"""并发下载图片(复用抖音图文的下载流程)
落 hexi/data/temp/bilibili(原先落系统 temp,cleanup 扫不到、永不清理)。
"""
from .douyin_api import _process_note_with_parsed
tmp_root = Path(tempfile.gettempdir()) / "bilibili"
tmp_root.mkdir(parents=True, exist_ok=True)
tmp_root = get_temp_root("bilibili")
headers = {
"Referer": BILI_REFERER,
"User-Agent": BILI_UA,
}
return await _process_note_with_parsed(
[[u] for u in image_urls], None, tmp_root, file_name, headers
[[u] for u in image_urls], None, tmp_root, rel_stem, headers
)
@@ -10,11 +10,9 @@ from nonebot import logger
from playwright.async_api import async_playwright
from ...models import DouyinFetchError
from ...utils import ensure_unique_path, get_temp_root
from ...utils import get_temp_root, unique_media_path
from .douyin_parser import (
ParsedDouyinContent,
extract_trailing_digits,
is_animated_note,
parse_animated_note_videos,
parse_douyin_response,
parse_note_images,
@@ -194,9 +192,9 @@ async def fetch_douyin_content(
if parsed.media_type == "视频":
content = await _process_video(
api_response, tmp_root, parsed.file_name, aweme_id, headers
api_response, tmp_root, parsed.rel_stem, aweme_id, headers
)
return parsed.file_name, content
return parsed.raw_title, content
elif parsed.media_type == "图片":
# 先解析 images 列表,区分纯动图和图文/图+视频
@@ -209,15 +207,15 @@ async def fetch_douyin_content(
images_urls,
video_url,
tmp_root,
parsed.file_name,
parsed.rel_stem,
headers,
)
else:
# 纯动图(所有项都是视频)
content = await _process_animated_note(
api_response, tmp_root, parsed.file_name, headers
api_response, tmp_root, parsed.rel_stem, headers
)
return parsed.file_name, content
return parsed.raw_title, content
return None, None
@@ -228,11 +226,14 @@ async def fetch_douyin_content(
async def _process_video(
api_response: dict,
tmp_root: Path,
file_name: str,
rel_stem: str,
aweme_id: str,
headers: Dict[str, str],
) -> Path:
"""处理视频内容,返回本地文件路径"""
"""处理视频内容,返回本地文件路径
rel_stem 是相对平台根的路径词干 `{作者目录}/{作品名}`(见 ParsedDouyinContent.rel_stem)。
"""
groups = parse_video_urls(api_response)
best_group = None
@@ -251,7 +252,7 @@ async def _process_video(
best = max(full, key=lambda x: x["br"])
logger.info(f"选择码率: {best['br']} - {best['url'][:60]}...")
output_path = ensure_unique_path(tmp_root / f"{file_name}.mp4")
output_path = unique_media_path(tmp_root / f"{rel_stem}.mp4")
async with httpx.AsyncClient(headers=headers) as client:
async with client.stream("GET", best["url"]) as resp:
resp.raise_for_status()
@@ -271,9 +272,10 @@ async def _process_video(
logger.info(f"选择视频码率: {video['br']}")
logger.info(f"选择音频码率: {audio['br']}")
video_path = tmp_root / f"{file_name}_v.mp4"
audio_path = tmp_root / f"{file_name}_a.mp4"
output_path = ensure_unique_path(tmp_root / f"{file_name}.mp4")
# 先定下产物名(顺带建好作者目录),分轨中间文件与产物同目录
output_path = unique_media_path(tmp_root / f"{rel_stem}.mp4")
video_path = output_path.parent / f"{output_path.stem}_v.mp4"
audio_path = output_path.parent / f"{output_path.stem}_a.mp4"
async with httpx.AsyncClient(headers=headers) as client:
logger.info("开始下载视频...")
@@ -291,7 +293,9 @@ async def _process_video(
f.write(chunk)
logger.info("合并视频和音频...")
merge_video_audio(video_path, audio_path, output_path)
# ffmpeg 是同步子进程,直接 await 不了:不丢线程池会卡住整个事件循环
# (合并期间 Web 轮询、群消息全都停摆)
await asyncio.to_thread(merge_video_audio, video_path, audio_path, output_path)
video_path.unlink()
audio_path.unlink()
@@ -306,11 +310,14 @@ async def _process_note_with_parsed(
images_urls: List[List[str]],
video_url: Optional[str],
tmp_root: Path,
file_name: str,
rel_stem: str,
headers: Dict[str, str],
) -> List[Path]:
"""根据已解析的图片/视频 URL 列表,并行下载"""
note_dir = ensure_unique_path(tmp_root / file_name)
"""根据已解析的图片/视频 URL 列表,并行下载
一个作品一个目录:`{平台根}/{作者目录}/{作品名}[_{短码}]/001.jpg…`
"""
note_dir = unique_media_path(tmp_root / rel_stem)
note_dir.mkdir(parents=True, exist_ok=True)
logger.info(f"图文保存目录: {note_dir}")
@@ -353,7 +360,7 @@ async def _process_note(
api_response: dict,
api_response_favorite: dict,
tmp_root: Path,
file_name: str,
rel_stem: str,
aweme_id: str,
headers: Dict[str, str],
) -> List[Path]:
@@ -367,7 +374,7 @@ async def _process_note(
raise DouyinFetchError("未找到图文链接")
return await _process_note_with_parsed(
images_urls, video_url, tmp_root, file_name, headers
images_urls, video_url, tmp_root, rel_stem, headers
)
@@ -377,14 +384,14 @@ async def _process_note(
async def _process_animated_note(
api_response: dict,
tmp_root: Path,
file_name: str,
rel_stem: str,
headers: Dict[str, str],
) -> List[Path]:
"""处理动图内容(media_type=42),并行下载所有无声 mp4 视频"""
video_urls = parse_animated_note_videos(api_response)
logger.info(f"解析到的动图视频链接: {video_urls}")
note_dir = ensure_unique_path(tmp_root / file_name)
note_dir = unique_media_path(tmp_root / rel_stem)
note_dir.mkdir(parents=True, exist_ok=True)
logger.info(f"动图保存目录: {note_dir}")
@@ -4,13 +4,12 @@ import json
import re
import urllib.parse
from dataclasses import dataclass
from datetime import datetime
from typing import Dict, List, Optional
from nonebot import logger
from ...models import DouyinFetchError
from ...utils import slugify
from ...utils import build_author_dir, build_work_stem
# ============================= 数据结构 =============================
@@ -22,7 +21,13 @@ class ParsedDouyinContent:
raw_title: str
raw_nickname: str
media_type: str # "视频" | "图片"
file_name: str # 构建好的文件名 stem
file_name: str # 作品名 stem(落盘文件名/多图子目录名)
author_dir: str = "" # 作者目录 `{昵称}_{uid}`(落盘与 S3 key 的首层)
@property
def rel_stem(self) -> str:
"""相对平台根目录的路径词干:`{作者目录}/{作品名}`"""
return f"{self.author_dir}/{self.file_name}" if self.author_dir else self.file_name
# ============================= URL 工具 =============================
@@ -109,6 +114,29 @@ def extract_author_nickname(api_response: dict) -> str:
return nickname
def extract_author_id(api_response: dict) -> str:
"""从 API 响应提取作者稳定 id,优先级: uid > unique_id(抖音号) > sec_uid
SSR 路径的 `aweme_list[0].author.uid` 与 API 路径的
`aweme_detail.author.uid` 走同一套取值;都拿不到返回空串,
作者目录退化成只用昵称(见 utils.build_author_dir)。
"""
aweme_detail = api_response.get("aweme_detail") or {}
author = aweme_detail.get("author") or {}
if not author:
aweme_list = api_response.get("aweme_list") or []
if aweme_list and isinstance(aweme_list, list):
author = (aweme_list[0] or {}).get("author") or {}
for key in ("uid", "unique_id", "sec_uid"):
value = str(author.get(key) or "").strip()
if value and value != "0":
logger.info(f"RAW作者id({key}):{value}")
return value
logger.info("未获取到作者 id,作者目录只用昵称")
return ""
def detect_media_type(referer_url: str | None) -> str | None:
"""根据页面 URL 检测媒体类型(视频/图片)"""
if referer_url is None:
@@ -326,36 +354,6 @@ def parse_ssr_page(html: str) -> Optional[dict]:
return {"aweme_list": [item]}
# ============================= 文件名构建 =============================
def build_file_name(
raw_title: str,
raw_nickname: str,
media_type: str,
) -> str:
"""
构建文件名 stem,格式: {作者}_{标题}_{类型}_{时间戳}
昵称不限长,标题最多 15 字符(slugify 后),末尾 HHMMSS 防覆盖。
"""
slug_nickname = slugify(raw_nickname)
if raw_title:
slug_title = slugify(raw_title, max_length=15)
else:
slug_title = ""
if not slug_title:
slug_title = datetime.now().strftime("%H%M%S")
logger.info(f"标题为空,使用短时间戳: {slug_title}")
slug_type = slugify(media_type)
time_suffix = datetime.now().strftime("%H%M%S")
return f"{slug_nickname}_{slug_title}_{slug_type}_{time_suffix}"
# ============================= 动图检测 =============================
@@ -410,11 +408,21 @@ def parse_douyin_response(
raw_title = extract_title_from_api(api_response)
raw_nickname = extract_author_nickname(api_response)
file_name = build_file_name(raw_title, raw_nickname, media_type)
raw_author_id = extract_author_id(api_response)
# 昵称/uid 都拿不到时的来源码:优先作品链接,其次响应里的作品 id
aweme_id = str(
(api_response.get("aweme_detail") or {}).get("aweme_id")
or ((api_response.get("aweme_list") or [{}])[0] or {}).get("aweme_id")
or ""
)
author_dir = build_author_dir(
raw_nickname, raw_author_id, source=referer_url or aweme_id
)
file_name = build_work_stem(raw_title)
logger.info(
f"内容标题: {raw_title}, 作者: {raw_nickname}, "
f"类型: {media_type}, 文件名: {file_name}"
f"内容标题: {raw_title}, 作者: {raw_nickname}({raw_author_id}), "
f"类型: {media_type}, 落盘路径: {author_dir}/{file_name}"
)
return ParsedDouyinContent(
@@ -422,4 +430,5 @@ def parse_douyin_response(
raw_nickname=raw_nickname,
media_type=media_type,
file_name=file_name,
author_dir=author_dir,
)
@@ -85,13 +85,13 @@ async def fetch_douyin_note_ssr(
images_urls, video_url = parse_note_images(api_response, None, vid)
if images_urls:
file_paths = await _process_note_with_parsed(
images_urls, video_url, tmp_root, parsed.file_name, headers
images_urls, video_url, tmp_root, parsed.rel_stem, headers
)
else:
file_paths = await _process_animated_note(
api_response, tmp_root, parsed.file_name, headers
api_response, tmp_root, parsed.rel_stem, headers
)
return parsed.file_name, file_paths
return parsed.raw_title, file_paths
async def _fetch_ssr_api_response(
@@ -19,16 +19,21 @@
import asyncio
import json
import re
import tempfile
import urllib.parse
from datetime import datetime
from pathlib import Path
from typing import Optional, Union
from nonebot import logger
from ...models import ContentFetchError
from ...utils import get_data_dir, get_temp_root, parse_netscape_cookies, slugify
from ...utils import (
build_author_dir,
build_work_stem,
get_data_dir,
get_temp_root,
parse_netscape_cookies,
unique_media_path,
)
DATA_DIR = get_data_dir()
@@ -214,6 +219,12 @@ async def _build_result(note: dict) -> tuple[str, Union[Path, list[Path]]]:
desc = note.get("desc") or ""
nickname = ((note.get("user") or {}).get("nickname")) or "小红书用户"
text = title or desc or "小红书笔记"
rel_stem = _build_rel_stem(
nickname,
_extract_author_id(note),
text,
source=str(note.get("noteId") or note.get("id") or ""),
)
# 1. 视频笔记 → 无水印原片优先
if note.get("type") == "video" and note.get("video"):
@@ -223,8 +234,7 @@ async def _build_result(note: dict) -> tuple[str, Union[Path, list[Path]]]:
if okey:
video_url = f"https://sns-video-bd.xhscdn.com/{okey}"
logger.info(f"小红书视频: 无水印原片 originVideoKey={okey[:30]}...")
file_name = _build_file_name(nickname, text, "视频")
video_path = await _download_video(video_url, file_name)
video_path = await _download_video(video_url, rel_stem)
return text, video_path
# 1b. 无 originVideoKey(国内站数据)→ 从 stream 分组选无水印原片
@@ -251,8 +261,7 @@ async def _build_result(note: dict) -> tuple[str, Union[Path, list[Path]]]:
f"{best.get('width')}x{best.get('height')} {best.get('fps')}fps "
f"size={best.get('size')} duration={duration}ms"
)
file_name = _build_file_name(nickname, text, "视频")
video_path = await _download_video(video_url, file_name)
video_path = await _download_video(video_url, rel_stem)
return text, video_path
raise ContentFetchError("小红书视频流解析失败")
@@ -266,25 +275,36 @@ async def _build_result(note: dict) -> tuple[str, Union[Path, list[Path]]]:
logger.info(f"小红书文字笔记: {text[:30]}")
return text, []
file_name = _build_file_name(nickname, text, "笔记")
file_paths = await _download_images(images, file_name)
file_paths = await _download_images(images, rel_stem)
logger.info(f"小红书图文笔记: 作者={nickname}, 图片={len(images)} 张")
return text, file_paths
def _build_file_name(nickname: str, title: str, kind: str) -> str:
"""构建文件名 stem: {作者}_{标题}_{类型}_{时间}"""
slug_nickname = slugify(nickname)
slug_title = slugify(title or "", max_length=15)
if not slug_title:
slug_title = datetime.now().strftime("%H%M%S")
time_suffix = datetime.now().strftime("%H%M%S")
return f"{slug_nickname}_{slug_title}_{kind}_{time_suffix}"
def _extract_author_id(note: dict) -> str:
"""小红书作者稳定 id(页面数据字段未实测,逐个兜底;取不到返回空串)"""
user = note.get("user") or {}
for key in ("userId", "user_id", "id"):
value = user.get(key)
if isinstance(value, (str, int)) and str(value).strip() not in ("", "0"):
return str(value).strip()
return ""
async def _download_images(image_urls: list[str], file_name: str) -> list[Path]:
def _build_rel_stem(
nickname: str, author_id: str, title: str, source: str = ""
) -> str:
"""相对平台根的路径词干:`{作者}_{userId}/{作品名}`
昵称/作者 id 都拿不到时用 source(笔记 id)当来源码(见 utils.build_author_dir)。
"""
return (
f"{build_author_dir(nickname, author_id, source=source)}"
f"/{build_work_stem(title)}"
)
async def _download_images(image_urls: list[str], rel_stem: str) -> list[Path]:
"""并发下载图片(复用抖音图文的下载流程)"""
import httpx
from .douyin_api import _process_note_with_parsed
@@ -294,11 +314,11 @@ async def _download_images(image_urls: list[str], file_name: str) -> list[Path]:
"User-Agent": REDNOTE_UA,
}
return await _process_note_with_parsed(
[[u] for u in image_urls], None, tmp_root, file_name, headers
[[u] for u in image_urls], None, tmp_root, rel_stem, headers
)
async def _download_video(video_url: str, file_name: str) -> Path:
async def _download_video(video_url: str, rel_stem: str) -> Path:
"""流式下载视频
注意:sns-video-bd(无水印原片)不带 Referer 或带 xiaohongshu.com
@@ -307,7 +327,7 @@ async def _download_video(video_url: str, file_name: str) -> Path:
import httpx
tmp_root = get_temp_root("xiaohongshu")
output_path = tmp_root / f"{file_name}.mp4"
output_path = unique_media_path(tmp_root / f"{rel_stem}.mp4")
headers = {"User-Agent": REDNOTE_UA}
async with httpx.AsyncClient(headers=headers, timeout=300) as client:
async with client.stream("GET", video_url) as resp:
@@ -3,9 +3,9 @@
import asyncio
import os
import re
import shutil
import sys
import tempfile
from datetime import datetime
from pathlib import Path
from typing import Optional
@@ -14,7 +14,14 @@ from nonebot import logger
from yt_dlp import YoutubeDL
from yt_dlp.utils import DownloadError
from ...utils import get_data_dir, get_temp_root, slugify, ensure_unique_path
from ...utils import (
build_author_dir,
build_work_stem,
get_data_dir,
get_temp_root,
slugify,
unique_media_path,
)
def detect_platform(url: str) -> str:
@@ -41,6 +48,16 @@ def extract_uploader(info: dict) -> Optional[str]:
)
def extract_uploader_id(info: dict) -> str:
"""作者稳定 id:channel_id > uploader_id(@handle / B站 mid),拿不到返回空串
不用 `id`(那是作品 id,会把同一作者的作品拆到不同目录)。
"""
if not info:
return ""
return str(info.get("channel_id") or info.get("uploader_id") or "").strip()
def get_ffmpeg_path() -> str:
scripts_dir = os.path.dirname(sys.executable)
ffmpeg_path = os.path.join(scripts_dir, "ffmpeg.exe")
@@ -78,19 +95,36 @@ async def _retry_download(
logger.error(
f"yt-dlp 重试 {max_retries} 次后仍失败: {str(e)[:120]}"
)
except Exception as e:
except Exception:
# 非 DownloadError(如 OSError)不重试,直接抛出
raise
raise last_error # type: ignore[misc]
async def download_video(url: str) -> Optional[Path]:
"""下载视频,支持直链和 yt-dlp"""
async def download_video(
url: str,
*,
author: Optional[str] = None,
author_id: Optional[str] = None,
) -> tuple[Optional[Path], str]:
"""下载视频,支持直链和 yt-dlp
author / author_id 可由调用方覆盖(如 B站 视频动态已从动态数据里拿到 mid,
传进来才能和同一位 up 的图文落在同一个作者目录)。
Returns:
(本地文件, 作品标题) — 失败时 (None, "")。
落盘位置:`temp/{平台}/{作者}_{作者id}/{作品名}[_{短码}].ext`
(直链拿不到作者信息,统一进 `未知作者/`)。
"""
platform = detect_platform(url)
temp_root = get_temp_root(platform)
# ---------- 1. 直链探测 ----------
# 不含 m3u8:HLS 播放列表直下只会得到一个文本文件,交给 yt-dlp 处理
direct_media_ext = re.search(
r"\.(mp4|m3u8|ts|webm|mov|flv)(?:$|\?)", url, re.IGNORECASE
r"\.(mp4|ts|webm|mov|flv)(?:$|\?)", url, re.IGNORECASE
)
is_direct = bool(direct_media_ext)
@@ -105,39 +139,35 @@ async def download_video(url: str) -> Optional[Path]:
is_direct = False
if is_direct:
temp_dir = tempfile.mkdtemp(prefix="direct_ytcache_", dir=get_temp_root("ytcache"))
ext = "mp4"
m = re.search(r"\.([a-zA-Z0-9]{2,5})(?:$|\?)", url)
if m and len(m.group(1)) <= 5:
ext = m.group(1)
url_stem = Path(url.split("?")[0]).stem or "video"
slug_stem = slugify(url_stem, max_length=15)
if not slug_stem:
slug_stem = datetime.now().strftime("%H%M%S")
time_suffix = datetime.now().strftime("%H%M%S")
new_name = f"{slug_stem}_视频_{time_suffix}.{ext}"
filename = os.path.join(temp_dir, new_name)
slug_stem = slugify(url_stem, max_length=15) or "视频"
# 直链拿不到作者信息 → `未知作者_{来源短码}`(同一链接稳定、不同链接不撞)
author_dir = build_author_dir(author, author_id, source=url)
final_path = unique_media_path(temp_root / author_dir / f"{slug_stem}.{ext}")
try:
async with AsyncClient(follow_redirects=True, timeout=300) as client:
async with client.stream("GET", url) as resp:
resp.raise_for_status()
with open(filename, "wb") as fh:
with open(final_path, "wb") as fh:
async for chunk in resp.aiter_bytes(chunk_size=8192):
fh.write(chunk)
final_path = ensure_unique_path(Path(filename))
logger.info(f"直接下载完成: {final_path}")
return final_path
return final_path, ""
except Exception:
logger.exception("直接下载失败,回退 yt-dlp")
if os.path.exists(filename):
os.remove(filename)
final_path.unlink(missing_ok=True)
# ---------- 2. yt-dlp 下载 ----------
platform = detect_platform(url)
temp_dir = tempfile.mkdtemp(prefix="ytcache_", dir=get_temp_root("ytcache"))
# 先下到 scratch 目录(outtmpl 必须在拿到 info 之前给定),拿到 info 后再
# 按作者归位到 temp/{平台}/{作者}_{作者id}/
temp_dir = tempfile.mkdtemp(prefix="_dl_", dir=temp_root)
output_path = os.path.join(temp_dir, "%(title).80s.%(ext)s")
base_opts = {
@@ -188,7 +218,10 @@ async def download_video(url: str) -> Optional[Path]:
"Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,image/webp,*/*;q=0.8",
"Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
}
base_opts["extractor_args"] = {"twitter": {"api": ["syndication"]}}
# 不要强制 twitter:api=syndication:该端点为未登录视角,对敏感/受限推文
# 只返回 tombstone(无 mediaDetails),且会无条件覆盖已登录 GraphQL 的结果,
# 表现为 "No video could be found in this tweet"。默认走 GraphQL + cookies,
# 遇 429 yt-dlp 会自行回退 syndication。
elif platform == "youtube":
base_opts["http_headers"] = {
"User-Agent": ua,
@@ -206,54 +239,53 @@ async def download_video(url: str) -> Optional[Path]:
info = await _retry_download(loop, url, base_opts)
except Exception:
logger.exception("yt-dlp 下载失败")
# YouTube: cookies 可能触发 bot 检测导致只返回图片无视频格式
# 回退无 cookie 模式重试
if platform == "youtube" and "cookiefile" in base_opts:
# YouTube: cookies 可能触发 bot 检测导致只返回图片无视频格式
# 回退无 cookie 模式重试
logger.info("YouTube 回退无 cookies 模式重试...")
base_opts.pop("cookiefile", None)
base_opts.pop("http_headers", None)
# 清理失败残留
for f in Path(temp_dir).glob("*.*"):
try:
f.unlink()
except Exception:
pass
try:
info = await _retry_download(loop, url, base_opts, max_retries=2)
except Exception:
logger.exception("yt-dlp 无 cookies 重试也失败")
return None
elif platform == "twitter":
# X 登录态失效(auth_token 过期)时 GraphQL 会直接拒绝请求;
# 退回未登录的 syndication 端点,公开推文仍可下载(敏感推文会失败)
logger.info("Twitter 回退 syndication 端点重试...")
base_opts["extractor_args"] = {"twitter": {"api": ["syndication"]}}
else:
return None
return None, ""
# 清理失败残留
for f in Path(temp_dir).glob("*.*"):
try:
f.unlink()
except Exception:
pass
try:
info = await _retry_download(loop, url, base_opts, max_retries=2)
except Exception:
logger.exception("yt-dlp 回退重试也失败")
return None, ""
files = list(Path(temp_dir).glob("*.*"))
if not files:
return None
return None, ""
original_file = files[0]
# 构建新文件名
uploader = extract_uploader(info or {})
# 归位到作者目录:{作者}_{作者id}/{作品名}[_{短码}].ext
uploader = author or extract_uploader(info or {})
uploader_id = author_id or extract_uploader_id(info or {})
title = ((info or {}).get("title") or "").strip()
slug_title = slugify(title, max_length=15) if title else ""
if not slug_title:
slug_title = datetime.now().strftime("%H%M%S")
time_suffix = datetime.now().strftime("%H%M%S")
if uploader:
slug_uploader = slugify(str(uploader))
new_stem = f"{slug_uploader}_{slug_title}_视频_{time_suffix}"
else:
new_stem = f"{slug_title}_视频_{time_suffix}"
new_path = ensure_unique_path(
original_file.with_name(f"{new_stem}{original_file.suffix}")
author_dir = build_author_dir(uploader, uploader_id, source=url)
new_path = unique_media_path(
temp_root / author_dir / f"{build_work_stem(title)}{original_file.suffix}"
)
original_file.rename(new_path)
shutil.move(str(original_file), str(new_path))
# 下载用的 scratch 目录已空,顺手收掉(cleanup 不删目录)
shutil.rmtree(temp_dir, ignore_errors=True)
logger.info(
f"yt-dlp 下载完成, 标题: {title}, "
f"作者: {uploader}, 重命名: {new_path}"
f"作者: {uploader}({uploader_id}), 落盘: {new_path}"
)
return new_path
return new_path, title