265 lines
9.0 KiB
Python
265 lines
9.0 KiB
Python
"""通用视频下载 — yt-dlp + 直链探测"""
|
||||
|
|
|
|||
|
|
import asyncio
|
|||
|
|
import os
|
|||
|
|
import re
|
|||
|
|
import sys
|
|||
|
|
import tempfile
|
|||
|
|
from datetime import datetime
|
|||
|
|
from pathlib import Path
|
|||
|
|
from typing import Optional
|
|||
|
|
|
|||
|
|
from httpx import AsyncClient
|
|||
|
|
from nonebot import logger
|
|||
|
|
from yt_dlp import YoutubeDL
|
|||
|
|
from yt_dlp.utils import DownloadError
|
|||
|
|
|
|||
|
|
from ..utils import get_temp_root, slugify, ensure_unique_path
|
|||
|
|
|
|||
|
|
|
|||
|
|
def detect_platform(url: str) -> str:
|
|||
|
|
if "bilibili.com" in url or "b23.tv" in url:
|
|||
|
|
return "bilibili"
|
|||
|
|
if "twitter.com" in url or "x.com" in url:
|
|||
|
|
return "twitter"
|
|||
|
|
if "youtube.com" in url or "youtu.be" in url:
|
|||
|
|
return "youtube"
|
|||
|
|
if "douyin.com" in url or "v.douyin.com" in url or "iesdouyin.com" in url:
|
|||
|
|
return "douyin"
|
|||
|
|
return "other"
|
|||
|
|
|
|||
|
|
|
|||
|
|
def extract_uploader(info: dict) -> Optional[str]:
|
|||
|
|
"""从 yt-dlp info dict 提取上传者,优先级: uploader > channel > creator > uploader_id"""
|
|||
|
|
if not info:
|
|||
|
|
return None
|
|||
|
|
return (
|
|||
|
|
info.get("uploader")
|
|||
|
|
or info.get("channel")
|
|||
|
|
or info.get("creator")
|
|||
|
|
or info.get("uploader_id")
|
|||
|
|
)
|
|||
|
|
|
|||
|
|
|
|||
|
|
def get_ffmpeg_path() -> str:
|
|||
|
|
scripts_dir = os.path.dirname(sys.executable)
|
|||
|
|
ffmpeg_path = os.path.join(scripts_dir, "ffmpeg.exe")
|
|||
|
|
if os.path.exists(ffmpeg_path):
|
|||
|
|
return ffmpeg_path
|
|||
|
|
return "ffmpeg"
|
|||
|
|
|
|||
|
|
|
|||
|
|
def _get_data_dir() -> Path:
|
|||
|
|
"""获取 data/ 目录路径"""
|
|||
|
|
return Path(__file__).resolve().parent.parent / "data"
|
|||
|
|
|
|||
|
|
|
|||
|
|
async def _retry_download(
|
|||
|
|
loop: asyncio.AbstractEventLoop,
|
|||
|
|
url: str,
|
|||
|
|
base_opts: dict,
|
|||
|
|
max_retries: int = 3,
|
|||
|
|
) -> dict:
|
|||
|
|
"""带重试的 yt-dlp 下载,处理 RemoteDisconnected 等瞬态错误"""
|
|||
|
|
|
|||
|
|
def _run_yt():
|
|||
|
|
with YoutubeDL(base_opts) as ydl:
|
|||
|
|
return ydl.extract_info(url, download=True)
|
|||
|
|
|
|||
|
|
last_error = None
|
|||
|
|
for attempt in range(1, max_retries + 1):
|
|||
|
|
try:
|
|||
|
|
return await loop.run_in_executor(None, _run_yt)
|
|||
|
|
except DownloadError as e:
|
|||
|
|
last_error = e
|
|||
|
|
if attempt < max_retries:
|
|||
|
|
delay = 2 ** attempt # 2s, 4s, 8s
|
|||
|
|
logger.warning(
|
|||
|
|
f"yt-dlp 下载失败 (第 {attempt}/{max_retries} 次),"
|
|||
|
|
f"{delay}s 后重试: {str(e)[:120]}"
|
|||
|
|
)
|
|||
|
|
await asyncio.sleep(delay)
|
|||
|
|
else:
|
|||
|
|
logger.error(
|
|||
|
|
f"yt-dlp 重试 {max_retries} 次后仍失败: {str(e)[:120]}"
|
|||
|
|
)
|
|||
|
|
except Exception as e:
|
|||
|
|
# 非 DownloadError(如 OSError)不重试,直接抛出
|
|||
|
|
raise
|
|||
|
|
|
|||
|
|
raise last_error # type: ignore[misc]
|
|||
|
|
|
|||
|
|
|
|||
|
|
async def download_video(url: str) -> Optional[Path]:
|
|||
|
|
"""下载视频,支持直链和 yt-dlp"""
|
|||
|
|
|
|||
|
|
# ---------- 1. 直链探测 ----------
|
|||
|
|
direct_media_ext = re.search(
|
|||
|
|
r"\.(mp4|m3u8|ts|webm|mov|flv)(?:$|\?)", url, re.IGNORECASE
|
|||
|
|
)
|
|||
|
|
is_direct = bool(direct_media_ext)
|
|||
|
|
|
|||
|
|
if not is_direct:
|
|||
|
|
try:
|
|||
|
|
async with AsyncClient(follow_redirects=True, timeout=30) as client:
|
|||
|
|
head = await client.head(url, follow_redirects=True)
|
|||
|
|
ctype = head.headers.get("content-type", "")
|
|||
|
|
if ctype.startswith("video/") or "application/octet-stream" in ctype:
|
|||
|
|
is_direct = True
|
|||
|
|
except Exception:
|
|||
|
|
is_direct = False
|
|||
|
|
|
|||
|
|
if is_direct:
|
|||
|
|
temp_dir = tempfile.mkdtemp(prefix="direct_ytcache_", dir=get_temp_root("ytcache"))
|
|||
|
|
ext = "mp4"
|
|||
|
|
m = re.search(r"\.([a-zA-Z0-9]{2,5})(?:$|\?)", url)
|
|||
|
|
if m and len(m.group(1)) <= 5:
|
|||
|
|
ext = m.group(1)
|
|||
|
|
|
|||
|
|
url_stem = Path(url.split("?")[0]).stem or "video"
|
|||
|
|
slug_stem = slugify(url_stem, max_length=15)
|
|||
|
|
if not slug_stem:
|
|||
|
|
slug_stem = datetime.now().strftime("%H%M%S")
|
|||
|
|
time_suffix = datetime.now().strftime("%H%M%S")
|
|||
|
|
new_name = f"{slug_stem}_视频_{time_suffix}.{ext}"
|
|||
|
|
filename = os.path.join(temp_dir, new_name)
|
|||
|
|
|
|||
|
|
try:
|
|||
|
|
async with AsyncClient(follow_redirects=True, timeout=300) as client:
|
|||
|
|
async with client.stream("GET", url) as resp:
|
|||
|
|
resp.raise_for_status()
|
|||
|
|
with open(filename, "wb") as fh:
|
|||
|
|
async for chunk in resp.aiter_bytes(chunk_size=8192):
|
|||
|
|
fh.write(chunk)
|
|||
|
|
|
|||
|
|
final_path = ensure_unique_path(Path(filename))
|
|||
|
|
logger.info(f"直接下载完成: {final_path}")
|
|||
|
|
return final_path
|
|||
|
|
except Exception:
|
|||
|
|
logger.exception("直接下载失败,回退 yt-dlp")
|
|||
|
|
if os.path.exists(filename):
|
|||
|
|
os.remove(filename)
|
|||
|
|
|
|||
|
|
# ---------- 2. yt-dlp 下载 ----------
|
|||
|
|
platform = detect_platform(url)
|
|||
|
|
temp_dir = tempfile.mkdtemp(prefix="ytcache_", dir=get_temp_root("ytcache"))
|
|||
|
|
output_path = os.path.join(temp_dir, "%(title).80s.%(ext)s")
|
|||
|
|
|
|||
|
|
base_opts = {
|
|||
|
|
"outtmpl": output_path,
|
|||
|
|
"format": "bestvideo+bestaudio/best",
|
|||
|
|
"merge_output_format": "mp4",
|
|||
|
|
"noplaylist": True,
|
|||
|
|
"quiet": True,
|
|||
|
|
"ffmpeg_location": get_ffmpeg_path(),
|
|||
|
|
}
|
|||
|
|
|
|||
|
|
cookie_path = _get_data_dir() / "cookies.txt"
|
|||
|
|
|
|||
|
|
if platform in ("bilibili", "twitter", "youtube"):
|
|||
|
|
if not cookie_path.exists():
|
|||
|
|
raise RuntimeError(f"{platform} 需要 cookies.txt,但未找到")
|
|||
|
|
base_opts["cookiefile"] = str(cookie_path)
|
|||
|
|
logger.info(f"{platform} 使用 cookies.txt 下载")
|
|||
|
|
|
|||
|
|
ua = (
|
|||
|
|
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) "
|
|||
|
|
"AppleWebKit/537.36 (KHTML, like Gecko) "
|
|||
|
|
"Chrome/122.0.0.0 Safari/537.36"
|
|||
|
|
)
|
|||
|
|
|
|||
|
|
if platform == "bilibili":
|
|||
|
|
base_opts["http_headers"] = {
|
|||
|
|
"User-Agent": ua,
|
|||
|
|
"Referer": "https://www.bilibili.com/",
|
|||
|
|
"Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,image/webp,*/*;q=0.8",
|
|||
|
|
"Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
|
|||
|
|
"Accept-Encoding": "gzip, deflate, br",
|
|||
|
|
}
|
|||
|
|
base_opts["extractor_args"] = {
|
|||
|
|
"bilibili": {
|
|||
|
|
"header": [
|
|||
|
|
"Referer:https://www.bilibili.com/",
|
|||
|
|
f"User-Agent:{ua}",
|
|||
|
|
"Accept:text/html,application/xhtml+xml,application/xml;q=0.9,image/webp,*/*;q=0.8",
|
|||
|
|
"Accept-Language:zh-CN,zh;q=0.9,en;q=0.8",
|
|||
|
|
]
|
|||
|
|
}
|
|||
|
|
}
|
|||
|
|
elif platform == "twitter":
|
|||
|
|
base_opts["http_headers"] = {
|
|||
|
|
"User-Agent": ua,
|
|||
|
|
"Referer": "https://x.com/",
|
|||
|
|
"Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,image/webp,*/*;q=0.8",
|
|||
|
|
"Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
|
|||
|
|
}
|
|||
|
|
base_opts["extractor_args"] = {"twitter": {"api": ["syndication"]}}
|
|||
|
|
elif platform == "youtube":
|
|||
|
|
base_opts["http_headers"] = {
|
|||
|
|
"User-Agent": ua,
|
|||
|
|
"Referer": "https://www.youtube.com/",
|
|||
|
|
"Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,image/webp,*/*;q=0.8",
|
|||
|
|
"Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
|
|||
|
|
}
|
|||
|
|
|
|||
|
|
else:
|
|||
|
|
logger.info("其他平台,默认无 cookies 下载")
|
|||
|
|
|
|||
|
|
loop = asyncio.get_event_loop()
|
|||
|
|
|
|||
|
|
try:
|
|||
|
|
info = await _retry_download(loop, url, base_opts)
|
|||
|
|
except Exception:
|
|||
|
|
logger.exception("yt-dlp 下载失败")
|
|||
|
|
# YouTube: cookies 可能触发 bot 检测导致只返回图片无视频格式
|
|||
|
|
# 回退无 cookie 模式重试
|
|||
|
|
if platform == "youtube" and "cookiefile" in base_opts:
|
|||
|
|
logger.info("YouTube 回退无 cookies 模式重试...")
|
|||
|
|
base_opts.pop("cookiefile", None)
|
|||
|
|
base_opts.pop("http_headers", None)
|
|||
|
|
# 清理失败残留
|
|||
|
|
for f in Path(temp_dir).glob("*.*"):
|
|||
|
|
try:
|
|||
|
|
f.unlink()
|
|||
|
|
except Exception:
|
|||
|
|
pass
|
|||
|
|
try:
|
|||
|
|
info = await _retry_download(loop, url, base_opts, max_retries=2)
|
|||
|
|
except Exception:
|
|||
|
|
logger.exception("yt-dlp 无 cookies 重试也失败")
|
|||
|
|
return None
|
|||
|
|
else:
|
|||
|
|
return None
|
|||
|
|
|
|||
|
|
files = list(Path(temp_dir).glob("*.*"))
|
|||
|
|
if not files:
|
|||
|
|
return None
|
|||
|
|
|
|||
|
|
original_file = files[0]
|
|||
|
|
|
|||
|
|
# 构建新文件名
|
|||
|
|
uploader = extract_uploader(info or {})
|
|||
|
|
title = ((info or {}).get("title") or "").strip()
|
|||
|
|
|
|||
|
|
slug_title = slugify(title, max_length=15) if title else ""
|
|||
|
|
if not slug_title:
|
|||
|
|
slug_title = datetime.now().strftime("%H%M%S")
|
|||
|
|
|
|||
|
|
time_suffix = datetime.now().strftime("%H%M%S")
|
|||
|
|
if uploader:
|
|||
|
|
slug_uploader = slugify(str(uploader))
|
|||
|
|
new_stem = f"{slug_uploader}_{slug_title}_视频_{time_suffix}"
|
|||
|
|
else:
|
|||
|
|
new_stem = f"{slug_title}_视频_{time_suffix}"
|
|||
|
|
|
|||
|
|
new_path = ensure_unique_path(
|
|||
|
|
original_file.with_name(f"{new_stem}{original_file.suffix}")
|
|||
|
|
)
|
|||
|
|
original_file.rename(new_path)
|
|||
|
|
logger.info(
|
|||
|
|
f"yt-dlp 下载完成, 标题: {title}, "
|
|||
|
|
f"作者: {uploader}, 重命名: {new_path}"
|
|||
|
|
)
|
|||
|
|
|
|||
|
|
return new_path
|