Add HeXi bot codebase: custom plugins, web frontends, tests
- hexi core: message handling, rate limiting, cooldown, plugin manager - Custom plugins: BF stats, daily check-in, quotes, persona cards, etc. - Community plugins vendored under hexi/plugins with local fixes - Web admin frontends (learning-chat, persona-admin), unified hexi/web - Tests for rate_limit/cooldown/memes/persona; poetry.lock Co-Authored-By: Claude <noreply@anthropic.com>
This commit is contained in:
@@ -0,0 +1,264 @@
|
||||
"""通用视频下载 — yt-dlp + 直链探测"""
|
||||
|
||||
import asyncio
|
||||
import os
|
||||
import re
|
||||
import sys
|
||||
import tempfile
|
||||
from datetime import datetime
|
||||
from pathlib import Path
|
||||
from typing import Optional
|
||||
|
||||
from httpx import AsyncClient
|
||||
from nonebot import logger
|
||||
from yt_dlp import YoutubeDL
|
||||
from yt_dlp.utils import DownloadError
|
||||
|
||||
from ..utils import get_temp_root, slugify, ensure_unique_path
|
||||
|
||||
|
||||
def detect_platform(url: str) -> str:
|
||||
if "bilibili.com" in url or "b23.tv" in url:
|
||||
return "bilibili"
|
||||
if "twitter.com" in url or "x.com" in url:
|
||||
return "twitter"
|
||||
if "youtube.com" in url or "youtu.be" in url:
|
||||
return "youtube"
|
||||
if "douyin.com" in url or "v.douyin.com" in url or "iesdouyin.com" in url:
|
||||
return "douyin"
|
||||
return "other"
|
||||
|
||||
|
||||
def extract_uploader(info: dict) -> Optional[str]:
|
||||
"""从 yt-dlp info dict 提取上传者,优先级: uploader > channel > creator > uploader_id"""
|
||||
if not info:
|
||||
return None
|
||||
return (
|
||||
info.get("uploader")
|
||||
or info.get("channel")
|
||||
or info.get("creator")
|
||||
or info.get("uploader_id")
|
||||
)
|
||||
|
||||
|
||||
def get_ffmpeg_path() -> str:
|
||||
scripts_dir = os.path.dirname(sys.executable)
|
||||
ffmpeg_path = os.path.join(scripts_dir, "ffmpeg.exe")
|
||||
if os.path.exists(ffmpeg_path):
|
||||
return ffmpeg_path
|
||||
return "ffmpeg"
|
||||
|
||||
|
||||
def _get_data_dir() -> Path:
|
||||
"""获取 data/ 目录路径"""
|
||||
return Path(__file__).resolve().parent.parent / "data"
|
||||
|
||||
|
||||
async def _retry_download(
|
||||
loop: asyncio.AbstractEventLoop,
|
||||
url: str,
|
||||
base_opts: dict,
|
||||
max_retries: int = 3,
|
||||
) -> dict:
|
||||
"""带重试的 yt-dlp 下载,处理 RemoteDisconnected 等瞬态错误"""
|
||||
|
||||
def _run_yt():
|
||||
with YoutubeDL(base_opts) as ydl:
|
||||
return ydl.extract_info(url, download=True)
|
||||
|
||||
last_error = None
|
||||
for attempt in range(1, max_retries + 1):
|
||||
try:
|
||||
return await loop.run_in_executor(None, _run_yt)
|
||||
except DownloadError as e:
|
||||
last_error = e
|
||||
if attempt < max_retries:
|
||||
delay = 2 ** attempt # 2s, 4s, 8s
|
||||
logger.warning(
|
||||
f"yt-dlp 下载失败 (第 {attempt}/{max_retries} 次),"
|
||||
f"{delay}s 后重试: {str(e)[:120]}"
|
||||
)
|
||||
await asyncio.sleep(delay)
|
||||
else:
|
||||
logger.error(
|
||||
f"yt-dlp 重试 {max_retries} 次后仍失败: {str(e)[:120]}"
|
||||
)
|
||||
except Exception as e:
|
||||
# 非 DownloadError(如 OSError)不重试,直接抛出
|
||||
raise
|
||||
|
||||
raise last_error # type: ignore[misc]
|
||||
|
||||
|
||||
async def download_video(url: str) -> Optional[Path]:
|
||||
"""下载视频,支持直链和 yt-dlp"""
|
||||
|
||||
# ---------- 1. 直链探测 ----------
|
||||
direct_media_ext = re.search(
|
||||
r"\.(mp4|m3u8|ts|webm|mov|flv)(?:$|\?)", url, re.IGNORECASE
|
||||
)
|
||||
is_direct = bool(direct_media_ext)
|
||||
|
||||
if not is_direct:
|
||||
try:
|
||||
async with AsyncClient(follow_redirects=True, timeout=30) as client:
|
||||
head = await client.head(url, follow_redirects=True)
|
||||
ctype = head.headers.get("content-type", "")
|
||||
if ctype.startswith("video/") or "application/octet-stream" in ctype:
|
||||
is_direct = True
|
||||
except Exception:
|
||||
is_direct = False
|
||||
|
||||
if is_direct:
|
||||
temp_dir = tempfile.mkdtemp(prefix="direct_ytcache_", dir=get_temp_root("ytcache"))
|
||||
ext = "mp4"
|
||||
m = re.search(r"\.([a-zA-Z0-9]{2,5})(?:$|\?)", url)
|
||||
if m and len(m.group(1)) <= 5:
|
||||
ext = m.group(1)
|
||||
|
||||
url_stem = Path(url.split("?")[0]).stem or "video"
|
||||
slug_stem = slugify(url_stem, max_length=15)
|
||||
if not slug_stem:
|
||||
slug_stem = datetime.now().strftime("%H%M%S")
|
||||
time_suffix = datetime.now().strftime("%H%M%S")
|
||||
new_name = f"{slug_stem}_视频_{time_suffix}.{ext}"
|
||||
filename = os.path.join(temp_dir, new_name)
|
||||
|
||||
try:
|
||||
async with AsyncClient(follow_redirects=True, timeout=300) as client:
|
||||
async with client.stream("GET", url) as resp:
|
||||
resp.raise_for_status()
|
||||
with open(filename, "wb") as fh:
|
||||
async for chunk in resp.aiter_bytes(chunk_size=8192):
|
||||
fh.write(chunk)
|
||||
|
||||
final_path = ensure_unique_path(Path(filename))
|
||||
logger.info(f"直接下载完成: {final_path}")
|
||||
return final_path
|
||||
except Exception:
|
||||
logger.exception("直接下载失败,回退 yt-dlp")
|
||||
if os.path.exists(filename):
|
||||
os.remove(filename)
|
||||
|
||||
# ---------- 2. yt-dlp 下载 ----------
|
||||
platform = detect_platform(url)
|
||||
temp_dir = tempfile.mkdtemp(prefix="ytcache_", dir=get_temp_root("ytcache"))
|
||||
output_path = os.path.join(temp_dir, "%(title).80s.%(ext)s")
|
||||
|
||||
base_opts = {
|
||||
"outtmpl": output_path,
|
||||
"format": "bestvideo+bestaudio/best",
|
||||
"merge_output_format": "mp4",
|
||||
"noplaylist": True,
|
||||
"quiet": True,
|
||||
"ffmpeg_location": get_ffmpeg_path(),
|
||||
}
|
||||
|
||||
cookie_path = _get_data_dir() / "cookies.txt"
|
||||
|
||||
if platform in ("bilibili", "twitter", "youtube"):
|
||||
if not cookie_path.exists():
|
||||
raise RuntimeError(f"{platform} 需要 cookies.txt,但未找到")
|
||||
base_opts["cookiefile"] = str(cookie_path)
|
||||
logger.info(f"{platform} 使用 cookies.txt 下载")
|
||||
|
||||
ua = (
|
||||
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) "
|
||||
"AppleWebKit/537.36 (KHTML, like Gecko) "
|
||||
"Chrome/122.0.0.0 Safari/537.36"
|
||||
)
|
||||
|
||||
if platform == "bilibili":
|
||||
base_opts["http_headers"] = {
|
||||
"User-Agent": ua,
|
||||
"Referer": "https://www.bilibili.com/",
|
||||
"Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,image/webp,*/*;q=0.8",
|
||||
"Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
|
||||
"Accept-Encoding": "gzip, deflate, br",
|
||||
}
|
||||
base_opts["extractor_args"] = {
|
||||
"bilibili": {
|
||||
"header": [
|
||||
"Referer:https://www.bilibili.com/",
|
||||
f"User-Agent:{ua}",
|
||||
"Accept:text/html,application/xhtml+xml,application/xml;q=0.9,image/webp,*/*;q=0.8",
|
||||
"Accept-Language:zh-CN,zh;q=0.9,en;q=0.8",
|
||||
]
|
||||
}
|
||||
}
|
||||
elif platform == "twitter":
|
||||
base_opts["http_headers"] = {
|
||||
"User-Agent": ua,
|
||||
"Referer": "https://x.com/",
|
||||
"Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,image/webp,*/*;q=0.8",
|
||||
"Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
|
||||
}
|
||||
base_opts["extractor_args"] = {"twitter": {"api": ["syndication"]}}
|
||||
elif platform == "youtube":
|
||||
base_opts["http_headers"] = {
|
||||
"User-Agent": ua,
|
||||
"Referer": "https://www.youtube.com/",
|
||||
"Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,image/webp,*/*;q=0.8",
|
||||
"Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
|
||||
}
|
||||
|
||||
else:
|
||||
logger.info("其他平台,默认无 cookies 下载")
|
||||
|
||||
loop = asyncio.get_event_loop()
|
||||
|
||||
try:
|
||||
info = await _retry_download(loop, url, base_opts)
|
||||
except Exception:
|
||||
logger.exception("yt-dlp 下载失败")
|
||||
# YouTube: cookies 可能触发 bot 检测导致只返回图片无视频格式
|
||||
# 回退无 cookie 模式重试
|
||||
if platform == "youtube" and "cookiefile" in base_opts:
|
||||
logger.info("YouTube 回退无 cookies 模式重试...")
|
||||
base_opts.pop("cookiefile", None)
|
||||
base_opts.pop("http_headers", None)
|
||||
# 清理失败残留
|
||||
for f in Path(temp_dir).glob("*.*"):
|
||||
try:
|
||||
f.unlink()
|
||||
except Exception:
|
||||
pass
|
||||
try:
|
||||
info = await _retry_download(loop, url, base_opts, max_retries=2)
|
||||
except Exception:
|
||||
logger.exception("yt-dlp 无 cookies 重试也失败")
|
||||
return None
|
||||
else:
|
||||
return None
|
||||
|
||||
files = list(Path(temp_dir).glob("*.*"))
|
||||
if not files:
|
||||
return None
|
||||
|
||||
original_file = files[0]
|
||||
|
||||
# 构建新文件名
|
||||
uploader = extract_uploader(info or {})
|
||||
title = ((info or {}).get("title") or "").strip()
|
||||
|
||||
slug_title = slugify(title, max_length=15) if title else ""
|
||||
if not slug_title:
|
||||
slug_title = datetime.now().strftime("%H%M%S")
|
||||
|
||||
time_suffix = datetime.now().strftime("%H%M%S")
|
||||
if uploader:
|
||||
slug_uploader = slugify(str(uploader))
|
||||
new_stem = f"{slug_uploader}_{slug_title}_视频_{time_suffix}"
|
||||
else:
|
||||
new_stem = f"{slug_title}_视频_{time_suffix}"
|
||||
|
||||
new_path = ensure_unique_path(
|
||||
original_file.with_name(f"{new_stem}{original_file.suffix}")
|
||||
)
|
||||
original_file.rename(new_path)
|
||||
logger.info(
|
||||
f"yt-dlp 下载完成, 标题: {title}, "
|
||||
f"作者: {uploader}, 重命名: {new_path}"
|
||||
)
|
||||
|
||||
return new_path
|
||||
Reference in New Issue
Block a user