Add HeXi bot codebase: custom plugins, web frontends, tests

- hexi core: message handling, rate limiting, cooldown, plugin manager
- Custom plugins: BF stats, daily check-in, quotes, persona cards, etc.
- Community plugins vendored under hexi/plugins with local fixes
- Web admin frontends (learning-chat, persona-admin), unified hexi/web
- Tests for rate_limit/cooldown/memes/persona; poetry.lock

Co-Authored-By: Claude <noreply@anthropic.com>
This commit is contained in:
2026-09-01 13:13:40 +08:00
co-authored by Claude
parent 1783c60afa
commit b61d09f09f
3201 changed files with 160436 additions and 171 deletions
@@ -0,0 +1,264 @@
"""通用视频下载 — yt-dlp + 直链探测"""
import asyncio
import os
import re
import sys
import tempfile
from datetime import datetime
from pathlib import Path
from typing import Optional
from httpx import AsyncClient
from nonebot import logger
from yt_dlp import YoutubeDL
from yt_dlp.utils import DownloadError
from ..utils import get_temp_root, slugify, ensure_unique_path
def detect_platform(url: str) -> str:
if "bilibili.com" in url or "b23.tv" in url:
return "bilibili"
if "twitter.com" in url or "x.com" in url:
return "twitter"
if "youtube.com" in url or "youtu.be" in url:
return "youtube"
if "douyin.com" in url or "v.douyin.com" in url or "iesdouyin.com" in url:
return "douyin"
return "other"
def extract_uploader(info: dict) -> Optional[str]:
"""从 yt-dlp info dict 提取上传者,优先级: uploader > channel > creator > uploader_id"""
if not info:
return None
return (
info.get("uploader")
or info.get("channel")
or info.get("creator")
or info.get("uploader_id")
)
def get_ffmpeg_path() -> str:
scripts_dir = os.path.dirname(sys.executable)
ffmpeg_path = os.path.join(scripts_dir, "ffmpeg.exe")
if os.path.exists(ffmpeg_path):
return ffmpeg_path
return "ffmpeg"
def _get_data_dir() -> Path:
"""获取 data/ 目录路径"""
return Path(__file__).resolve().parent.parent / "data"
async def _retry_download(
loop: asyncio.AbstractEventLoop,
url: str,
base_opts: dict,
max_retries: int = 3,
) -> dict:
"""带重试的 yt-dlp 下载,处理 RemoteDisconnected 等瞬态错误"""
def _run_yt():
with YoutubeDL(base_opts) as ydl:
return ydl.extract_info(url, download=True)
last_error = None
for attempt in range(1, max_retries + 1):
try:
return await loop.run_in_executor(None, _run_yt)
except DownloadError as e:
last_error = e
if attempt < max_retries:
delay = 2 ** attempt # 2s, 4s, 8s
logger.warning(
f"yt-dlp 下载失败 (第 {attempt}/{max_retries} 次),"
f"{delay}s 后重试: {str(e)[:120]}"
)
await asyncio.sleep(delay)
else:
logger.error(
f"yt-dlp 重试 {max_retries} 次后仍失败: {str(e)[:120]}"
)
except Exception as e:
# 非 DownloadError(如 OSError)不重试,直接抛出
raise
raise last_error # type: ignore[misc]
async def download_video(url: str) -> Optional[Path]:
"""下载视频,支持直链和 yt-dlp"""
# ---------- 1. 直链探测 ----------
direct_media_ext = re.search(
r"\.(mp4|m3u8|ts|webm|mov|flv)(?:$|\?)", url, re.IGNORECASE
)
is_direct = bool(direct_media_ext)
if not is_direct:
try:
async with AsyncClient(follow_redirects=True, timeout=30) as client:
head = await client.head(url, follow_redirects=True)
ctype = head.headers.get("content-type", "")
if ctype.startswith("video/") or "application/octet-stream" in ctype:
is_direct = True
except Exception:
is_direct = False
if is_direct:
temp_dir = tempfile.mkdtemp(prefix="direct_ytcache_", dir=get_temp_root("ytcache"))
ext = "mp4"
m = re.search(r"\.([a-zA-Z0-9]{2,5})(?:$|\?)", url)
if m and len(m.group(1)) <= 5:
ext = m.group(1)
url_stem = Path(url.split("?")[0]).stem or "video"
slug_stem = slugify(url_stem, max_length=15)
if not slug_stem:
slug_stem = datetime.now().strftime("%H%M%S")
time_suffix = datetime.now().strftime("%H%M%S")
new_name = f"{slug_stem}_视频_{time_suffix}.{ext}"
filename = os.path.join(temp_dir, new_name)
try:
async with AsyncClient(follow_redirects=True, timeout=300) as client:
async with client.stream("GET", url) as resp:
resp.raise_for_status()
with open(filename, "wb") as fh:
async for chunk in resp.aiter_bytes(chunk_size=8192):
fh.write(chunk)
final_path = ensure_unique_path(Path(filename))
logger.info(f"直接下载完成: {final_path}")
return final_path
except Exception:
logger.exception("直接下载失败,回退 yt-dlp")
if os.path.exists(filename):
os.remove(filename)
# ---------- 2. yt-dlp 下载 ----------
platform = detect_platform(url)
temp_dir = tempfile.mkdtemp(prefix="ytcache_", dir=get_temp_root("ytcache"))
output_path = os.path.join(temp_dir, "%(title).80s.%(ext)s")
base_opts = {
"outtmpl": output_path,
"format": "bestvideo+bestaudio/best",
"merge_output_format": "mp4",
"noplaylist": True,
"quiet": True,
"ffmpeg_location": get_ffmpeg_path(),
}
cookie_path = _get_data_dir() / "cookies.txt"
if platform in ("bilibili", "twitter", "youtube"):
if not cookie_path.exists():
raise RuntimeError(f"{platform} 需要 cookies.txt,但未找到")
base_opts["cookiefile"] = str(cookie_path)
logger.info(f"{platform} 使用 cookies.txt 下载")
ua = (
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) "
"AppleWebKit/537.36 (KHTML, like Gecko) "
"Chrome/122.0.0.0 Safari/537.36"
)
if platform == "bilibili":
base_opts["http_headers"] = {
"User-Agent": ua,
"Referer": "https://www.bilibili.com/",
"Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,image/webp,*/*;q=0.8",
"Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
"Accept-Encoding": "gzip, deflate, br",
}
base_opts["extractor_args"] = {
"bilibili": {
"header": [
"Referer:https://www.bilibili.com/",
f"User-Agent:{ua}",
"Accept:text/html,application/xhtml+xml,application/xml;q=0.9,image/webp,*/*;q=0.8",
"Accept-Language:zh-CN,zh;q=0.9,en;q=0.8",
]
}
}
elif platform == "twitter":
base_opts["http_headers"] = {
"User-Agent": ua,
"Referer": "https://x.com/",
"Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,image/webp,*/*;q=0.8",
"Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
}
base_opts["extractor_args"] = {"twitter": {"api": ["syndication"]}}
elif platform == "youtube":
base_opts["http_headers"] = {
"User-Agent": ua,
"Referer": "https://www.youtube.com/",
"Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,image/webp,*/*;q=0.8",
"Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
}
else:
logger.info("其他平台,默认无 cookies 下载")
loop = asyncio.get_event_loop()
try:
info = await _retry_download(loop, url, base_opts)
except Exception:
logger.exception("yt-dlp 下载失败")
# YouTube: cookies 可能触发 bot 检测导致只返回图片无视频格式
# 回退无 cookie 模式重试
if platform == "youtube" and "cookiefile" in base_opts:
logger.info("YouTube 回退无 cookies 模式重试...")
base_opts.pop("cookiefile", None)
base_opts.pop("http_headers", None)
# 清理失败残留
for f in Path(temp_dir).glob("*.*"):
try:
f.unlink()
except Exception:
pass
try:
info = await _retry_download(loop, url, base_opts, max_retries=2)
except Exception:
logger.exception("yt-dlp 无 cookies 重试也失败")
return None
else:
return None
files = list(Path(temp_dir).glob("*.*"))
if not files:
return None
original_file = files[0]
# 构建新文件名
uploader = extract_uploader(info or {})
title = ((info or {}).get("title") or "").strip()
slug_title = slugify(title, max_length=15) if title else ""
if not slug_title:
slug_title = datetime.now().strftime("%H%M%S")
time_suffix = datetime.now().strftime("%H%M%S")
if uploader:
slug_uploader = slugify(str(uploader))
new_stem = f"{slug_uploader}_{slug_title}_视频_{time_suffix}"
else:
new_stem = f"{slug_title}_视频_{time_suffix}"
new_path = ensure_unique_path(
original_file.with_name(f"{new_stem}{original_file.suffix}")
)
original_file.rename(new_path)
logger.info(
f"yt-dlp 下载完成, 标题: {title}, "
f"作者: {uploader}, 重命名: {new_path}"
)
return new_path