Add HeXi bot codebase: custom plugins, web frontends, tests
- hexi core: message handling, rate limiting, cooldown, plugin manager - Custom plugins: BF stats, daily check-in, quotes, persona cards, etc. - Community plugins vendored under hexi/plugins with local fixes - Web admin frontends (learning-chat, persona-admin), unified hexi/web - Tests for rate_limit/cooldown/memes/persona; poetry.lock Co-Authored-By: Claude <noreply@anthropic.com>
This commit is contained in:
@@ -0,0 +1,9 @@
|
||||
"""handlers 包:消息入口与各平台解析处理。"""
|
||||
|
||||
from . import douyin, sender, universal # noqa: F401
|
||||
from .entry import ( # noqa: F401
|
||||
active_video_handler,
|
||||
auto_video_handler,
|
||||
dispatch_url,
|
||||
match_message,
|
||||
)
|
||||
@@ -0,0 +1,131 @@
|
||||
"""抖音视频/图文解析编排层"""
|
||||
|
||||
import os
|
||||
import re
|
||||
import urllib.parse
|
||||
from pathlib import Path
|
||||
from typing import Optional, Union
|
||||
|
||||
import httpx
|
||||
from nonebot import logger
|
||||
|
||||
from ..fetchers.douyin_api import fetch_douyin_content
|
||||
from ..fetchers.douyin_ssr import MOBILE_UA, fetch_douyin_note_ssr
|
||||
from ..models import DouyinFetchError
|
||||
from ..utils import parse_netscape_cookies
|
||||
from .sender import PendingMedia, _as_paths
|
||||
|
||||
SHORT_LINK_PATTERN = re.compile(r"(v\.douyin\.com/[A-Za-z0-9_\-]+)")
|
||||
|
||||
|
||||
def _get_data_dir() -> str:
|
||||
return os.path.join(os.path.dirname(__file__), "..", "data")
|
||||
|
||||
|
||||
async def _resolve_short_link(url: str) -> Optional[str]:
|
||||
"""短链重定向:httpx 取 Location(最多 3 跳),失败返回 None"""
|
||||
try:
|
||||
async with httpx.AsyncClient(
|
||||
headers={"User-Agent": MOBILE_UA},
|
||||
follow_redirects=False,
|
||||
timeout=10,
|
||||
) as client:
|
||||
current = url
|
||||
for _ in range(3):
|
||||
resp = await client.get(current)
|
||||
if resp.status_code >= 400:
|
||||
return None
|
||||
location = resp.headers.get("Location")
|
||||
if not location:
|
||||
final = str(resp.url)
|
||||
# 200 但仍是短链本身(如 JS 挑战壳页)→ 静默失败,显式记录;
|
||||
# 图文短链若此处失败将退化到 playwright 旧链路
|
||||
if "v.douyin.com" in final:
|
||||
logger.warning(f"短链返回挑战页(未跳转): {final}")
|
||||
return final
|
||||
# Location 可能是相对路径,需拼上当前 URL
|
||||
current = urllib.parse.urljoin(current, location)
|
||||
return current
|
||||
except Exception:
|
||||
logger.warning(f"短链重定向失败: {url}")
|
||||
return None
|
||||
|
||||
|
||||
async def parse_douyin(
|
||||
url: str,
|
||||
) -> tuple[Optional[str], Optional[Union[Path, list[Path]]], bool]:
|
||||
"""解析抖音链接(短链 / 全链接),返回 (title, file_paths, is_image_post)
|
||||
|
||||
分派规则(不能依赖 URL 路径:短链重定向后图文也统一变成
|
||||
iesdouyin.com/share/video/{id} 形态):
|
||||
- note 形态链接 → SSR 静态解析,失败不回退 playwright
|
||||
(其图文链路存在 aweme/post 取到作者其他作品的缺陷)
|
||||
- 其余形态 → 先试 SSR 按内容判定:有 images 即图文 → SSR 秒级下载;
|
||||
真视频 / SSR 失败 → playwright 拦截(保留全部清晰度能力)
|
||||
"""
|
||||
# 优先匹配短链接 v.douyin.com/xxx
|
||||
short = SHORT_LINK_PATTERN.search(url)
|
||||
if short:
|
||||
target_url = "https://" + short.group(1)
|
||||
# 短链先重定向拿到全链接,确定图文/视频类型后分派
|
||||
resolved = await _resolve_short_link(target_url)
|
||||
if resolved:
|
||||
logger.info(f"短链重定向: {target_url} -> {resolved}")
|
||||
target_url = resolved
|
||||
else:
|
||||
# 全链接: www.douyin.com/note/xxx 或 www.douyin.com/video/xxx
|
||||
m = re.search(
|
||||
r"(https?://(?:www\.)?douyin\.com/(?:note|video)/\d+)", url
|
||||
)
|
||||
if not m:
|
||||
return None, None, False
|
||||
target_url = m.group(1)
|
||||
|
||||
try:
|
||||
is_img_post = False
|
||||
cookies_path = os.path.join(_get_data_dir(), "cookies.txt")
|
||||
cookies = parse_netscape_cookies(cookies_path)
|
||||
# 先试 SSR(秒级、免浏览器),失败一律回退 playwright 链路。
|
||||
# 2026-08-13 起抖音对 iesdouyin SSR 端点整体降级(登录态也拿不到
|
||||
# videoInfoRes,只返回 33KB 壳页),note/video 形态统一回退,
|
||||
# 图文作品的"取到作者其他作品"缺陷由 douyin_api 按 aweme_id 精确匹配修复。
|
||||
try:
|
||||
title, file_path = await fetch_douyin_note_ssr(target_url, cookies)
|
||||
except DouyinFetchError:
|
||||
logger.info("SSR 解析失败,回退 playwright 链路")
|
||||
title, file_path = None, None
|
||||
if file_path is None:
|
||||
title, file_path = await fetch_douyin_content(
|
||||
target_url, cookies, 10, False
|
||||
)
|
||||
if isinstance(file_path, list):
|
||||
is_img_post = True
|
||||
return title, file_path, is_img_post
|
||||
except Exception:
|
||||
logger.exception(f"获取直链失败 {target_url}")
|
||||
return None, None, False
|
||||
|
||||
|
||||
async def process_douyin_res(
|
||||
title: str,
|
||||
file_paths: Union[Path, list[Path]],
|
||||
is_private: bool,
|
||||
image_post: bool,
|
||||
plan: str | None = None,
|
||||
) -> tuple[Optional[PendingMedia], Optional[str]]:
|
||||
"""下载已完成 → 打包为待发送媒体(不上传、不发送、不清理)
|
||||
|
||||
多级发送(temp 本地 → S3 链接 → 回退本地)由 send_pending_media 统一处理。
|
||||
"""
|
||||
if not file_paths:
|
||||
return None, None
|
||||
return (
|
||||
PendingMedia(
|
||||
files=_as_paths(file_paths),
|
||||
image_post=image_post,
|
||||
is_private=is_private,
|
||||
plan=plan,
|
||||
title=title,
|
||||
),
|
||||
None,
|
||||
)
|
||||
@@ -0,0 +1,266 @@
|
||||
"""消息入口与分派:自动解析 / 主动解析(文本链接 + QQ小程序/分享卡片)。"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import html
|
||||
import os
|
||||
import re
|
||||
from typing import Optional
|
||||
|
||||
from nonebot import on_message, logger
|
||||
from nonebot.adapters import Event
|
||||
from nonebot.rule import to_me
|
||||
from nonebot_plugin_alconna import UniMessage
|
||||
from nonebot_plugin_alconna.uniseg import get_target
|
||||
|
||||
from ..fetchers.bilibili_content import fetch_bilibili_content, resolve_short_link
|
||||
from ..fetchers.rednote_content import fetch_rednote_content
|
||||
from .douyin import parse_douyin, process_douyin_res
|
||||
from .sender import PendingMedia, send_pending_media
|
||||
from .universal import handle_universal
|
||||
from ..list_proc import AUTO_LINK_KEYWORDS, get_group_auto_link, verify_user
|
||||
|
||||
BASE_DIR = os.path.dirname(os.path.abspath(__file__))
|
||||
FILE_PATH = os.path.join(BASE_DIR, "data", "list.json")
|
||||
|
||||
URL_PATTERN = re.compile(r"(https?://\S+)")
|
||||
XCX_PATTERN = r"QQ小程序(?:&#93;|]|\])"
|
||||
|
||||
VALID_HOSTS = [
|
||||
"b23.tv",
|
||||
"bilibili.com",
|
||||
"youtube.com",
|
||||
"youtu.be",
|
||||
"douyin.com",
|
||||
"v.douyin.com",
|
||||
"iesdouyin.com",
|
||||
"m.douyin.com",
|
||||
"jingxuan.douyin.com",
|
||||
"x.com",
|
||||
"twitter.com",
|
||||
"xiaohongshu.com",
|
||||
"xhslink.com",
|
||||
"xhslink.cn",
|
||||
]
|
||||
|
||||
auto_video_handler = on_message(priority=10, block=False)
|
||||
active_video_handler = on_message(priority=10, block=False, rule=to_me())
|
||||
|
||||
|
||||
async def _check_access(
|
||||
event: Event, *, auto_only: bool = False, msg: str | None = None
|
||||
) -> tuple[bool, str | None]:
|
||||
"""统一权限检查。
|
||||
|
||||
auto_only=True → 自动解析:需白名单 + 开启自动解析,或 auto_link 关键词命中。
|
||||
auto_only=False → 主动触发:需非黑名单,群聊还需白名单。
|
||||
|
||||
Returns:
|
||||
(allowed, plan) — plan 用于 S3 路由,不允许时为 None
|
||||
"""
|
||||
white, black, auto, plan = await verify_user(event)
|
||||
target = get_target(event)
|
||||
|
||||
if target.private:
|
||||
if auto_only:
|
||||
logger.info("权限分析:自动解析不处理私聊")
|
||||
return False, None
|
||||
if black:
|
||||
logger.info(f"权限分析:黑名单用户私聊,不回复: {event.get_user_id()}")
|
||||
return False, None
|
||||
logger.info("权限分析:私聊,直接解析")
|
||||
return True, None
|
||||
|
||||
group_id = str(event.group_id)
|
||||
|
||||
if not white:
|
||||
logger.info(f"权限分析:群 {group_id} 不在白名单,不做处理")
|
||||
return False, None
|
||||
|
||||
if auto_only and not auto:
|
||||
if msg is not None and await _match_auto_link(event, msg):
|
||||
logger.info(f"权限分析:群 {group_id} 未开启自动解析,但自动链接关键词命中")
|
||||
else:
|
||||
logger.info(f"权限分析:群 {group_id} 未开启自动解析")
|
||||
return False, None
|
||||
|
||||
if not auto_only and black:
|
||||
logger.info(f"权限分析:黑名单用户,不回复: {event.get_user_id()}")
|
||||
return False, None
|
||||
|
||||
logger.info(
|
||||
f"权限分析:群 {group_id} 权限通过 — "
|
||||
f"自动解析: {auto}, 方案: {plan or '默认(PLANC)'}"
|
||||
)
|
||||
return True, plan
|
||||
|
||||
|
||||
async def _match_auto_link(event: Event, msg: str) -> bool:
|
||||
"""消息中的 URL 是否命中群配置的 auto_link 关键词。"""
|
||||
keywords = await get_group_auto_link(event)
|
||||
if not keywords:
|
||||
return False
|
||||
urls = URL_PATTERN.findall(msg)
|
||||
if not urls:
|
||||
return False
|
||||
for kw in keywords:
|
||||
domains = AUTO_LINK_KEYWORDS.get(kw, (kw,))
|
||||
if any(any(domain in url for domain in domains) for url in urls):
|
||||
logger.info(f"自动链接:关键词 {kw} 命中消息 {urls}")
|
||||
return True
|
||||
return False
|
||||
|
||||
|
||||
@auto_video_handler.handle()
|
||||
async def handle_auto_video(event: Event):
|
||||
msg = str(event.get_message()).strip()
|
||||
allowed, plan = await _check_access(event, auto_only=True, msg=msg)
|
||||
if allowed:
|
||||
await match_message(event, plan=plan)
|
||||
|
||||
|
||||
@active_video_handler.handle()
|
||||
async def handle_active_video(event: Event):
|
||||
allowed, plan = await _check_access(event, auto_only=False)
|
||||
if allowed:
|
||||
await match_message(event, plan=plan)
|
||||
|
||||
|
||||
async def match_message(event: Event, plan: str | None = None):
|
||||
"""消息匹配与分派:文本链接 / QQ小程序卡片统一走 dispatch_url。"""
|
||||
msg = str(event.get_message()).strip()
|
||||
logger.info(f"消息解析:获取到的消息:{msg}")
|
||||
is_private = get_target(event).private
|
||||
|
||||
message = None
|
||||
public_url = None
|
||||
|
||||
if re.search(XCX_PATTERN, msg) or "CQ:json" in msg or "CQ:share" in msg:
|
||||
logger.info("消息解析:检测到 CQ 卡片")
|
||||
url = await _extract_xcx_url(msg)
|
||||
logger.info(f"消息解析:卡片链接:{url}")
|
||||
if not url or not any(domain in url for domain in VALID_HOSTS):
|
||||
return
|
||||
message, public_url = await dispatch_url(url, is_private, plan=plan)
|
||||
if not message:
|
||||
return
|
||||
else:
|
||||
urls = URL_PATTERN.findall(msg)
|
||||
for url in urls:
|
||||
logger.info(f"消息解析:作品链接:{url}")
|
||||
message, public_url = await dispatch_url(url, is_private, plan=plan)
|
||||
if message:
|
||||
break
|
||||
if not message:
|
||||
return
|
||||
|
||||
if isinstance(message, PendingMedia):
|
||||
ok, pub = await send_pending_media(message)
|
||||
if pub:
|
||||
await UniMessage.text(f"{pub}").send()
|
||||
if not ok:
|
||||
await UniMessage.text(f"媒体发送失败:{url}").send()
|
||||
else:
|
||||
if public_url:
|
||||
await UniMessage.text(f"{public_url}").send()
|
||||
await message.send()
|
||||
|
||||
|
||||
async def dispatch_url(
|
||||
url: str,
|
||||
is_private: bool,
|
||||
plan: str | None = None,
|
||||
) -> tuple[Optional[UniMessage], Optional[str]]:
|
||||
"""按平台分派解析(文本链接与小程序卡片共用)。"""
|
||||
url = url.rstrip(",。!?、;:)】》\"')")
|
||||
|
||||
if "b23.tv" in url or "bili2233.cn" in url:
|
||||
resolved = await resolve_short_link(url)
|
||||
if resolved:
|
||||
logger.info(f"b23 短链重定向: {url} -> {resolved}")
|
||||
url = resolved
|
||||
|
||||
if "douyin.com" in url or "v.douyin.com" in url or "iesdouyin.com" in url:
|
||||
await UniMessage.text("检测到链接,正在处理,请稍候...").send()
|
||||
try:
|
||||
title, parsed_path, image_post = await parse_douyin(url)
|
||||
return await process_douyin_res(
|
||||
title, parsed_path, is_private, image_post, plan=plan
|
||||
)
|
||||
except Exception:
|
||||
await UniMessage.text(f"无法解析到媒体:{url}").send()
|
||||
logger.exception(f"媒体解析:无法解析到媒体:{url}")
|
||||
return None, None
|
||||
|
||||
if any(
|
||||
kw in url
|
||||
for kw in (
|
||||
"bilibili.com/opus",
|
||||
"bilibili.com/dynamic",
|
||||
"t.bilibili.com",
|
||||
"bilibili.com/read",
|
||||
)
|
||||
):
|
||||
await UniMessage.text("检测到链接,正在处理,请稍候...").send()
|
||||
try:
|
||||
title, parsed_path = await fetch_bilibili_content(url)
|
||||
if not title and parsed_path is None:
|
||||
await UniMessage.text(f"无法解析到内容:{url}").send()
|
||||
return None, None
|
||||
if parsed_path == []:
|
||||
await UniMessage.text(title).send()
|
||||
return None, None
|
||||
return await process_douyin_res(
|
||||
title, parsed_path, is_private,
|
||||
isinstance(parsed_path, list), plan=plan,
|
||||
)
|
||||
except Exception:
|
||||
await UniMessage.text(f"无法解析到内容:{url}").send()
|
||||
logger.exception(f"B站内容解析:无法解析:{url}")
|
||||
return None, None
|
||||
|
||||
if "xiaohongshu.com" in url or "xhslink." in url:
|
||||
await UniMessage.text("检测到链接,正在处理,请稍候...").send()
|
||||
try:
|
||||
title, parsed_path = await fetch_rednote_content(url)
|
||||
if not title and parsed_path is None:
|
||||
await UniMessage.text(f"无法解析到内容:{url}").send()
|
||||
return None, None
|
||||
if parsed_path == []:
|
||||
await UniMessage.text(title).send()
|
||||
return None, None
|
||||
return await process_douyin_res(
|
||||
title, parsed_path, is_private,
|
||||
isinstance(parsed_path, list), plan=plan,
|
||||
)
|
||||
except Exception:
|
||||
await UniMessage.text(f"无法解析到内容:{url}").send()
|
||||
logger.exception(f"小红书解析:无法解析:{url}")
|
||||
return None, None
|
||||
|
||||
if any(domain in url for domain in VALID_HOSTS):
|
||||
await UniMessage.text("检测到链接,正在处理,请稍候...").send()
|
||||
try:
|
||||
return await handle_universal(url, is_private, plan=plan)
|
||||
except Exception as e:
|
||||
logger.exception(e)
|
||||
await UniMessage.text("下载过程中出现错误。").send()
|
||||
return None, None
|
||||
|
||||
return None, None
|
||||
|
||||
|
||||
async def _extract_xcx_url(msg: str) -> Optional[str]:
|
||||
"""从 CQ 卡片消息中提取跳转 URL(保留 query 参数)。"""
|
||||
match = re.search(r'"qqdocurl":"(.*?)"', msg)
|
||||
if not match:
|
||||
match = re.search(r'"jumpUrl":"(.*?)"', msg)
|
||||
if not match:
|
||||
logger.warning("未找到 qqdocurl/jumpUrl 字段")
|
||||
return None
|
||||
|
||||
raw_url = match.group(1)
|
||||
unescaped = html.unescape(raw_url)
|
||||
cleaned_url = unescaped.replace(r"\/", "/")
|
||||
logger.info(f"卡片提取链接: {cleaned_url}")
|
||||
return cleaned_url
|
||||
@@ -0,0 +1,107 @@
|
||||
"""媒体多级发送 — temp 本地文件优先,失败逐级降级
|
||||
|
||||
发送策略(2026-08-22 用户需求):
|
||||
1. temp 本地文件直接发送(最快,不经 S3)
|
||||
2. 失败 → 上传本地 S3,用预签名链接发送
|
||||
3. 再失败 → 回退 temp 本地文件再发一次
|
||||
|
||||
temp 下的文件发送成功后也不清理(用户手动处理 data/temp)。
|
||||
"""
|
||||
|
||||
from dataclasses import dataclass
|
||||
from pathlib import Path
|
||||
from typing import Optional, Union
|
||||
|
||||
from nonebot import logger
|
||||
from nonebot_plugin_alconna import UniMessage
|
||||
|
||||
from ..storage.s3 import upload_with_plan
|
||||
|
||||
|
||||
@dataclass
|
||||
class PendingMedia:
|
||||
"""待发送媒体:本地文件 + 上传元数据(发送前不做任何上传/清理)"""
|
||||
|
||||
files: list[Path]
|
||||
image_post: bool = False
|
||||
is_private: bool = False
|
||||
plan: Optional[str] = None
|
||||
title: str = ""
|
||||
|
||||
|
||||
def _as_paths(file_paths: Union[Path, list[Path]]) -> list[Path]:
|
||||
if isinstance(file_paths, list):
|
||||
return [Path(p) for p in file_paths]
|
||||
return [Path(file_paths)]
|
||||
|
||||
|
||||
def _build_local_msg(files: list[Path], image_post: bool) -> UniMessage:
|
||||
"""本地文件版消息(mp4 → 视频,其余 → 图片)"""
|
||||
msg = UniMessage()
|
||||
for fp in files:
|
||||
if fp.suffix.lower() == ".mp4":
|
||||
msg.video(path=fp)
|
||||
else:
|
||||
msg.image(path=fp)
|
||||
return msg
|
||||
|
||||
|
||||
def _build_s3_msg(
|
||||
media: PendingMedia,
|
||||
) -> tuple[UniMessage, Optional[str]]:
|
||||
"""上传本地 S3 并构建链接版消息,返回 (message, public_url)"""
|
||||
msg = UniMessage()
|
||||
public_url = None
|
||||
for fp in media.files:
|
||||
local_url, pub = upload_with_plan(
|
||||
fp,
|
||||
plan=media.plan,
|
||||
is_private=media.is_private,
|
||||
title=media.title,
|
||||
image_post=media.image_post,
|
||||
)
|
||||
if not local_url:
|
||||
raise RuntimeError(f"上传本地 S3 失败: {fp}")
|
||||
if pub:
|
||||
public_url = pub
|
||||
if fp.suffix.lower() == ".mp4":
|
||||
msg.video(url=local_url)
|
||||
else:
|
||||
msg.image(url=local_url)
|
||||
return msg, public_url
|
||||
|
||||
|
||||
async def send_pending_media(media: PendingMedia) -> tuple[bool, Optional[str]]:
|
||||
"""多级发送,返回 (是否成功, public_url)
|
||||
|
||||
public_url 仅在走 S3 链接发送成功时返回(调用方决定是否发文字)。
|
||||
temp 文件发送成功后保留(用户手动清理 data/temp)。
|
||||
"""
|
||||
if not media.files:
|
||||
return False, None
|
||||
|
||||
# ── 1. temp 本地文件直接发送 ──────────────────────────────
|
||||
try:
|
||||
await _build_local_msg(media.files, media.image_post).send()
|
||||
logger.info("媒体发送成功(temp 本地文件直达)")
|
||||
return True, None
|
||||
except Exception as e:
|
||||
logger.warning(f"temp 本地文件发送失败,切换本地 S3 链接: {e}")
|
||||
|
||||
# ── 2. 上传本地 S3 → 链接发送 ─────────────────────────────
|
||||
try:
|
||||
s3_msg, public_url = _build_s3_msg(media)
|
||||
await s3_msg.send()
|
||||
logger.info("媒体发送成功(本地 S3 链接)")
|
||||
return True, public_url
|
||||
except Exception as e:
|
||||
logger.warning(f"本地 S3 链接发送失败,回退 temp 本地文件: {e}")
|
||||
|
||||
# ── 3. 回退:temp 本地文件再发一次 ────────────────────────
|
||||
try:
|
||||
await _build_local_msg(media.files, media.image_post).send()
|
||||
logger.info("媒体发送成功(回退 temp 本地文件)")
|
||||
return True, None
|
||||
except Exception as e:
|
||||
logger.exception(f"回退发送失败: {e}")
|
||||
return False, None
|
||||
@@ -0,0 +1,40 @@
|
||||
"""通用平台视频解析编排层 — B站 / YouTube / Twitter 等"""
|
||||
|
||||
from pathlib import Path
|
||||
from typing import Optional
|
||||
|
||||
from nonebot import logger
|
||||
from nonebot_plugin_alconna import UniMessage
|
||||
|
||||
from ..fetchers.video_downloader import download_video
|
||||
from .sender import PendingMedia, _as_paths
|
||||
|
||||
|
||||
async def handle_universal(
|
||||
url: str,
|
||||
is_private: bool,
|
||||
plan: str | None = None,
|
||||
) -> tuple[Optional[PendingMedia], Optional[str]]:
|
||||
"""
|
||||
下载通用平台视频 → 打包待发送媒体(上传/发送由 sender 多级处理)
|
||||
|
||||
Returns:
|
||||
(PendingMedia, public_url) — None 表示下载失败
|
||||
"""
|
||||
video_file = await download_video(url)
|
||||
if not video_file:
|
||||
await UniMessage.text("视频下载失败。").send()
|
||||
return None, None
|
||||
|
||||
logger.info(f"文件路径:{video_file}")
|
||||
|
||||
return (
|
||||
PendingMedia(
|
||||
files=_as_paths(video_file),
|
||||
image_post=False,
|
||||
is_private=is_private,
|
||||
plan=plan,
|
||||
title="title",
|
||||
),
|
||||
None,
|
||||
)
|
||||
Reference in New Issue
Block a user