Files
HeXi/hexi/plugins/nonebot_plugin_video_analysis/handlers/entry.py
T
sansenhoshiandClaude Code 4badcfcf32 feat(video-analysis): 群策略 v3 / 群文件投递通道 / Web 管理页
- policy.py:per-group 正交策略(自动解析 / 自动策略 / 禁用策略 / 存储 A·B·C /
  公网 / 链接 / 群文件 + 平台限定),list.json v1/v2 → v3 自动迁移,
  写入统一走 PolicyStore(加锁 + .tmp 原子替换 + 字段归一)
- 群文件并行通道 group_file.py:打包 zip(可选 pyzipper AES-256)后优先走 S3 预签名、
  本地直传兜底;设了密码但 pyzipper 不可用就放弃上传,不退化成明文
- list_proc.py 收敛到「视频策略」统一入口,权限判定改走 policy
- Web 管理页 /hub/video_analysis(群策略 + 链接解析面板)与 services/web_jobs.py
  (只复用纯函数层,Web 上下文不发消息;内存任务表 + 并发闸门 + 超时)
- 媒体命名统一到 utils.py({作者}_{作者id}/{作品名}[_短码]),cleanup 回收空目录
- 测试:policy / 命名 / 群文件 / web_jobs 四组

顺带 pyproject 的 pytest 加 testpaths=tests(避免收进 debug/ 下的调试脚本)。

Co-Authored-By: Claude Code <noreply@anthropic.com>
2026-09-22 14:23:32 +08:00

286 lines
10 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""消息入口与分派:自动解析 / 主动解析(文本链接 + QQ小程序/分享卡片)。"""
from __future__ import annotations
import html
import re
from nonebot import on_message, logger
from nonebot.adapters import Event
from nonebot.rule import to_me
from nonebot_plugin_alconna import UniMessage
from nonebot_plugin_alconna.uniseg import get_target
from ..services.fetchers.bilibili_content import fetch_bilibili_content, resolve_short_link
from ..services.fetchers.rednote_content import fetch_rednote_content
from .douyin import parse_douyin, process_douyin_res
from .sender import PendingMedia, send_pending_media
from .universal import handle_universal
from ..list_proc import get_policy, is_group_whitelisted, is_user_blacklisted
from ..policy import Policy, match_platform
URL_PATTERN = re.compile(r"(https?://\S+)")
XCX_PATTERN = r"QQ小程序(?:&amp;#93;|&#93;|\])"
VALID_HOSTS = [
"b23.tv",
"bilibili.com",
"youtube.com",
"youtu.be",
"douyin.com",
"v.douyin.com",
"iesdouyin.com",
"m.douyin.com",
"jingxuan.douyin.com",
"x.com",
"twitter.com",
"xiaohongshu.com",
"xhslink.com",
"xhslink.cn",
]
auto_video_handler = on_message(priority=10, block=False)
active_video_handler = on_message(priority=10, block=False, rule=to_me())
async def _check_access(
event: Event, *, auto_only: bool = False, msg: str | None = None
) -> tuple[bool, Policy | None]:
"""统一权限检查。
auto_only=True → 自动解析:需白名单群 + (开启自动解析或消息命中自动策略)。
auto_only=False → 主动触发:需白名单群 + 非黑名单用户。
私聊不做自动解析,读 default 节策略后直接解析。
禁用策略(ban_link)是消息级的(按链接判定),见 match_message。
Returns:
(allowed, policy) — policy 决定存储/投递,不允许时为 None
"""
target = get_target(event)
if target.private:
if auto_only:
logger.info("权限分析:自动解析不处理私聊")
return False, None
if is_user_blacklisted(event):
logger.info(f"权限分析:黑名单用户私聊,不回复: {event.get_user_id()}")
return False, None
logger.info("权限分析:私聊,直接解析")
return True, get_policy(event)
group_id = str(event.group_id)
if not is_group_whitelisted(event):
logger.info(f"权限分析:群 {group_id} 不在白名单,不做处理")
return False, None
if is_user_blacklisted(event):
logger.info(f"权限分析:黑名单用户,不回复: {event.get_user_id()}")
return False, None
policy = get_policy(event)
if auto_only and not policy.auto and not _match_auto_link(msg or "", policy):
logger.info(f"权限分析:群 {group_id} 未开启自动解析且未命中自动策略")
return False, None
logger.info(
f"权限分析:群 {group_id} 权限通过 — 自动解析: {policy.auto}, "
f"存储: {policy.plan}, 公网: {policy.upload_public}, "
f"群文件: {policy.upload_group_file}"
)
return True, policy
def _match_auto_link(msg: str, policy: Policy) -> bool:
"""消息中的 URL 是否命中群策略的自动策略(auto_link)。"""
if not policy.auto_link:
return False
for url in URL_PATTERN.findall(msg):
platform = policy.auto_matched(url)
if platform:
logger.info(f"自动策略:{platform} 命中消息 {url}")
return True
return False
def _skip_banned(url: str, policy: Policy) -> bool:
"""链接是否命中禁用策略(命中即静默丢弃,只记日志)。"""
platform = policy.banned(url)
if platform:
logger.info(f"禁用策略:{platform} 已禁用,忽略链接 {url}")
return True
return False
@auto_video_handler.handle()
async def handle_auto_video(event: Event):
msg = str(event.get_message()).strip()
allowed, policy = await _check_access(event, auto_only=True, msg=msg)
if allowed and policy is not None:
await match_message(event, policy)
@active_video_handler.handle()
async def handle_active_video(event: Event):
allowed, policy = await _check_access(event, auto_only=False)
if allowed and policy is not None:
await match_message(event, policy)
async def match_message(event: Event, policy: Policy):
"""消息匹配与分派:文本链接 / QQ小程序卡片统一走 dispatch_url。
命中禁用策略的链接在这里丢弃;消息里还有其它可用链接则继续解析。
"""
msg = str(event.get_message()).strip()
logger.info(f"消息解析:获取到的消息:{msg}")
is_private = get_target(event).private
message = None
public_url = None
url = ""
if re.search(XCX_PATTERN, msg) or "CQ:json" in msg or "CQ:share" in msg:
logger.info("消息解析:检测到 CQ 卡片")
url = await _extract_xcx_url(msg) or ""
logger.info(f"消息解析:卡片链接:{url}")
if not url or not any(domain in url for domain in VALID_HOSTS):
return
if _skip_banned(url, policy):
return
message, public_url = await dispatch_url(url, is_private, policy)
if not message:
return
else:
urls = [u for u in URL_PATTERN.findall(msg) if not _skip_banned(u, policy)]
for url in urls:
logger.info(f"消息解析:作品链接:{url}")
message, public_url = await dispatch_url(url, is_private, policy)
if message:
break
if not message:
return
if isinstance(message, PendingMedia):
ok, pub = await send_pending_media(message, event)
if pub and policy.sends_link:
await UniMessage.text(f"{pub}").send()
if not ok:
await UniMessage.text(f"媒体发送失败:{url}").send()
else:
if public_url and policy.sends_link:
await UniMessage.text(f"{public_url}").send()
await message.send()
async def dispatch_url(
url: str,
is_private: bool,
policy: Policy,
) -> tuple[UniMessage | None, str | None]:
"""按平台分派解析(文本链接与小程序卡片共用)。"""
url = url.rstrip(",。!?、;:)】》\"')")
if "b23.tv" in url or "bili2233.cn" in url:
resolved = await resolve_short_link(url)
if resolved:
logger.info(f"b23 短链重定向: {url} -> {resolved}")
url = resolved
# 平台标签:短链重定向之后再判定(群文件限定平台用)
platform = match_platform(url)
if "douyin.com" in url or "v.douyin.com" in url or "iesdouyin.com" in url:
await UniMessage.text("检测到链接,正在处理,请稍候...").send()
try:
title, parsed_path, image_post = await parse_douyin(url)
if not title and not parsed_path:
# 链接识别成功但未能获取到作品(网络/风控/链接失效等),
# 给出明确失败提示,避免只发"请稍候"后无下文。
await UniMessage.text(f"无法解析到媒体:{url}").send()
logger.warning(f"媒体解析:未能获取到作品:{url}")
return None, None
return await process_douyin_res(
title, parsed_path, is_private, image_post,
policy=policy, platform=platform,
)
except Exception:
await UniMessage.text(f"无法解析到媒体:{url}").send()
logger.exception(f"媒体解析:无法解析到媒体:{url}")
return None, None
if any(
kw in url
for kw in (
"bilibili.com/opus",
"bilibili.com/dynamic",
"t.bilibili.com",
"bilibili.com/read",
)
):
await UniMessage.text("检测到链接,正在处理,请稍候...").send()
try:
title, parsed_path = await fetch_bilibili_content(url)
if not title and parsed_path is None:
await UniMessage.text(f"无法解析到内容:{url}").send()
return None, None
if parsed_path == []:
await UniMessage.text(title).send()
return None, None
return await process_douyin_res(
title, parsed_path, is_private,
isinstance(parsed_path, list), policy=policy, platform=platform,
)
except Exception:
await UniMessage.text(f"无法解析到内容:{url}").send()
logger.exception(f"B站内容解析:无法解析:{url}")
return None, None
if "xiaohongshu.com" in url or "xhslink." in url:
await UniMessage.text("检测到链接,正在处理,请稍候...").send()
try:
title, parsed_path = await fetch_rednote_content(url)
if not title and parsed_path is None:
await UniMessage.text(f"无法解析到内容:{url}").send()
return None, None
if parsed_path == []:
await UniMessage.text(title).send()
return None, None
return await process_douyin_res(
title, parsed_path, is_private,
isinstance(parsed_path, list), policy=policy, platform=platform,
)
except Exception:
await UniMessage.text(f"无法解析到内容:{url}").send()
logger.exception(f"小红书解析:无法解析:{url}")
return None, None
if any(domain in url for domain in VALID_HOSTS):
await UniMessage.text("检测到链接,正在处理,请稍候...").send()
try:
return await handle_universal(url, is_private, policy, platform)
except Exception as e:
logger.exception(e)
await UniMessage.text("下载过程中出现错误。").send()
return None, None
return None, None
async def _extract_xcx_url(msg: str) -> str | None:
"""从 CQ 卡片消息中提取跳转 URL(保留 query 参数)。"""
match = re.search(r'"qqdocurl":"(.*?)"', msg)
if not match:
match = re.search(r'"jumpUrl":"(.*?)"', msg)
if not match:
logger.warning("未找到 qqdocurl/jumpUrl 字段")
return None
raw_url = match.group(1)
unescaped = html.unescape(raw_url)
cleaned_url = unescaped.replace(r"\/", "/")
logger.info(f"卡片提取链接: {cleaned_url}")
return cleaned_url