Files
HeXi/hexi/plugins/nonebot_plugin_video_analysis/handlers/entry.py
T
sansenhoshi 131b92b319 结构调整
视频解析多图/多媒体结构 消息体适配
2026-09-08 14:25:32 +08:00

268 lines
9.6 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""消息入口与分派:自动解析 / 主动解析(文本链接 + QQ小程序/分享卡片)。"""
from __future__ import annotations
import html
import re
from nonebot import on_message, logger
from nonebot.adapters import Event
from nonebot.rule import to_me
from nonebot_plugin_alconna import UniMessage
from nonebot_plugin_alconna.uniseg import get_target
from ..services.fetchers.bilibili_content import fetch_bilibili_content, resolve_short_link
from ..services.fetchers.rednote_content import fetch_rednote_content
from .douyin import parse_douyin, process_douyin_res
from .sender import PendingMedia, send_pending_media
from .universal import handle_universal
from ..list_proc import AUTO_LINK_KEYWORDS, get_group_auto_link, verify_user
URL_PATTERN = re.compile(r"(https?://\S+)")
XCX_PATTERN = r"QQ小程序(?:]|]|\])"
VALID_HOSTS = [
"b23.tv",
"bilibili.com",
"youtube.com",
"youtu.be",
"douyin.com",
"v.douyin.com",
"iesdouyin.com",
"m.douyin.com",
"jingxuan.douyin.com",
"x.com",
"twitter.com",
"xiaohongshu.com",
"xhslink.com",
"xhslink.cn",
]
auto_video_handler = on_message(priority=10, block=False)
active_video_handler = on_message(priority=10, block=False, rule=to_me())
async def _check_access(
event: Event, *, auto_only: bool = False, msg: str | None = None
) -> tuple[bool, str | None]:
"""统一权限检查。
auto_only=True → 自动解析:需白名单 + 开启自动解析,或 auto_link 关键词命中。
auto_only=False → 主动触发:需非黑名单,群聊还需白名单。
Returns:
(allowed, plan) — plan 用于 S3 路由,不允许时为 None
"""
white, black, auto, plan = await verify_user(event)
target = get_target(event)
if target.private:
if auto_only:
logger.info("权限分析:自动解析不处理私聊")
return False, None
if black:
logger.info(f"权限分析:黑名单用户私聊,不回复: {event.get_user_id()}")
return False, None
logger.info("权限分析:私聊,直接解析")
return True, None
group_id = str(event.group_id)
if not white:
logger.info(f"权限分析:群 {group_id} 不在白名单,不做处理")
return False, None
if auto_only and not auto:
if msg is not None and await _match_auto_link(event, msg):
logger.info(f"权限分析:群 {group_id} 未开启自动解析,但自动链接关键词命中")
else:
logger.info(f"权限分析:群 {group_id} 未开启自动解析")
return False, None
if not auto_only and black:
logger.info(f"权限分析:黑名单用户,不回复: {event.get_user_id()}")
return False, None
logger.info(
f"权限分析:群 {group_id} 权限通过 — "
f"自动解析: {auto}, 方案: {plan or '默认(PLANC)'}"
)
return True, plan
async def _match_auto_link(event: Event, msg: str) -> bool:
"""消息中的 URL 是否命中群配置的 auto_link 关键词。"""
keywords = await get_group_auto_link(event)
if not keywords:
return False
urls = URL_PATTERN.findall(msg)
if not urls:
return False
for kw in keywords:
domains = AUTO_LINK_KEYWORDS.get(kw, (kw,))
if any(any(domain in url for domain in domains) for url in urls):
logger.info(f"自动链接:关键词 {kw} 命中消息 {urls}")
return True
return False
@auto_video_handler.handle()
async def handle_auto_video(event: Event):
msg = str(event.get_message()).strip()
allowed, plan = await _check_access(event, auto_only=True, msg=msg)
if allowed:
await match_message(event, plan=plan)
@active_video_handler.handle()
async def handle_active_video(event: Event):
allowed, plan = await _check_access(event, auto_only=False)
if allowed:
await match_message(event, plan=plan)
async def match_message(event: Event, plan: str | None = None):
"""消息匹配与分派:文本链接 / QQ小程序卡片统一走 dispatch_url。"""
msg = str(event.get_message()).strip()
logger.info(f"消息解析:获取到的消息:{msg}")
is_private = get_target(event).private
message = None
public_url = None
if re.search(XCX_PATTERN, msg) or "CQ:json" in msg or "CQ:share" in msg:
logger.info("消息解析:检测到 CQ 卡片")
url = await _extract_xcx_url(msg)
logger.info(f"消息解析:卡片链接:{url}")
if not url or not any(domain in url for domain in VALID_HOSTS):
return
message, public_url = await dispatch_url(url, is_private, plan=plan)
if not message:
return
else:
urls = URL_PATTERN.findall(msg)
for url in urls:
logger.info(f"消息解析:作品链接:{url}")
message, public_url = await dispatch_url(url, is_private, plan=plan)
if message:
break
if not message:
return
if isinstance(message, PendingMedia):
ok, pub = await send_pending_media(message, event)
if pub:
await UniMessage.text(f"{pub}").send()
if not ok:
await UniMessage.text(f"媒体发送失败:{url}").send()
else:
if public_url:
await UniMessage.text(f"{public_url}").send()
await message.send()
async def dispatch_url(
url: str,
is_private: bool,
plan: str | None = None,
) -> tuple[UniMessage | None, str | None]:
"""按平台分派解析(文本链接与小程序卡片共用)。"""
url = url.rstrip(",。!?、;:)】》\"')")
if "b23.tv" in url or "bili2233.cn" in url:
resolved = await resolve_short_link(url)
if resolved:
logger.info(f"b23 短链重定向: {url} -> {resolved}")
url = resolved
if "douyin.com" in url or "v.douyin.com" in url or "iesdouyin.com" in url:
await UniMessage.text("检测到链接,正在处理,请稍候...").send()
try:
title, parsed_path, image_post = await parse_douyin(url)
if not title and not parsed_path:
# 链接识别成功但未能获取到作品(网络/风控/链接失效等),
# 给出明确失败提示,避免只发"请稍候"后无下文。
await UniMessage.text(f"无法解析到媒体:{url}").send()
logger.warning(f"媒体解析:未能获取到作品:{url}")
return None, None
return await process_douyin_res(
title, parsed_path, is_private, image_post, plan=plan
)
except Exception:
await UniMessage.text(f"无法解析到媒体:{url}").send()
logger.exception(f"媒体解析:无法解析到媒体:{url}")
return None, None
if any(
kw in url
for kw in (
"bilibili.com/opus",
"bilibili.com/dynamic",
"t.bilibili.com",
"bilibili.com/read",
)
):
await UniMessage.text("检测到链接,正在处理,请稍候...").send()
try:
title, parsed_path = await fetch_bilibili_content(url)
if not title and parsed_path is None:
await UniMessage.text(f"无法解析到内容:{url}").send()
return None, None
if parsed_path == []:
await UniMessage.text(title).send()
return None, None
return await process_douyin_res(
title, parsed_path, is_private,
isinstance(parsed_path, list), plan=plan,
)
except Exception:
await UniMessage.text(f"无法解析到内容:{url}").send()
logger.exception(f"B站内容解析:无法解析:{url}")
return None, None
if "xiaohongshu.com" in url or "xhslink." in url:
await UniMessage.text("检测到链接,正在处理,请稍候...").send()
try:
title, parsed_path = await fetch_rednote_content(url)
if not title and parsed_path is None:
await UniMessage.text(f"无法解析到内容:{url}").send()
return None, None
if parsed_path == []:
await UniMessage.text(title).send()
return None, None
return await process_douyin_res(
title, parsed_path, is_private,
isinstance(parsed_path, list), plan=plan,
)
except Exception:
await UniMessage.text(f"无法解析到内容:{url}").send()
logger.exception(f"小红书解析:无法解析:{url}")
return None, None
if any(domain in url for domain in VALID_HOSTS):
await UniMessage.text("检测到链接,正在处理,请稍候...").send()
try:
return await handle_universal(url, is_private, plan=plan)
except Exception as e:
logger.exception(e)
await UniMessage.text("下载过程中出现错误。").send()
return None, None
return None, None
async def _extract_xcx_url(msg: str) -> str | None:
"""从 CQ 卡片消息中提取跳转 URL(保留 query 参数)。"""
match = re.search(r'"qqdocurl":"(.*?)"', msg)
if not match:
match = re.search(r'"jumpUrl":"(.*?)"', msg)
if not match:
logger.warning("未找到 qqdocurl/jumpUrl 字段")
return None
raw_url = match.group(1)
unescaped = html.unescape(raw_url)
cleaned_url = unescaped.replace(r"\/", "/")
logger.info(f"卡片提取链接: {cleaned_url}")
return cleaned_url