268 lines
9.6 KiB
Python
268 lines
9.6 KiB
Python
"""消息入口与分派:自动解析 / 主动解析(文本链接 + QQ小程序/分享卡片)。"""
|
||
|
||
from __future__ import annotations
|
||
|
||
import html
|
||
import re
|
||
|
||
from nonebot import on_message, logger
|
||
from nonebot.adapters import Event
|
||
from nonebot.rule import to_me
|
||
from nonebot_plugin_alconna import UniMessage
|
||
from nonebot_plugin_alconna.uniseg import get_target
|
||
|
||
from ..services.fetchers.bilibili_content import fetch_bilibili_content, resolve_short_link
|
||
from ..services.fetchers.rednote_content import fetch_rednote_content
|
||
from .douyin import parse_douyin, process_douyin_res
|
||
from .sender import PendingMedia, send_pending_media
|
||
from .universal import handle_universal
|
||
from ..list_proc import AUTO_LINK_KEYWORDS, get_group_auto_link, verify_user
|
||
|
||
URL_PATTERN = re.compile(r"(https?://\S+)")
|
||
XCX_PATTERN = r"QQ小程序(?:]|]|\])"
|
||
|
||
VALID_HOSTS = [
|
||
"b23.tv",
|
||
"bilibili.com",
|
||
"youtube.com",
|
||
"youtu.be",
|
||
"douyin.com",
|
||
"v.douyin.com",
|
||
"iesdouyin.com",
|
||
"m.douyin.com",
|
||
"jingxuan.douyin.com",
|
||
"x.com",
|
||
"twitter.com",
|
||
"xiaohongshu.com",
|
||
"xhslink.com",
|
||
"xhslink.cn",
|
||
]
|
||
|
||
auto_video_handler = on_message(priority=10, block=False)
|
||
active_video_handler = on_message(priority=10, block=False, rule=to_me())
|
||
|
||
|
||
async def _check_access(
|
||
event: Event, *, auto_only: bool = False, msg: str | None = None
|
||
) -> tuple[bool, str | None]:
|
||
"""统一权限检查。
|
||
|
||
auto_only=True → 自动解析:需白名单 + 开启自动解析,或 auto_link 关键词命中。
|
||
auto_only=False → 主动触发:需非黑名单,群聊还需白名单。
|
||
|
||
Returns:
|
||
(allowed, plan) — plan 用于 S3 路由,不允许时为 None
|
||
"""
|
||
white, black, auto, plan = await verify_user(event)
|
||
target = get_target(event)
|
||
|
||
if target.private:
|
||
if auto_only:
|
||
logger.info("权限分析:自动解析不处理私聊")
|
||
return False, None
|
||
if black:
|
||
logger.info(f"权限分析:黑名单用户私聊,不回复: {event.get_user_id()}")
|
||
return False, None
|
||
logger.info("权限分析:私聊,直接解析")
|
||
return True, None
|
||
|
||
group_id = str(event.group_id)
|
||
|
||
if not white:
|
||
logger.info(f"权限分析:群 {group_id} 不在白名单,不做处理")
|
||
return False, None
|
||
|
||
if auto_only and not auto:
|
||
if msg is not None and await _match_auto_link(event, msg):
|
||
logger.info(f"权限分析:群 {group_id} 未开启自动解析,但自动链接关键词命中")
|
||
else:
|
||
logger.info(f"权限分析:群 {group_id} 未开启自动解析")
|
||
return False, None
|
||
|
||
if not auto_only and black:
|
||
logger.info(f"权限分析:黑名单用户,不回复: {event.get_user_id()}")
|
||
return False, None
|
||
|
||
logger.info(
|
||
f"权限分析:群 {group_id} 权限通过 — "
|
||
f"自动解析: {auto}, 方案: {plan or '默认(PLANC)'}"
|
||
)
|
||
return True, plan
|
||
|
||
|
||
async def _match_auto_link(event: Event, msg: str) -> bool:
|
||
"""消息中的 URL 是否命中群配置的 auto_link 关键词。"""
|
||
keywords = await get_group_auto_link(event)
|
||
if not keywords:
|
||
return False
|
||
urls = URL_PATTERN.findall(msg)
|
||
if not urls:
|
||
return False
|
||
for kw in keywords:
|
||
domains = AUTO_LINK_KEYWORDS.get(kw, (kw,))
|
||
if any(any(domain in url for domain in domains) for url in urls):
|
||
logger.info(f"自动链接:关键词 {kw} 命中消息 {urls}")
|
||
return True
|
||
return False
|
||
|
||
|
||
@auto_video_handler.handle()
|
||
async def handle_auto_video(event: Event):
|
||
msg = str(event.get_message()).strip()
|
||
allowed, plan = await _check_access(event, auto_only=True, msg=msg)
|
||
if allowed:
|
||
await match_message(event, plan=plan)
|
||
|
||
|
||
@active_video_handler.handle()
|
||
async def handle_active_video(event: Event):
|
||
allowed, plan = await _check_access(event, auto_only=False)
|
||
if allowed:
|
||
await match_message(event, plan=plan)
|
||
|
||
|
||
async def match_message(event: Event, plan: str | None = None):
|
||
"""消息匹配与分派:文本链接 / QQ小程序卡片统一走 dispatch_url。"""
|
||
msg = str(event.get_message()).strip()
|
||
logger.info(f"消息解析:获取到的消息:{msg}")
|
||
is_private = get_target(event).private
|
||
|
||
message = None
|
||
public_url = None
|
||
|
||
if re.search(XCX_PATTERN, msg) or "CQ:json" in msg or "CQ:share" in msg:
|
||
logger.info("消息解析:检测到 CQ 卡片")
|
||
url = await _extract_xcx_url(msg)
|
||
logger.info(f"消息解析:卡片链接:{url}")
|
||
if not url or not any(domain in url for domain in VALID_HOSTS):
|
||
return
|
||
message, public_url = await dispatch_url(url, is_private, plan=plan)
|
||
if not message:
|
||
return
|
||
else:
|
||
urls = URL_PATTERN.findall(msg)
|
||
for url in urls:
|
||
logger.info(f"消息解析:作品链接:{url}")
|
||
message, public_url = await dispatch_url(url, is_private, plan=plan)
|
||
if message:
|
||
break
|
||
if not message:
|
||
return
|
||
|
||
if isinstance(message, PendingMedia):
|
||
ok, pub = await send_pending_media(message, event)
|
||
if pub:
|
||
await UniMessage.text(f"{pub}").send()
|
||
if not ok:
|
||
await UniMessage.text(f"媒体发送失败:{url}").send()
|
||
else:
|
||
if public_url:
|
||
await UniMessage.text(f"{public_url}").send()
|
||
await message.send()
|
||
|
||
|
||
async def dispatch_url(
|
||
url: str,
|
||
is_private: bool,
|
||
plan: str | None = None,
|
||
) -> tuple[UniMessage | None, str | None]:
|
||
"""按平台分派解析(文本链接与小程序卡片共用)。"""
|
||
url = url.rstrip(",。!?、;:)】》\"')")
|
||
|
||
if "b23.tv" in url or "bili2233.cn" in url:
|
||
resolved = await resolve_short_link(url)
|
||
if resolved:
|
||
logger.info(f"b23 短链重定向: {url} -> {resolved}")
|
||
url = resolved
|
||
|
||
if "douyin.com" in url or "v.douyin.com" in url or "iesdouyin.com" in url:
|
||
await UniMessage.text("检测到链接,正在处理,请稍候...").send()
|
||
try:
|
||
title, parsed_path, image_post = await parse_douyin(url)
|
||
if not title and not parsed_path:
|
||
# 链接识别成功但未能获取到作品(网络/风控/链接失效等),
|
||
# 给出明确失败提示,避免只发"请稍候"后无下文。
|
||
await UniMessage.text(f"无法解析到媒体:{url}").send()
|
||
logger.warning(f"媒体解析:未能获取到作品:{url}")
|
||
return None, None
|
||
return await process_douyin_res(
|
||
title, parsed_path, is_private, image_post, plan=plan
|
||
)
|
||
except Exception:
|
||
await UniMessage.text(f"无法解析到媒体:{url}").send()
|
||
logger.exception(f"媒体解析:无法解析到媒体:{url}")
|
||
return None, None
|
||
|
||
if any(
|
||
kw in url
|
||
for kw in (
|
||
"bilibili.com/opus",
|
||
"bilibili.com/dynamic",
|
||
"t.bilibili.com",
|
||
"bilibili.com/read",
|
||
)
|
||
):
|
||
await UniMessage.text("检测到链接,正在处理,请稍候...").send()
|
||
try:
|
||
title, parsed_path = await fetch_bilibili_content(url)
|
||
if not title and parsed_path is None:
|
||
await UniMessage.text(f"无法解析到内容:{url}").send()
|
||
return None, None
|
||
if parsed_path == []:
|
||
await UniMessage.text(title).send()
|
||
return None, None
|
||
return await process_douyin_res(
|
||
title, parsed_path, is_private,
|
||
isinstance(parsed_path, list), plan=plan,
|
||
)
|
||
except Exception:
|
||
await UniMessage.text(f"无法解析到内容:{url}").send()
|
||
logger.exception(f"B站内容解析:无法解析:{url}")
|
||
return None, None
|
||
|
||
if "xiaohongshu.com" in url or "xhslink." in url:
|
||
await UniMessage.text("检测到链接,正在处理,请稍候...").send()
|
||
try:
|
||
title, parsed_path = await fetch_rednote_content(url)
|
||
if not title and parsed_path is None:
|
||
await UniMessage.text(f"无法解析到内容:{url}").send()
|
||
return None, None
|
||
if parsed_path == []:
|
||
await UniMessage.text(title).send()
|
||
return None, None
|
||
return await process_douyin_res(
|
||
title, parsed_path, is_private,
|
||
isinstance(parsed_path, list), plan=plan,
|
||
)
|
||
except Exception:
|
||
await UniMessage.text(f"无法解析到内容:{url}").send()
|
||
logger.exception(f"小红书解析:无法解析:{url}")
|
||
return None, None
|
||
|
||
if any(domain in url for domain in VALID_HOSTS):
|
||
await UniMessage.text("检测到链接,正在处理,请稍候...").send()
|
||
try:
|
||
return await handle_universal(url, is_private, plan=plan)
|
||
except Exception as e:
|
||
logger.exception(e)
|
||
await UniMessage.text("下载过程中出现错误。").send()
|
||
return None, None
|
||
|
||
return None, None
|
||
|
||
|
||
async def _extract_xcx_url(msg: str) -> str | None:
|
||
"""从 CQ 卡片消息中提取跳转 URL(保留 query 参数)。"""
|
||
match = re.search(r'"qqdocurl":"(.*?)"', msg)
|
||
if not match:
|
||
match = re.search(r'"jumpUrl":"(.*?)"', msg)
|
||
if not match:
|
||
logger.warning("未找到 qqdocurl/jumpUrl 字段")
|
||
return None
|
||
|
||
raw_url = match.group(1)
|
||
unescaped = html.unescape(raw_url)
|
||
cleaned_url = unescaped.replace(r"\/", "/")
|
||
logger.info(f"卡片提取链接: {cleaned_url}")
|
||
return cleaned_url
|