- policy.py:per-group 正交策略(自动解析 / 自动策略 / 禁用策略 / 存储 A·B·C /
公网 / 链接 / 群文件 + 平台限定),list.json v1/v2 → v3 自动迁移,
写入统一走 PolicyStore(加锁 + .tmp 原子替换 + 字段归一)
- 群文件并行通道 group_file.py:打包 zip(可选 pyzipper AES-256)后优先走 S3 预签名、
本地直传兜底;设了密码但 pyzipper 不可用就放弃上传,不退化成明文
- list_proc.py 收敛到「视频策略」统一入口,权限判定改走 policy
- Web 管理页 /hub/video_analysis(群策略 + 链接解析面板)与 services/web_jobs.py
(只复用纯函数层,Web 上下文不发消息;内存任务表 + 并发闸门 + 超时)
- 媒体命名统一到 utils.py({作者}_{作者id}/{作品名}[_短码]),cleanup 回收空目录
- 测试:policy / 命名 / 群文件 / web_jobs 四组
顺带 pyproject 的 pytest 加 testpaths=tests(避免收进 debug/ 下的调试脚本)。
Co-Authored-By: Claude Code <noreply@anthropic.com>
286 lines
10 KiB
Python
286 lines
10 KiB
Python
"""消息入口与分派:自动解析 / 主动解析(文本链接 + QQ小程序/分享卡片)。"""
|
||
|
||
from __future__ import annotations
|
||
|
||
import html
|
||
import re
|
||
|
||
from nonebot import on_message, logger
|
||
from nonebot.adapters import Event
|
||
from nonebot.rule import to_me
|
||
from nonebot_plugin_alconna import UniMessage
|
||
from nonebot_plugin_alconna.uniseg import get_target
|
||
|
||
from ..services.fetchers.bilibili_content import fetch_bilibili_content, resolve_short_link
|
||
from ..services.fetchers.rednote_content import fetch_rednote_content
|
||
from .douyin import parse_douyin, process_douyin_res
|
||
from .sender import PendingMedia, send_pending_media
|
||
from .universal import handle_universal
|
||
from ..list_proc import get_policy, is_group_whitelisted, is_user_blacklisted
|
||
from ..policy import Policy, match_platform
|
||
|
||
URL_PATTERN = re.compile(r"(https?://\S+)")
|
||
XCX_PATTERN = r"QQ小程序(?:&#93;|]|\])"
|
||
|
||
VALID_HOSTS = [
|
||
"b23.tv",
|
||
"bilibili.com",
|
||
"youtube.com",
|
||
"youtu.be",
|
||
"douyin.com",
|
||
"v.douyin.com",
|
||
"iesdouyin.com",
|
||
"m.douyin.com",
|
||
"jingxuan.douyin.com",
|
||
"x.com",
|
||
"twitter.com",
|
||
"xiaohongshu.com",
|
||
"xhslink.com",
|
||
"xhslink.cn",
|
||
]
|
||
|
||
auto_video_handler = on_message(priority=10, block=False)
|
||
active_video_handler = on_message(priority=10, block=False, rule=to_me())
|
||
|
||
|
||
async def _check_access(
|
||
event: Event, *, auto_only: bool = False, msg: str | None = None
|
||
) -> tuple[bool, Policy | None]:
|
||
"""统一权限检查。
|
||
|
||
auto_only=True → 自动解析:需白名单群 + (开启自动解析或消息命中自动策略)。
|
||
auto_only=False → 主动触发:需白名单群 + 非黑名单用户。
|
||
|
||
私聊不做自动解析,读 default 节策略后直接解析。
|
||
禁用策略(ban_link)是消息级的(按链接判定),见 match_message。
|
||
|
||
Returns:
|
||
(allowed, policy) — policy 决定存储/投递,不允许时为 None
|
||
"""
|
||
target = get_target(event)
|
||
|
||
if target.private:
|
||
if auto_only:
|
||
logger.info("权限分析:自动解析不处理私聊")
|
||
return False, None
|
||
if is_user_blacklisted(event):
|
||
logger.info(f"权限分析:黑名单用户私聊,不回复: {event.get_user_id()}")
|
||
return False, None
|
||
logger.info("权限分析:私聊,直接解析")
|
||
return True, get_policy(event)
|
||
|
||
group_id = str(event.group_id)
|
||
|
||
if not is_group_whitelisted(event):
|
||
logger.info(f"权限分析:群 {group_id} 不在白名单,不做处理")
|
||
return False, None
|
||
|
||
if is_user_blacklisted(event):
|
||
logger.info(f"权限分析:黑名单用户,不回复: {event.get_user_id()}")
|
||
return False, None
|
||
|
||
policy = get_policy(event)
|
||
|
||
if auto_only and not policy.auto and not _match_auto_link(msg or "", policy):
|
||
logger.info(f"权限分析:群 {group_id} 未开启自动解析且未命中自动策略")
|
||
return False, None
|
||
|
||
logger.info(
|
||
f"权限分析:群 {group_id} 权限通过 — 自动解析: {policy.auto}, "
|
||
f"存储: {policy.plan}, 公网: {policy.upload_public}, "
|
||
f"群文件: {policy.upload_group_file}"
|
||
)
|
||
return True, policy
|
||
|
||
|
||
def _match_auto_link(msg: str, policy: Policy) -> bool:
|
||
"""消息中的 URL 是否命中群策略的自动策略(auto_link)。"""
|
||
if not policy.auto_link:
|
||
return False
|
||
for url in URL_PATTERN.findall(msg):
|
||
platform = policy.auto_matched(url)
|
||
if platform:
|
||
logger.info(f"自动策略:{platform} 命中消息 {url}")
|
||
return True
|
||
return False
|
||
|
||
|
||
def _skip_banned(url: str, policy: Policy) -> bool:
|
||
"""链接是否命中禁用策略(命中即静默丢弃,只记日志)。"""
|
||
platform = policy.banned(url)
|
||
if platform:
|
||
logger.info(f"禁用策略:{platform} 已禁用,忽略链接 {url}")
|
||
return True
|
||
return False
|
||
|
||
|
||
@auto_video_handler.handle()
|
||
async def handle_auto_video(event: Event):
|
||
msg = str(event.get_message()).strip()
|
||
allowed, policy = await _check_access(event, auto_only=True, msg=msg)
|
||
if allowed and policy is not None:
|
||
await match_message(event, policy)
|
||
|
||
|
||
@active_video_handler.handle()
|
||
async def handle_active_video(event: Event):
|
||
allowed, policy = await _check_access(event, auto_only=False)
|
||
if allowed and policy is not None:
|
||
await match_message(event, policy)
|
||
|
||
|
||
async def match_message(event: Event, policy: Policy):
|
||
"""消息匹配与分派:文本链接 / QQ小程序卡片统一走 dispatch_url。
|
||
|
||
命中禁用策略的链接在这里丢弃;消息里还有其它可用链接则继续解析。
|
||
"""
|
||
msg = str(event.get_message()).strip()
|
||
logger.info(f"消息解析:获取到的消息:{msg}")
|
||
is_private = get_target(event).private
|
||
|
||
message = None
|
||
public_url = None
|
||
url = ""
|
||
|
||
if re.search(XCX_PATTERN, msg) or "CQ:json" in msg or "CQ:share" in msg:
|
||
logger.info("消息解析:检测到 CQ 卡片")
|
||
url = await _extract_xcx_url(msg) or ""
|
||
logger.info(f"消息解析:卡片链接:{url}")
|
||
if not url or not any(domain in url for domain in VALID_HOSTS):
|
||
return
|
||
if _skip_banned(url, policy):
|
||
return
|
||
message, public_url = await dispatch_url(url, is_private, policy)
|
||
if not message:
|
||
return
|
||
else:
|
||
urls = [u for u in URL_PATTERN.findall(msg) if not _skip_banned(u, policy)]
|
||
for url in urls:
|
||
logger.info(f"消息解析:作品链接:{url}")
|
||
message, public_url = await dispatch_url(url, is_private, policy)
|
||
if message:
|
||
break
|
||
if not message:
|
||
return
|
||
|
||
if isinstance(message, PendingMedia):
|
||
ok, pub = await send_pending_media(message, event)
|
||
if pub and policy.sends_link:
|
||
await UniMessage.text(f"{pub}").send()
|
||
if not ok:
|
||
await UniMessage.text(f"媒体发送失败:{url}").send()
|
||
else:
|
||
if public_url and policy.sends_link:
|
||
await UniMessage.text(f"{public_url}").send()
|
||
await message.send()
|
||
|
||
|
||
async def dispatch_url(
|
||
url: str,
|
||
is_private: bool,
|
||
policy: Policy,
|
||
) -> tuple[UniMessage | None, str | None]:
|
||
"""按平台分派解析(文本链接与小程序卡片共用)。"""
|
||
url = url.rstrip(",。!?、;:)】》\"')")
|
||
|
||
if "b23.tv" in url or "bili2233.cn" in url:
|
||
resolved = await resolve_short_link(url)
|
||
if resolved:
|
||
logger.info(f"b23 短链重定向: {url} -> {resolved}")
|
||
url = resolved
|
||
|
||
# 平台标签:短链重定向之后再判定(群文件限定平台用)
|
||
platform = match_platform(url)
|
||
|
||
if "douyin.com" in url or "v.douyin.com" in url or "iesdouyin.com" in url:
|
||
await UniMessage.text("检测到链接,正在处理,请稍候...").send()
|
||
try:
|
||
title, parsed_path, image_post = await parse_douyin(url)
|
||
if not title and not parsed_path:
|
||
# 链接识别成功但未能获取到作品(网络/风控/链接失效等),
|
||
# 给出明确失败提示,避免只发"请稍候"后无下文。
|
||
await UniMessage.text(f"无法解析到媒体:{url}").send()
|
||
logger.warning(f"媒体解析:未能获取到作品:{url}")
|
||
return None, None
|
||
return await process_douyin_res(
|
||
title, parsed_path, is_private, image_post,
|
||
policy=policy, platform=platform,
|
||
)
|
||
except Exception:
|
||
await UniMessage.text(f"无法解析到媒体:{url}").send()
|
||
logger.exception(f"媒体解析:无法解析到媒体:{url}")
|
||
return None, None
|
||
|
||
if any(
|
||
kw in url
|
||
for kw in (
|
||
"bilibili.com/opus",
|
||
"bilibili.com/dynamic",
|
||
"t.bilibili.com",
|
||
"bilibili.com/read",
|
||
)
|
||
):
|
||
await UniMessage.text("检测到链接,正在处理,请稍候...").send()
|
||
try:
|
||
title, parsed_path = await fetch_bilibili_content(url)
|
||
if not title and parsed_path is None:
|
||
await UniMessage.text(f"无法解析到内容:{url}").send()
|
||
return None, None
|
||
if parsed_path == []:
|
||
await UniMessage.text(title).send()
|
||
return None, None
|
||
return await process_douyin_res(
|
||
title, parsed_path, is_private,
|
||
isinstance(parsed_path, list), policy=policy, platform=platform,
|
||
)
|
||
except Exception:
|
||
await UniMessage.text(f"无法解析到内容:{url}").send()
|
||
logger.exception(f"B站内容解析:无法解析:{url}")
|
||
return None, None
|
||
|
||
if "xiaohongshu.com" in url or "xhslink." in url:
|
||
await UniMessage.text("检测到链接,正在处理,请稍候...").send()
|
||
try:
|
||
title, parsed_path = await fetch_rednote_content(url)
|
||
if not title and parsed_path is None:
|
||
await UniMessage.text(f"无法解析到内容:{url}").send()
|
||
return None, None
|
||
if parsed_path == []:
|
||
await UniMessage.text(title).send()
|
||
return None, None
|
||
return await process_douyin_res(
|
||
title, parsed_path, is_private,
|
||
isinstance(parsed_path, list), policy=policy, platform=platform,
|
||
)
|
||
except Exception:
|
||
await UniMessage.text(f"无法解析到内容:{url}").send()
|
||
logger.exception(f"小红书解析:无法解析:{url}")
|
||
return None, None
|
||
|
||
if any(domain in url for domain in VALID_HOSTS):
|
||
await UniMessage.text("检测到链接,正在处理,请稍候...").send()
|
||
try:
|
||
return await handle_universal(url, is_private, policy, platform)
|
||
except Exception as e:
|
||
logger.exception(e)
|
||
await UniMessage.text("下载过程中出现错误。").send()
|
||
return None, None
|
||
|
||
return None, None
|
||
|
||
|
||
async def _extract_xcx_url(msg: str) -> str | None:
|
||
"""从 CQ 卡片消息中提取跳转 URL(保留 query 参数)。"""
|
||
match = re.search(r'"qqdocurl":"(.*?)"', msg)
|
||
if not match:
|
||
match = re.search(r'"jumpUrl":"(.*?)"', msg)
|
||
if not match:
|
||
logger.warning("未找到 qqdocurl/jumpUrl 字段")
|
||
return None
|
||
|
||
raw_url = match.group(1)
|
||
unescaped = html.unescape(raw_url)
|
||
cleaned_url = unescaped.replace(r"\/", "/")
|
||
logger.info(f"卡片提取链接: {cleaned_url}")
|
||
return cleaned_url
|