From 4badcfcf3211a3936c1359feca7b75aa7a24a04f Mon Sep 17 00:00:00 2001 From: sans Date: Tue, 22 Sep 2026 14:23:32 +0800 Subject: [PATCH] =?UTF-8?q?feat(video-analysis):=20=E7=BE=A4=E7=AD=96?= =?UTF-8?q?=E7=95=A5=20v3=20/=20=E7=BE=A4=E6=96=87=E4=BB=B6=E6=8A=95?= =?UTF-8?q?=E9=80=92=E9=80=9A=E9=81=93=20/=20Web=20=E7=AE=A1=E7=90=86?= =?UTF-8?q?=E9=A1=B5?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - policy.py:per-group 正交策略(自动解析 / 自动策略 / 禁用策略 / 存储 A·B·C / 公网 / 链接 / 群文件 + 平台限定),list.json v1/v2 → v3 自动迁移, 写入统一走 PolicyStore(加锁 + .tmp 原子替换 + 字段归一) - 群文件并行通道 group_file.py:打包 zip(可选 pyzipper AES-256)后优先走 S3 预签名、 本地直传兜底;设了密码但 pyzipper 不可用就放弃上传,不退化成明文 - list_proc.py 收敛到「视频策略」统一入口,权限判定改走 policy - Web 管理页 /hub/video_analysis(群策略 + 链接解析面板)与 services/web_jobs.py (只复用纯函数层,Web 上下文不发消息;内存任务表 + 并发闸门 + 超时) - 媒体命名统一到 utils.py({作者}_{作者id}/{作品名}[_短码]),cleanup 回收空目录 - 测试:policy / 命名 / 群文件 / web_jobs 四组 顺带 pyproject 的 pytest 加 testpaths=tests(避免收进 debug/ 下的调试脚本)。 Co-Authored-By: Claude Code --- .../nonebot_plugin_video_analysis/__init__.py | 17 + .../nonebot_plugin_video_analysis/cleanup.py | 58 +- .../nonebot_plugin_video_analysis/config.py | 71 +- .../handlers/douyin.py | 10 +- .../handlers/entry.py | 118 ++- .../handlers/sender.py | 88 +- .../handlers/universal.py | 14 +- .../list_proc.py | 719 +++++++------- .../nonebot_plugin_video_analysis/policy.py | 486 ++++++++++ .../services/fetchers/bilibili_content.py | 78 +- .../services/fetchers/douyin_api.py | 51 +- .../services/fetchers/douyin_parser.py | 81 +- .../services/fetchers/douyin_ssr.py | 6 +- .../services/fetchers/rednote_content.py | 64 +- .../services/fetchers/video_downloader.py | 140 +-- .../services/storage/group_file.py | 271 ++++++ .../services/storage/s3.py | 54 +- .../services/web_jobs.py | 508 ++++++++++ .../nonebot_plugin_video_analysis/utils.py | 155 ++- .../nonebot_plugin_video_analysis/web_hub.py | 176 ++++ hexi/web/package-lock.json | 51 +- hexi/web/package.json | 3 +- hexi/web/src/plugins/video_analysis/index.tsx | 881 ++++++++++++++++++ pyproject.toml | 4 + requirements.txt | 1 + tests/test_video_group_file.py | 303 ++++++ tests/test_video_media_naming.py | 214 +++++ tests/test_video_policy.py | 315 +++++++ tests/test_web_jobs.py | 571 ++++++++++++ 29 files changed, 4838 insertions(+), 670 deletions(-) create mode 100644 hexi/plugins/nonebot_plugin_video_analysis/policy.py create mode 100644 hexi/plugins/nonebot_plugin_video_analysis/services/storage/group_file.py create mode 100644 hexi/plugins/nonebot_plugin_video_analysis/services/web_jobs.py create mode 100644 hexi/plugins/nonebot_plugin_video_analysis/web_hub.py create mode 100644 hexi/web/src/plugins/video_analysis/index.tsx create mode 100644 tests/test_video_group_file.py create mode 100644 tests/test_video_media_naming.py create mode 100644 tests/test_video_policy.py create mode 100644 tests/test_web_jobs.py diff --git a/hexi/plugins/nonebot_plugin_video_analysis/__init__.py b/hexi/plugins/nonebot_plugin_video_analysis/__init__.py index 31358ed..e9365a0 100644 --- a/hexi/plugins/nonebot_plugin_video_analysis/__init__.py +++ b/hexi/plugins/nonebot_plugin_video_analysis/__init__.py @@ -18,5 +18,22 @@ __plugin_meta__ = PluginMetadata( # 显式导入子模块:注册配置 schema + 消息 matcher(配合 load_plugins 只加载到包层) from . import config as _config # noqa: E402 from . import handlers as _handlers # noqa: E402 +from . import list_proc as _list_proc # noqa: E402 +from . import web_hub as _web_hub # noqa: E402 + +from hexi.web_hub.web_plugin_registry import register_web_plugin # noqa: E402 + +# 注册到统一 Web 管理台(/hub):hub 启动时挂载 /api/video_analysis, +# 前端页面 hexi/web/src/plugins/video_analysis/ +register_web_plugin( + "video_analysis", + "视频解析", + "video", + lambda: _web_hub.build_admin_app(), + module_name=__name__, +) _config.register_config() +# 导入期把群策略读进内存(含 list.json v1/v2 → v3 迁移), +# 之后 verify_user 走内存,不再碰磁盘。 +_list_proc.warmup() diff --git a/hexi/plugins/nonebot_plugin_video_analysis/cleanup.py b/hexi/plugins/nonebot_plugin_video_analysis/cleanup.py index 7fb1f5f..e8e4d68 100644 --- a/hexi/plugins/nonebot_plugin_video_analysis/cleanup.py +++ b/hexi/plugins/nonebot_plugin_video_analysis/cleanup.py @@ -10,7 +10,9 @@ hexi/data/temp 是下载媒体中转区(见 utils.get_temp_root),发送成功后 - /清理temp [天数] 清理 temp 下超过 N 天(默认取配置)未修改的文件 - /temp统计 查看 temp 目录占用情况 -目录结构始终保留,占用中的文件自动跳过。 +媒体按「作者目录/作品目录」分层落盘,文件删完后这些空目录会一并收掉 +(只删同为过期、且确实为空的目录 —— 正在落盘的目录 mtime 很新,不会误删)。 +占用中的文件自动跳过。 """ import asyncio @@ -71,11 +73,39 @@ def _walk_files(root: Path) -> list[Path]: return files +def _prune_empty_dirs(root: Path, max_age: float) -> int: + """自底向上收掉空目录(作者的层与作品的层都算),返回删除的目录数。 + + 只删「本身就是空」且 mtime 已过期的目录:正在落盘的目录刚建出来、 + mtime 很新,不会被误删;刚删完文件的目录 mtime 会被刷新,留到下一轮。 + root 自身不在 rglob 结果里,不会被删。 + """ + removed = 0 + # 目录另有 60s 下限:清理temp 0 时不能把"刚建出来、还没写第一个文件"的 + # 目录(下载落盘点先 mkdir 再 open)删掉 + dir_age = max(max_age, 60) + dirs = [p for p in root.rglob("*") if p.is_dir()] + # 深的先处理:子目录删掉后父目录才可能变空,同一轮里能被顺带收掉 + for path in sorted(dirs, key=lambda p: len(p.parts), reverse=True): + try: + if any(path.iterdir()): + continue + if not _file_is_stale(path, dir_age): + continue + path.rmdir() + removed += 1 + logger.info(f"temp 清理: 删除空目录 {path}") + except OSError: + # 被占用 / 刚被别的进程删掉 → 留待下轮 + continue + return removed + + def clean_temp_files(sub: str = "", days: int | None = None) -> tuple[int, int]: - """清理 temp[/sub] 下超过期限的文件。 + """清理 temp[/sub] 下超过期限的文件与随之空掉的目录。 Returns: - (removed, total) — 删除数、统计到的文件总数 + (removed, total) — 删除的文件数、统计到的文件总数 """ root = get_temp_root(sub) if not root.is_dir(): @@ -94,17 +124,22 @@ def clean_temp_files(sub: str = "", days: int | None = None) -> tuple[int, int]: except OSError as e: # 文件被占用(如发送中)等场景,留待下轮 logger.warning(f"temp 清理: 跳过 {path} ({e})") + + dirs = _prune_empty_dirs(root, max_age) + if dirs: + logger.info(f"temp 清理: 同时收掉 {dirs} 个空目录") return removed, len(files) def temp_stats(sub: str = "") -> dict: - """统计 temp[/sub] 目录:文件数、总大小(字节)""" + """统计 temp[/sub] 目录:文件数、总大小(字节)、目录数(含作者/作品层)""" root = get_temp_root(sub) if not root.is_dir(): - return {"files": 0, "bytes": 0} + return {"files": 0, "bytes": 0, "dirs": 0} files = _walk_files(root) total_bytes = sum(p.stat().st_size for p in files if p.exists()) - return {"files": len(files), "bytes": total_bytes} + dirs = sum(1 for p in root.rglob("*") if p.is_dir()) + return {"files": len(files), "bytes": total_bytes, "dirs": dirs} # ── 手动清理命令(manual 模式,auto 模式下也可用) ────────────── @@ -141,11 +176,18 @@ async def _handle_clean(bot: Bot, event: MessageEvent): async def _handle_stats(event: MessageEvent): st = await _run_stats() if st["files"] == 0: - await UniMessage.text("temp 目录目前是空的。").send() + if st.get("dirs"): + await UniMessage.text( + f"temp 目录下没有文件了,还剩 {st['dirs']} 个空目录" + f"(下次清理/清理temp 会一并收掉)。" + ).send() + else: + await UniMessage.text("temp 目录目前是空的。").send() else: size_mb = st["bytes"] / 1024 / 1024 await UniMessage.text( - f"temp 目录:共 {st['files']} 个文件,占用 {size_mb:.1f} MB。" + f"temp 目录:共 {st['files']} 个文件,占用 {size_mb:.1f} MB," + f"{st.get('dirs', 0)} 个目录。" ).send() diff --git a/hexi/plugins/nonebot_plugin_video_analysis/config.py b/hexi/plugins/nonebot_plugin_video_analysis/config.py index f29d679..8673e31 100644 --- a/hexi/plugins/nonebot_plugin_video_analysis/config.py +++ b/hexi/plugins/nonebot_plugin_video_analysis/config.py @@ -1,21 +1,48 @@ """统一配置注册:把插件配置按文档标准接入 hexi.web_hub.web_config(Web 可读可改)。 - temp 清理配置(env pydantic) → register_model_config +- 群文件投递配置(打包/解压密码, env pydantic) → register_model_config - S3 存储配置(原硬编码在 services/storage/s3.py) → register_config_items(store=s3 模块) -- 群分组配置(data/list.json) → register_config_items(type=json, nosave) + +群策略(data/list.json v3)不在这里注册:它有独立的 Web 页面与 API +(见 web_hub.py),不再走配置抽屉里的裸 JSON 编辑。 """ from __future__ import annotations +from nonebot import get_plugin_config +from pydantic import BaseModel + from hexi.web_hub.config_standard import register_config_items, register_model_config -from . import cleanup, list_proc # noqa: F401 +from . import cleanup # noqa: F401 from .services.storage import s3 as _s3mod # 插件模块名 = plugin_id(与 NoneBot 模块名一致) _PLUGIN_ID = __package__ +class GroupFileConfig(BaseModel): + # 上传群文件时打包成一个 zip(关掉则逐个上传原文件) + video_analysis_group_file_zip: bool = True + # 压缩包解压密码(留空 = 不加密; 设置后要求 pyzipper 可用, 否则放弃群文件上传) + video_analysis_group_file_password: str = "" + + +group_file_config = get_plugin_config(GroupFileConfig) + + +def group_file_settings() -> tuple[bool, str]: + """群文件投递设置:(是否打包成 zip, 解压密码)。 + + 运行期读取实例属性,所以 Web 保存后立即生效(apply 是 setattr)。 + """ + return ( + bool(group_file_config.video_analysis_group_file_zip), + str(group_file_config.video_analysis_group_file_password or ""), + ) + + def _reset_s3_caches(_values=None, store=None): """保存 S3 配置后清空懒加载客户端缓存,让下次上传用新配置重建。""" if store is None: @@ -59,7 +86,29 @@ def register_config() -> None: apply_extra=lambda _values, _conf: cleanup.reload_cleanup_config(), ) - # 2) S3 存储配置(来源无关) + # 2) 群文件投递:打包 / 解压密码 + register_model_config( + _PLUGIN_ID, + group_file_config, + fields=[ + "video_analysis_group_file_zip", + "video_analysis_group_file_password", + ], + labels={ + "video_analysis_group_file_zip": "群文件打包成 zip", + "video_analysis_group_file_password": "压缩包解压密码", + }, + descriptions={ + "video_analysis_group_file_zip": "开启后群文件收到的是一个压缩包;关闭则逐个上传原文件", + "video_analysis_group_file_password": "留空 = 不加密;设置后用 AES-256 加密(需要 7-Zip/WinRAR 等工具解压),缺 pyzipper 时会放弃群文件上传而不是传明文", + }, + types={ + "video_analysis_group_file_zip": "bool", + "video_analysis_group_file_password": "password", + }, + ) + + # 3) S3 存储配置(来源无关) register_config_items( _PLUGIN_ID, [ @@ -86,19 +135,3 @@ def register_config() -> None: store=_s3mod, apply_extra=_reset_s3_caches, ) - - # 3) 群分组配置(data/list.json, 权威源在插件自身) - register_config_items( - _PLUGIN_ID, - [ - { - "key": "group_config", - "label": "群分组配置", - "type": "json", - "description": "data/list.json 内容。groups 为 {群号: {auto, plan(A/B), auto_link[]}},blacklist 为禁用用户 QQ 列表。白名单即 groups 的键。", - "getter": list_proc.get_group_config_sync, - "setter": list_proc.set_group_config_sync, - "nosave": True, - } - ], - ) diff --git a/hexi/plugins/nonebot_plugin_video_analysis/handlers/douyin.py b/hexi/plugins/nonebot_plugin_video_analysis/handlers/douyin.py index 145e903..5afc0b7 100644 --- a/hexi/plugins/nonebot_plugin_video_analysis/handlers/douyin.py +++ b/hexi/plugins/nonebot_plugin_video_analysis/handlers/douyin.py @@ -12,6 +12,7 @@ from ..services.fetchers.douyin_api import fetch_douyin_content from ..services.fetchers.douyin_ssr import MOBILE_UA, fetch_douyin_note_ssr from ..models import DouyinFetchError from ..utils import get_data_dir, parse_netscape_cookies +from ..policy import Policy from .sender import PendingMedia, _as_paths SHORT_LINK_PATTERN = re.compile(r"(v\.douyin\.com/[A-Za-z0-9_\-]+)") @@ -111,11 +112,13 @@ async def process_douyin_res( file_paths: Union[Path, list[Path]], is_private: bool, image_post: bool, - plan: str | None = None, + policy: Policy | None = None, + platform: str | None = None, ) -> tuple[Optional[PendingMedia], Optional[str]]: """下载已完成 → 打包为待发送媒体(不上传、不发送、不清理) - 多级发送(temp 本地 → S3 链接 → 回退本地)由 send_pending_media 统一处理。 + 多级发送(temp 本地 → S3 链接 → 回退本地)与群文件上传由 + send_pending_media 统一按 policy 处理;platform 用于群文件限定平台。 """ if not file_paths: return None, None @@ -124,7 +127,8 @@ async def process_douyin_res( files=_as_paths(file_paths), image_post=image_post, is_private=is_private, - plan=plan, + policy=policy, + platform=platform, title=title, ), None, diff --git a/hexi/plugins/nonebot_plugin_video_analysis/handlers/entry.py b/hexi/plugins/nonebot_plugin_video_analysis/handlers/entry.py index 025439e..146bf31 100644 --- a/hexi/plugins/nonebot_plugin_video_analysis/handlers/entry.py +++ b/hexi/plugins/nonebot_plugin_video_analysis/handlers/entry.py @@ -16,7 +16,8 @@ from ..services.fetchers.rednote_content import fetch_rednote_content from .douyin import parse_douyin, process_douyin_res from .sender import PendingMedia, send_pending_media from .universal import handle_universal -from ..list_proc import AUTO_LINK_KEYWORDS, get_group_auto_link, verify_user +from ..list_proc import get_policy, is_group_whitelisted, is_user_blacklisted +from ..policy import Policy, match_platform URL_PATTERN = re.compile(r"(https?://\S+)") XCX_PATTERN = r"QQ小程序(?:&#93;|]|\])" @@ -44,106 +45,119 @@ active_video_handler = on_message(priority=10, block=False, rule=to_me()) async def _check_access( event: Event, *, auto_only: bool = False, msg: str | None = None -) -> tuple[bool, str | None]: +) -> tuple[bool, Policy | None]: """统一权限检查。 - auto_only=True → 自动解析:需白名单 + 开启自动解析,或 auto_link 关键词命中。 - auto_only=False → 主动触发:需非黑名单,群聊还需白名单。 + auto_only=True → 自动解析:需白名单群 + (开启自动解析或消息命中自动策略)。 + auto_only=False → 主动触发:需白名单群 + 非黑名单用户。 + + 私聊不做自动解析,读 default 节策略后直接解析。 + 禁用策略(ban_link)是消息级的(按链接判定),见 match_message。 Returns: - (allowed, plan) — plan 用于 S3 路由,不允许时为 None + (allowed, policy) — policy 决定存储/投递,不允许时为 None """ - white, black, auto, plan = await verify_user(event) target = get_target(event) if target.private: if auto_only: logger.info("权限分析:自动解析不处理私聊") return False, None - if black: + if is_user_blacklisted(event): logger.info(f"权限分析:黑名单用户私聊,不回复: {event.get_user_id()}") return False, None logger.info("权限分析:私聊,直接解析") - return True, None + return True, get_policy(event) group_id = str(event.group_id) - if not white: + if not is_group_whitelisted(event): logger.info(f"权限分析:群 {group_id} 不在白名单,不做处理") return False, None - if auto_only and not auto: - if msg is not None and await _match_auto_link(event, msg): - logger.info(f"权限分析:群 {group_id} 未开启自动解析,但自动链接关键词命中") - else: - logger.info(f"权限分析:群 {group_id} 未开启自动解析") - return False, None - - if not auto_only and black: + if is_user_blacklisted(event): logger.info(f"权限分析:黑名单用户,不回复: {event.get_user_id()}") return False, None + policy = get_policy(event) + + if auto_only and not policy.auto and not _match_auto_link(msg or "", policy): + logger.info(f"权限分析:群 {group_id} 未开启自动解析且未命中自动策略") + return False, None + logger.info( - f"权限分析:群 {group_id} 权限通过 — " - f"自动解析: {auto}, 方案: {plan or '默认(PLANC)'}" + f"权限分析:群 {group_id} 权限通过 — 自动解析: {policy.auto}, " + f"存储: {policy.plan}, 公网: {policy.upload_public}, " + f"群文件: {policy.upload_group_file}" ) - return True, plan + return True, policy -async def _match_auto_link(event: Event, msg: str) -> bool: - """消息中的 URL 是否命中群配置的 auto_link 关键词。""" - keywords = await get_group_auto_link(event) - if not keywords: +def _match_auto_link(msg: str, policy: Policy) -> bool: + """消息中的 URL 是否命中群策略的自动策略(auto_link)。""" + if not policy.auto_link: return False - urls = URL_PATTERN.findall(msg) - if not urls: - return False - for kw in keywords: - domains = AUTO_LINK_KEYWORDS.get(kw, (kw,)) - if any(any(domain in url for domain in domains) for url in urls): - logger.info(f"自动链接:关键词 {kw} 命中消息 {urls}") + for url in URL_PATTERN.findall(msg): + platform = policy.auto_matched(url) + if platform: + logger.info(f"自动策略:{platform} 命中消息 {url}") return True return False +def _skip_banned(url: str, policy: Policy) -> bool: + """链接是否命中禁用策略(命中即静默丢弃,只记日志)。""" + platform = policy.banned(url) + if platform: + logger.info(f"禁用策略:{platform} 已禁用,忽略链接 {url}") + return True + return False + + @auto_video_handler.handle() async def handle_auto_video(event: Event): msg = str(event.get_message()).strip() - allowed, plan = await _check_access(event, auto_only=True, msg=msg) - if allowed: - await match_message(event, plan=plan) + allowed, policy = await _check_access(event, auto_only=True, msg=msg) + if allowed and policy is not None: + await match_message(event, policy) @active_video_handler.handle() async def handle_active_video(event: Event): - allowed, plan = await _check_access(event, auto_only=False) - if allowed: - await match_message(event, plan=plan) + allowed, policy = await _check_access(event, auto_only=False) + if allowed and policy is not None: + await match_message(event, policy) -async def match_message(event: Event, plan: str | None = None): - """消息匹配与分派:文本链接 / QQ小程序卡片统一走 dispatch_url。""" +async def match_message(event: Event, policy: Policy): + """消息匹配与分派:文本链接 / QQ小程序卡片统一走 dispatch_url。 + + 命中禁用策略的链接在这里丢弃;消息里还有其它可用链接则继续解析。 + """ msg = str(event.get_message()).strip() logger.info(f"消息解析:获取到的消息:{msg}") is_private = get_target(event).private message = None public_url = None + url = "" if re.search(XCX_PATTERN, msg) or "CQ:json" in msg or "CQ:share" in msg: logger.info("消息解析:检测到 CQ 卡片") - url = await _extract_xcx_url(msg) + url = await _extract_xcx_url(msg) or "" logger.info(f"消息解析:卡片链接:{url}") if not url or not any(domain in url for domain in VALID_HOSTS): return - message, public_url = await dispatch_url(url, is_private, plan=plan) + if _skip_banned(url, policy): + return + message, public_url = await dispatch_url(url, is_private, policy) if not message: return else: - urls = URL_PATTERN.findall(msg) + urls = [u for u in URL_PATTERN.findall(msg) if not _skip_banned(u, policy)] for url in urls: logger.info(f"消息解析:作品链接:{url}") - message, public_url = await dispatch_url(url, is_private, plan=plan) + message, public_url = await dispatch_url(url, is_private, policy) if message: break if not message: @@ -151,12 +165,12 @@ async def match_message(event: Event, plan: str | None = None): if isinstance(message, PendingMedia): ok, pub = await send_pending_media(message, event) - if pub: + if pub and policy.sends_link: await UniMessage.text(f"{pub}").send() if not ok: await UniMessage.text(f"媒体发送失败:{url}").send() else: - if public_url: + if public_url and policy.sends_link: await UniMessage.text(f"{public_url}").send() await message.send() @@ -164,7 +178,7 @@ async def match_message(event: Event, plan: str | None = None): async def dispatch_url( url: str, is_private: bool, - plan: str | None = None, + policy: Policy, ) -> tuple[UniMessage | None, str | None]: """按平台分派解析(文本链接与小程序卡片共用)。""" url = url.rstrip(",。!?、;:)】》\"')") @@ -175,6 +189,9 @@ async def dispatch_url( logger.info(f"b23 短链重定向: {url} -> {resolved}") url = resolved + # 平台标签:短链重定向之后再判定(群文件限定平台用) + platform = match_platform(url) + if "douyin.com" in url or "v.douyin.com" in url or "iesdouyin.com" in url: await UniMessage.text("检测到链接,正在处理,请稍候...").send() try: @@ -186,7 +203,8 @@ async def dispatch_url( logger.warning(f"媒体解析:未能获取到作品:{url}") return None, None return await process_douyin_res( - title, parsed_path, is_private, image_post, plan=plan + title, parsed_path, is_private, image_post, + policy=policy, platform=platform, ) except Exception: await UniMessage.text(f"无法解析到媒体:{url}").send() @@ -213,7 +231,7 @@ async def dispatch_url( return None, None return await process_douyin_res( title, parsed_path, is_private, - isinstance(parsed_path, list), plan=plan, + isinstance(parsed_path, list), policy=policy, platform=platform, ) except Exception: await UniMessage.text(f"无法解析到内容:{url}").send() @@ -232,7 +250,7 @@ async def dispatch_url( return None, None return await process_douyin_res( title, parsed_path, is_private, - isinstance(parsed_path, list), plan=plan, + isinstance(parsed_path, list), policy=policy, platform=platform, ) except Exception: await UniMessage.text(f"无法解析到内容:{url}").send() @@ -242,7 +260,7 @@ async def dispatch_url( if any(domain in url for domain in VALID_HOSTS): await UniMessage.text("检测到链接,正在处理,请稍候...").send() try: - return await handle_universal(url, is_private, plan=plan) + return await handle_universal(url, is_private, policy, platform) except Exception as e: logger.exception(e) await UniMessage.text("下载过程中出现错误。").send() diff --git a/hexi/plugins/nonebot_plugin_video_analysis/handlers/sender.py b/hexi/plugins/nonebot_plugin_video_analysis/handlers/sender.py index 99cc0fc..25a5a29 100644 --- a/hexi/plugins/nonebot_plugin_video_analysis/handlers/sender.py +++ b/hexi/plugins/nonebot_plugin_video_analysis/handlers/sender.py @@ -10,6 +10,10 @@ (视频段不能与其他段混合,一条消息也放不下多段视频/图集体验差) → 统一走合并转发,一个图/视频一个节点。 +群文件(并行通道): `policy.upload_group_file` 开着时,消息链跑完后额外把文件 +传到群文件(默认打包成一个 zip,可配解压密码,见 config.group_file_settings), +失败只记日志、不影响发送结果(打包/上传实现在 services/storage/group_file.py)。 + temp 下的文件发送成功后也不清理(用户手动处理 data/temp)。 """ @@ -20,7 +24,11 @@ from nonebot import get_bot, get_driver, logger from nonebot.adapters import Event from nonebot_plugin_alconna import UniMessage +from ..config import group_file_settings +from ..policy import Policy +from ..services.storage.group_file import upload_group_files from ..services.storage.s3 import upload_with_plan +from ..utils import media_rel_dir_of @dataclass @@ -30,7 +38,9 @@ class PendingMedia: files: list[Path] image_post: bool = False is_private: bool = False - plan: str | None = None + policy: Policy | None = None + #: 平台规范标签(见 policy.match_platform),群文件限定平台用 + platform: str | None = None title: str = "" @@ -43,8 +53,12 @@ def _as_paths(file_paths: Path | list[Path]) -> list[Path]: # ─────────────────────── 合并转发(多媒体专用) ─────────────────────── +#: 按视频段发送的扩展名(其余按图片发;直链下载可能落 webm/mov 等) +_VIDEO_SUFFIXES = {".mp4", ".webm", ".mov", ".flv", ".mkv", ".ts"} + + def _is_video(fp: Path) -> bool: - return fp.suffix.lower() == ".mp4" + return fp.suffix.lower() in _VIDEO_SUFFIXES def _needs_forward(files: list[Path]) -> bool: @@ -117,13 +131,7 @@ async def _build_s3_forward_items( items: list[tuple[Path, str | None]] = [] public_url = None for fp in media.files: - local_url, pub = upload_with_plan( - fp, - plan=media.plan, - is_private=media.is_private, - title=media.title, - image_post=media.image_post, - ) + local_url, pub = upload_with_plan(fp, policy=media.policy) if not local_url: raise RuntimeError(f"上传本地 S3 失败: {fp}") if pub: @@ -136,10 +144,10 @@ async def _build_s3_forward_items( def _build_local_msg(files: list[Path], image_post: bool) -> UniMessage: - """本地文件版消息(mp4 → 视频,其余 → 图片)""" + """本地文件版消息(视频扩展名 → 视频段,其余 → 图片段)""" msg = UniMessage() for fp in files: - if fp.suffix.lower() == ".mp4": + if _is_video(fp): msg.video(path=fp) else: msg.image(path=fp) @@ -153,28 +161,22 @@ def _build_s3_msg( msg = UniMessage() public_url = None for fp in media.files: - local_url, pub = upload_with_plan( - fp, - plan=media.plan, - is_private=media.is_private, - title=media.title, - image_post=media.image_post, - ) + local_url, pub = upload_with_plan(fp, policy=media.policy) if not local_url: raise RuntimeError(f"上传本地 S3 失败: {fp}") if pub: public_url = pub - if fp.suffix.lower() == ".mp4": + if _is_video(fp): msg.video(url=local_url) else: msg.image(url=local_url) return msg, public_url -async def send_pending_media( +async def _send_media_core( media: PendingMedia, event: Event | None = None ) -> tuple[bool, str | None]: - """多级发送,返回 (是否成功, public_url) + """多级发送主体,返回 (是否成功, public_url) public_url 仅在走 S3 链接发送成功时返回(调用方决定是否发文字)。 temp 文件发送成功后保留(用户手动清理 data/temp)。 @@ -236,3 +238,47 @@ async def send_pending_media( except Exception as e: logger.exception(f"回退发送失败: {e}") return False, None + + +async def _upload_group_files(media: PendingMedia, event: Event | None) -> None: + """群文件并行通道:消息链跑完后按策略额外传一份(失败只记日志)。 + + 是否打包成 zip / 是否加密由全局配置决定(config.group_file_settings); + 平台清单(policy.group_file_platforms)非空时只传清单里的平台,其余平台 + 照常走消息、不传群文件。 + """ + policy = media.policy + if policy is None or event is None: + return + if not policy.allows_group_file(media.platform): + if policy.upload_group_file: + logger.info( + f"群文件限定平台 {policy.group_file_platforms}," + f"本次为 {media.platform or '未知平台'},跳过群文件上传" + ) + return + group_id = getattr(event, "group_id", None) + if group_id is None: + return + zip_files, password = group_file_settings() + await upload_group_files( + media.files, + int(group_id), + title=media.title, + zip_files=zip_files, + password=password, + policy=policy, + rel_dir=media_rel_dir_of(media.files[0]), + ) + + +async def send_pending_media( + media: PendingMedia, event: Event | None = None +) -> tuple[bool, str | None]: + """多级发送 + 群文件并行通道,返回 (是否成功, public_url)。""" + if not media.files: + return False, None + + ok, public_url = await _send_media_core(media, event) + await _upload_group_files(media, event) + return ok, public_url diff --git a/hexi/plugins/nonebot_plugin_video_analysis/handlers/universal.py b/hexi/plugins/nonebot_plugin_video_analysis/handlers/universal.py index f647043..a7a80d5 100644 --- a/hexi/plugins/nonebot_plugin_video_analysis/handlers/universal.py +++ b/hexi/plugins/nonebot_plugin_video_analysis/handlers/universal.py @@ -1,11 +1,11 @@ """通用平台视频解析编排层 — B站 / YouTube / Twitter 等""" -from pathlib import Path from typing import Optional from nonebot import logger from nonebot_plugin_alconna import UniMessage +from ..policy import Policy from ..services.fetchers.video_downloader import download_video from .sender import PendingMedia, _as_paths @@ -13,15 +13,16 @@ from .sender import PendingMedia, _as_paths async def handle_universal( url: str, is_private: bool, - plan: str | None = None, + policy: Policy | None = None, + platform: str | None = None, ) -> tuple[Optional[PendingMedia], Optional[str]]: """ - 下载通用平台视频 → 打包待发送媒体(上传/发送由 sender 多级处理) + 下载通用平台视频 → 打包待发送媒体(上传/发送由 sender 按 policy 多级处理) Returns: (PendingMedia, public_url) — None 表示下载失败 """ - video_file = await download_video(url) + video_file, title = await download_video(url) if not video_file: await UniMessage.text("视频下载失败。").send() return None, None @@ -33,8 +34,9 @@ async def handle_universal( files=_as_paths(video_file), image_post=False, is_private=is_private, - plan=plan, - title="title", + policy=policy, + platform=platform, + title=title, ), None, ) diff --git a/hexi/plugins/nonebot_plugin_video_analysis/list_proc.py b/hexi/plugins/nonebot_plugin_video_analysis/list_proc.py index 16aa01d..4917c80 100644 --- a/hexi/plugins/nonebot_plugin_video_analysis/list_proc.py +++ b/hexi/plugins/nonebot_plugin_video_analysis/list_proc.py @@ -1,359 +1,434 @@ -import json -import asyncio -import os +"""群策略管理 —— 命令入口与访问层。 + +策略模型、平台定义与存储(list.json v3)见 `policy.py`。 +本模块只做两件事:给 handlers 提供同步访问器(走内存),以及把命令解析成 +`STORE.update_group(...)` 调用。 + +命令统一入口 `视频策略`: + + 视频策略 查看当前群策略面板 + 视频策略 <群号> ... 管理员操作指定群 + 视频策略 自动 on|off + 视频策略 自动策略 +小红书 -抖音 命中即解析(即使关了自动解析) + 视频策略 禁用策略 +X 命中即不解析(自动/手动都不解析) + 视频策略 存储 A|B|C + 视频策略 公网 on|off + 视频策略 链接 on|off 发送下载链接(需先开公网) + 视频策略 群文件 on|off + 视频策略 黑名单 +QQ -QQ 全局用户黑名单 + 视频策略 白名单 +群号 -群号 + +所有设置项都要求超管(`check_admin`);设置时群会自动加入白名单。 +""" + +from __future__ import annotations + +from typing import Any -from nonebot.plugin.on import on_command -from nonebot import logger from nonebot.adapters import Event from nonebot.matcher import Matcher -from nonebot_plugin_alconna import UniMessage, get_target +from nonebot.plugin.on import on_command +from nonebot_plugin_alconna import UniMessage +from nonebot_plugin_alconna.uniseg import get_target from hexi.core.custom_utils import check_admin -# from hexi.core.message_utils import send_poke -BASE_DIR = os.path.dirname(os.path.abspath(__file__)) -FILE_PATH = os.path.join(BASE_DIR, "data", "list.json") +from .policy import ( + PLANS, + PLATFORMS, + STORE, + Policy, + apply_platform_diff, + normalize_platforms, +) -USER_DATA: dict[str, str] = {} -_LOADED = False -LOCK = asyncio.Lock() +# ─────────────────────────── 访问层 ─────────────────────────── -add_black_list = on_command("添加黑名单",rule=check_admin) -add_white_list = on_command("添加白名单",rule=check_admin) -add_auto_list = on_command("添加自动名单",rule=check_admin) + +def warmup() -> None: + """导入期调用一次:把 list.json 读进内存(含 v1/v2 → v3 迁移)。""" + STORE.load() + + +def get_policy(event: Event) -> Policy: + """事件对应的策略:私聊读 default 节,群聊读该群策略。""" + if get_target(event).private: + return STORE.default_policy() + return STORE.get(str(event.group_id)) + + +def is_group_whitelisted(event: Event) -> bool: + """私聊不适用(返回 False,调用方应先判断私聊)。""" + return STORE.is_whitelisted(str(getattr(event, "group_id", ""))) + + +def is_user_blacklisted(event: Event) -> bool: + return STORE.is_blacklisted(event.get_user_id()) + + +# ─────────────────────────── 命令解析 ─────────────────────────── + +_TRUE_WORDS = {"on", "开", "true", "1", "yes", "启用"} +_FALSE_WORDS = {"off", "关", "false", "0", "no", "禁用"} + +TOGGLE_USAGE = "用法:视频策略 自动|公网|链接|群文件 on|off" + + +def _args(event: Event, matcher: Matcher) -> list[str]: + """去掉命令前缀后的参数列表。""" + text = event.get_message().extract_plain_text().strip() + cmd = matcher.state["_prefix"]["command"][0] + return text.replace(cmd, "", 1).strip().split() + + +def _parse_onoff(raw: str) -> bool | None: + word = raw.strip().lower() + if word in _TRUE_WORDS: + return True + if word in _FALSE_WORDS: + return False + return None + + +def _split_target(args: list[str], event: Event) -> tuple[str | None, list[str]]: + """首个参数是纯数字群号时视为操作目标群,返回 (群号, 剩余参数)。""" + if args and args[0].isdigit() and len(args[0]) >= 5: + return args[0], args[1:] + return None, args + + +def _fmt_platforms(values: list[str]) -> str: + return "、".join(values) if values else "—" + + +def _panel(group_id: str | None, policy: Policy, whitelisted: bool) -> str: + head = f"群 {group_id} 策略" if group_id else "私聊/默认策略" + if group_id and not whitelisted: + head += "(未加入白名单,本群不会自动解析)" + link_state = "开" if policy.sends_link else "关" + if policy.send_link and not policy.upload_public: + link_state += "(未上传公网,实际不发)" + file_state = "开" if policy.upload_group_file else "关" + if policy.upload_group_file and policy.group_file_platforms: + file_state += f"(仅 {'、'.join(policy.group_file_platforms)})" + return "\n".join( + [ + head, + f"自动解析:{'开' if policy.auto else '关'}", + f"自动策略:{_fmt_platforms(policy.auto_link)}", + f"禁用策略:{_fmt_platforms(policy.ban_link)}", + f"存储策略:{policy.plan}", + f"上传公网:{'开' if policy.upload_public else '关'}", + f"发送链接:{link_state}", + f"上传群文件:{file_state}", + f"平台标签:{'、'.join(PLATFORMS)}", + ] + ) + + +# ─────────────────────────── 命令注册 ─────────────────────────── + +policy_cmd = on_command("视频策略", rule=check_admin) +add_white_list = on_command("添加白名单", rule=check_admin) +remove_white_list = on_command("移除白名单", rule=check_admin) +add_black_list = on_command("添加黑名单", rule=check_admin) +remove_black_list = on_command("移除黑名单", rule=check_admin) +# 旧命令保留兼容 +add_auto_list = on_command("添加自动名单", rule=check_admin) set_plan_cmd = on_command("设置方案", rule=check_admin) set_auto_link_cmd = on_command("设置自动链接", rule=check_admin) -# 自动链接关键词 → 域名匹配表(auto_link 配置项使用) -AUTO_LINK_KEYWORDS = { - "xhs": ("xiaohongshu.com", "xhslink.com", "xhslink.cn"), - "bilibili": ("bilibili.com", "b23.tv", "bili2233.cn"), - "b23": ("bilibili.com", "b23.tv", "bili2233.cn"), - "douyin": ("douyin.com", "v.douyin.com", "iesdouyin.com", - "m.douyin.com", "jingxuan.douyin.com"), - "yt": ("youtube.com", "youtu.be"), - "youtube": ("youtube.com", "youtu.be"), - "x": ("x.com", "twitter.com"), - "twitter": ("x.com", "twitter.com"), -} + +async def _target_group( + args: list[str], event: Event, matcher: Matcher +) -> tuple[str | None, list[str]]: + """解析目标群号与剩余参数;未指定时用当前群(私聊则报错返回 None)。""" + group_id, rest = _split_target(args, event) + if group_id is not None: + return group_id, rest + if get_target(event).private: + cmd = matcher.state["_prefix"]["command"][0] + await UniMessage.text(f"私聊下请带上群号,如:{cmd} 123456789 自动 on").send() + return None, rest + return str(event.group_id), rest -@add_black_list.handle() -async def handle_add_black(event: Event, matcher: Matcher): - input_id = event.get_message().extract_plain_text().strip() - cmd = matcher.state["_prefix"]["command"][0] - input_id = input_id.replace(cmd, "").strip() - ok = await _add_to_blacklist(input_id) - if ok: - await UniMessage.text(f"{input_id} 已加入黑名单").send() +@policy_cmd.handle() +async def handle_policy(event: Event, matcher: Matcher): + args = _args(event, matcher) + group_id, rest = await _target_group(args, event, matcher) + if group_id is None: + return + + # 无子命令 → 面板 + if not rest: + policy = STORE.get(group_id) + await UniMessage.text( + _panel(group_id, policy, STORE.is_whitelisted(group_id)) + ).send() + return + + what, options = rest[0], rest[1:] + changes: dict[str, Any] = {} + note = "" + + if what in ("自动", "自动解析"): + if len(options) != 1 or (value := _parse_onoff(options[0])) is None: + await UniMessage.text(TOGGLE_USAGE).send() + return + changes["auto"] = value + note = f"自动解析已{'开启' if value else '关闭'}" + + elif what in ("自动策略", "自动链接"): + platforms = apply_platform_diff(STORE.get(group_id).auto_link, options) + if platforms is None: + await UniMessage.text( + f"用法:视频策略 自动策略 +平台/-平台(平台:{'、'.join(PLATFORMS)})" + ).send() + return + changes["auto_link"] = platforms + note = f"自动策略已设为:{_fmt_platforms(platforms)}" + + elif what in ("禁用策略", "禁用链接"): + platforms = apply_platform_diff(STORE.get(group_id).ban_link, options) + if platforms is None: + await UniMessage.text( + f"用法:视频策略 禁用策略 +平台/-平台(平台:{'、'.join(PLATFORMS)})" + ).send() + return + changes["ban_link"] = platforms + note = f"禁用策略已设为:{_fmt_platforms(platforms)}" + + elif what in ("存储", "方案", "存储策略"): + plan = options[0].strip().upper() if options else "" + if plan not in PLANS: + await UniMessage.text("用法:视频策略 存储 A|B|C").send() + return + changes["plan"] = plan + note = f"存储策略已设为 {plan}" + + elif what in ("公网", "上传公网"): + if len(options) != 1 or (value := _parse_onoff(options[0])) is None: + await UniMessage.text(f"用法:视频策略 {what} on|off").send() + return + changes["upload_public"] = value + note = f"上传公网已{'开启' if value else '关闭'}" + if not value and STORE.get(group_id).send_link: + note += "(发送链接已开但无公网链接,实际不会发)" + + elif what in ("链接", "发送链接", "下载链接"): + if len(options) != 1 or (value := _parse_onoff(options[0])) is None: + await UniMessage.text(f"用法:视频策略 {what} on|off").send() + return + changes["send_link"] = value + note = f"发送下载链接已{'开启' if value else '关闭'}" + if value and not STORE.get(group_id).upload_public: + note += ";当前未开启上传公网,需先:视频策略 公网 on" + + elif what in ("群文件平台", "群文件限定平台"): + platforms = apply_platform_diff( + STORE.get(group_id).group_file_platforms, options + ) + if platforms is None: + await UniMessage.text( + "用法:视频策略 群文件平台 +平台 -平台" + f"(平台:{'、'.join(PLATFORMS)};留空 = 全部平台)" + ).send() + return + changes["group_file_platforms"] = platforms + note = ( + f"群文件限定平台已设为:{_fmt_platforms(platforms)}" + if platforms + else "群文件限定平台已清空(所有平台都传群文件)" + ) + + elif what in ("群文件", "上传群文件"): + if len(options) != 1 or (value := _parse_onoff(options[0])) is None: + await UniMessage.text(f"用法:视频策略 {what} on|off").send() + return + changes["upload_group_file"] = value + note = f"上传群文件已{'开启' if value else '关闭'}" + + elif what == "黑名单": + note = await _handle_blacklist(options) + if note is None: + return + + elif what == "白名单": + note = await _handle_whitelist(options) + if note is None: + return + else: - await UniMessage.text(f"{input_id} 已在黑名单中").send() + await UniMessage.text( + "用法:视频策略 [群号] 自动|自动策略|禁用策略|存储|公网|链接|" + "群文件|群文件平台|黑名单|白名单 ..." + ).send() + return + + if changes: + policy = await STORE.update_group(group_id, create=True, **changes) + if policy is None: + await UniMessage.text(f"群 {group_id} 不在白名单中,请先添加白名单").send() + return + note += "\n\n" + _panel(group_id, policy, STORE.is_whitelisted(group_id)) + + await UniMessage.text(note).send() + + +async def _handle_blacklist(options: list[str]) -> str | None: + """全局黑名单增删,返回提示语;参数非法返回 None(已回复用法)。""" + if not options: + current = "、".join(STORE.blacklist()) or "—" + return f"全局黑名单:{current}\n用法:视频策略 黑名单 +QQ -QQ" + for token in options: + qq = token[1:].strip() + if token[:1] not in "+-" or not qq.isdigit(): + return "用法:视频策略 黑名单 +QQ -QQ" + added, removed = [], [] + for token in options: + qq = token[1:].strip() + if token[0] == "+" and await STORE.add_blacklist(qq): + added.append(qq) + elif token[0] == "-" and await STORE.remove_blacklist(qq): + removed.append(qq) + parts = [] + if added: + parts.append(f"已加入黑名单:{'、'.join(added)}") + if removed: + parts.append(f"已移出黑名单:{'、'.join(removed)}") + return "\n".join(parts) if parts else "黑名单无变化" + + +async def _handle_whitelist(options: list[str]) -> str | None: + """白名单增删,返回提示语;参数非法返回 None(已回复用法)。""" + if not options: + current = "、".join(sorted(STORE.all_groups())) or "—" + return f"白名单:{current}\n用法:视频策略 白名单 +群号 -群号" + for token in options: + gid = token[1:].strip() + if token[:1] not in "+-" or not gid.isdigit(): + return "用法:视频策略 白名单 +群号 -群号" + added, removed = [], [] + for token in options: + gid = token[1:].strip() + if token[0] == "+" and await STORE.update_group(gid, create=True) is not None: + added.append(gid) + elif token[0] == "-" and await STORE.remove_group(gid): + removed.append(gid) + parts = [] + if added: + parts.append(f"已加入白名单:{'、'.join(added)}") + if removed: + parts.append(f"已移除白名单:{'、'.join(removed)}") + return "\n".join(parts) if parts else "白名单无变化" + + +# ───────────────────── 旧命令(兼容保留) ───────────────────── @add_white_list.handle() async def handle_add_white(event: Event, matcher: Matcher): - input_id = event.get_message().extract_plain_text().strip() - cmd = matcher.state["_prefix"]["command"][0] - input_id = input_id.replace(cmd, "").strip() - ok = await add_white_user(input_id) - if ok: - await UniMessage.text(f"群 {input_id} 已加入白名单").send() - else: - await UniMessage.text(f"群 {input_id} 已在白名单中").send() + args = _args(event, matcher) + group_id = args[0].strip() if args else "" + if not group_id.isdigit(): + await UniMessage.text("用法:添加白名单 <群号>").send() + return + if STORE.is_whitelisted(group_id): + await UniMessage.text(f"群 {group_id} 已在白名单中").send() + return + await STORE.update_group(group_id, create=True) + await UniMessage.text(f"群 {group_id} 已加入白名单").send() + + +@remove_white_list.handle() +async def handle_remove_white(event: Event, matcher: Matcher): + args = _args(event, matcher) + group_id = args[0].strip() if args else "" + if not group_id.isdigit(): + await UniMessage.text("用法:移除白名单 <群号>").send() + return + ok = await STORE.remove_group(group_id) + await UniMessage.text( + f"群 {group_id} 已移出白名单" if ok else f"群 {group_id} 不在白名单中" + ).send() + + +@add_black_list.handle() +async def handle_add_black(event: Event, matcher: Matcher): + args = _args(event, matcher) + user_id = args[0].strip() if args else "" + if not user_id.isdigit(): + await UniMessage.text("用法:添加黑名单 ").send() + return + ok = await STORE.add_blacklist(user_id) + await UniMessage.text( + f"{user_id} 已加入黑名单" if ok else f"{user_id} 已在黑名单中" + ).send() + + +@remove_black_list.handle() +async def handle_remove_black(event: Event, matcher: Matcher): + args = _args(event, matcher) + user_id = args[0].strip() if args else "" + if not user_id.isdigit(): + await UniMessage.text("用法:移除黑名单 ").send() + return + ok = await STORE.remove_blacklist(user_id) + await UniMessage.text( + f"{user_id} 已移出黑名单" if ok else f"{user_id} 不在黑名单中" + ).send() @add_auto_list.handle() async def handle_add_auto(event: Event, matcher: Matcher): - input_id = event.get_message().extract_plain_text().strip() - cmd = matcher.state["_prefix"]["command"][0] - input_id = input_id.replace(cmd, "").strip() - ok = await add_auto_user(input_id) - if ok: - await UniMessage.text(f"群 {input_id} 已开启自动解析").send() - else: - await UniMessage.text(f"群 {input_id} 已开启自动解析,无需重复设置").send() + args = _args(event, matcher) + group_id = args[0].strip() if args else "" + if not group_id.isdigit(): + await UniMessage.text("用法:添加自动名单 <群号>").send() + return + policy = STORE.get(group_id) + if policy.auto and STORE.is_whitelisted(group_id): + await UniMessage.text(f"群 {group_id} 已开启自动解析,无需重复设置").send() + return + await STORE.update_group(group_id, create=True, auto=True) + await UniMessage.text(f"群 {group_id} 已开启自动解析").send() @set_plan_cmd.handle() async def handle_set_plan(event: Event, matcher: Matcher): - input_text = event.get_message().extract_plain_text().strip() - cmd = matcher.state["_prefix"]["command"][0] - args = input_text.replace(cmd, "").strip().split() - if len(args) != 2: - await UniMessage.text("用法:设置方案 <群号> A/B").send() + args = _args(event, matcher) + if len(args) != 2 or args[1].upper() not in PLANS: + await UniMessage.text("用法:设置方案 <群号> A/B/C").send() return group_id, plan = args[0], args[1].upper() - ok = await set_group_plan(group_id, plan) - if ok: - await UniMessage.text(f"群 {group_id} 存储方案已设为 {plan}").send() - elif plan not in ("A", "B"): - await UniMessage.text("方案必须是 A 或 B").send() - else: + policy = await STORE.update_group(group_id, plan=plan) + if policy is None: await UniMessage.text(f"群 {group_id} 不在白名单中,请先添加白名单").send() + return + await UniMessage.text(f"群 {group_id} 存储方案已设为 {plan}").send() + @set_auto_link_cmd.handle() async def handle_set_auto_link(event: Event, matcher: Matcher): - input_text = event.get_message().extract_plain_text().strip() - cmd = matcher.state["_prefix"]["command"][0] - args = input_text.replace(cmd, "").strip().split() - if len(args) < 2: + args = _args(event, matcher) + if not args: await UniMessage.text( - f"用法:设置自动链接 <群号> <关键词...>(关键词:{'/'.join(AUTO_LINK_KEYWORDS)})" + f"用法:设置自动链接 <群号> <平台...>(平台:{'、'.join(PLATFORMS)})" ).send() return group_id = args[0] - keywords = args[1:] - ok = await set_group_auto_link(group_id, keywords) - if ok: - if keywords: - await UniMessage.text( - f"群 {group_id} 自动链接关键词已设为: {'、'.join(keywords)}\n" - "匹配到对应平台链接时,即使未开启自动解析也会自动下载" - ).send() - else: - await UniMessage.text(f"群 {group_id} 的自动链接关键词已清空").send() - else: - await UniMessage.text(f"群 {group_id} 不在白名单中,请先添加白名单").send() - - -async def _safe_write_json(data, file_path: str): - tmp_path = file_path + ".tmp" - - # 在线程池中执行耗时的文件写入 - await asyncio.to_thread(_write_json_sync, data, tmp_path) - - # os.replace 是轻量级系统调用,通常很快,可直接在主线程执行 - # (也可放 to_thread,但一般没必要) - os.replace(tmp_path, file_path) - -def _write_json_sync(data, tmp_path: str): - """同步写入函数,供 to_thread 调用""" - with open(tmp_path, "w", encoding="utf-8") as f: - json.dump(data, f, ensure_ascii=False, indent=2) - - -def _migrate_to_v2(data: dict) -> tuple[dict, bool]: - """将旧格式(平铺数组)转换为新格式(group-centric map)""" - if "groups" in data: - return data, False # 已是 v2 - - white = data.get("WHITE_LIST", []) - auto_list = data.get("AUTO_ANALYSIS", []) - pa = data.get("PLANA", []) - pb = data.get("PLANB", []) - - groups: dict[str, dict] = {} - for gid in white: - entry: dict = {} - if gid in auto_list: - entry["auto"] = True - if gid in pa: - entry["plan"] = "A" - elif gid in pb: - entry["plan"] = "B" - groups[gid] = entry - - return { - "groups": groups, - "blacklist": data.get("BLACK_LIST", []), - }, True - -async def load_list(file_path: str = FILE_PATH): - global USER_DATA, _LOADED - - if _LOADED: + platforms = normalize_platforms(args[1:]) if len(args) > 1 else [] + if len(args) > 1 and not platforms: + await UniMessage.text(f"平台标签无效,可选:{'、'.join(PLATFORMS)}").send() return - - async with LOCK: - if _LOADED: - return - - exists = await asyncio.to_thread(os.path.exists, file_path) - if not exists: - USER_DATA = {"groups": {}, "blacklist": []} - await _safe_write_json(USER_DATA, file_path) - else: - content = await asyncio.to_thread(_read_file_sync, file_path) - raw = json.loads(content) - USER_DATA, migrated = _migrate_to_v2(raw) - if migrated: - await _safe_write_json(USER_DATA, file_path) - logger.info("list.json 已从旧格式迁移为新 group-centric 格式") - - _LOADED = True - - -def _read_file_sync(file_path: str) -> str: - with open(file_path, "r", encoding="utf-8") as f: - return f.read() - -async def verify_user(event: Event) -> tuple[bool, bool, bool, str | None]: - """ - 验证用户/群权限 - - Returns: - is_white — 群在白名单 - is_black — 用户在黑名单 - is_auto — 群开启自动解析 - plan — 存储方案 "A" / "B" / None - """ - await load_list() - user_id = event.get_user_id() - - groups = USER_DATA.get("groups", {}) - blacklist = USER_DATA.get("blacklist", []) - - is_black = user_id in blacklist - plan = None - is_auto = False - - if get_target(event).private: - logger.debug(f"[DEBUG] private chat user_id: {repr(user_id)}") - return True, is_black, False, None - - group_id = str(event.group_id) - group_config = groups.get(group_id, {}) - is_white = group_id in groups - is_auto = group_config.get("auto", False) - plan = group_config.get("plan") - - logger.debug(f"[DEBUG] user_id: {repr(user_id)} group_id: {group_id}") - logger.debug(f"[DEBUG] is_white: {is_white} is_black: {is_black} " - f"is_auto: {is_auto} plan: {plan}") - - return is_white, is_black, is_auto, plan - - -async def _add_to_blacklist(new_id: str) -> bool: - await load_list() - async with LOCK: - lst: list = USER_DATA.get("blacklist", []) - if new_id in lst: - return False - lst.append(new_id) - USER_DATA["blacklist"] = lst - await _safe_write_json(USER_DATA, FILE_PATH) - return True - - -async def add_white_user(new_id: str) -> bool: - """添加群到白名单(groups map)""" - await load_list() - async with LOCK: - groups: dict = USER_DATA.get("groups", {}) - if new_id in groups: - return False - groups[new_id] = {} - USER_DATA["groups"] = groups - await _safe_write_json(USER_DATA, FILE_PATH) - return True - - -async def add_auto_user(new_id: str) -> bool: - """设置群自动解析(群不在白名单则自动加入)""" - await load_list() - async with LOCK: - groups: dict = USER_DATA.get("groups", {}) - if new_id not in groups: - groups[new_id] = {} - if groups[new_id].get("auto"): - return False - groups[new_id]["auto"] = True - USER_DATA["groups"] = groups - await _safe_write_json(USER_DATA, FILE_PATH) - return True - - -async def set_group_plan(group_id: str, plan: str) -> bool: - """设置群的存储方案(A 或 B),群必须在白名单中""" - if plan not in ("A", "B"): - return False - await load_list() - async with LOCK: - groups: dict = USER_DATA.get("groups", {}) - if group_id not in groups: - return False - groups[group_id]["plan"] = plan - USER_DATA["groups"] = groups - await _safe_write_json(USER_DATA, FILE_PATH) - return True - - -async def get_group_auto_link(event: Event) -> list[str]: - """获取群配置的自动链接关键词(auto_link),私聊返回空列表""" - await load_list() - if get_target(event).private: - return [] - group_id = str(event.group_id) - return USER_DATA.get("groups", {}).get(group_id, {}).get("auto_link", []) - - -async def set_group_auto_link(group_id: str, keywords: list[str]) -> bool: - """设置群的自动链接关键词(auto_link),群必须在白名单中 - - keywords 为空列表时清空该配置。 - """ - await load_list() - async with LOCK: - groups: dict = USER_DATA.get("groups", {}) - if group_id not in groups: - return False - if keywords: - groups[group_id]["auto_link"] = keywords - else: - groups[group_id].pop("auto_link", None) - USER_DATA["groups"] = groups - await _safe_write_json(USER_DATA, FILE_PATH) - return True - - -# ── Web「分组配置」同步读写(list.json) ───────────────────────────── -def get_group_config_sync() -> dict: - """Web 读取用:确保已加载并返回 list.json 的完整结构(dict)。""" - global USER_DATA, _LOADED - if not _LOADED: - if os.path.exists(FILE_PATH): - try: - raw = json.loads(_read_file_sync(FILE_PATH)) - USER_DATA, _ = _migrate_to_v2(raw) - except Exception: # noqa: BLE001 - USER_DATA = {"groups": {}, "blacklist": []} - else: - USER_DATA = {"groups": {}, "blacklist": []} - _LOADED = True - return USER_DATA - - -def set_group_config_sync(new_data: dict) -> bool: - """Web 保存用:校验结构 → 更新内存 → 同步写回 list.json。 - - 只保留合法字段(auto/plan/auto_link/blacklist),避免脏数据。 - """ - global USER_DATA - if not isinstance(new_data, dict): - return False - groups = new_data.get("groups", {}) - blacklist = new_data.get("blacklist", []) - if not isinstance(groups, dict) or not isinstance(blacklist, list): - return False - - norm_groups: dict[str, dict] = {} - for gid, entry in groups.items(): - if not isinstance(entry, dict): - continue - e: dict = {} - if entry.get("auto"): - e["auto"] = True - plan = str(entry.get("plan", "")).upper() - if plan in ("A", "B"): - e["plan"] = plan - if isinstance(entry.get("auto_link"), list): - e["auto_link"] = [str(x) for x in entry["auto_link"]] - norm_groups[str(gid)] = e - - norm = {"groups": norm_groups, "blacklist": [str(x) for x in blacklist]} - USER_DATA = norm - _write_json_sync(norm, FILE_PATH + ".tmp") - os.replace(FILE_PATH + ".tmp", FILE_PATH) - return True - + if await STORE.update_group(group_id, auto_link=platforms) is None: + await UniMessage.text(f"群 {group_id} 不在白名单中,请先添加白名单").send() + return + if platforms: + await UniMessage.text( + f"群 {group_id} 自动策略已设为:{_fmt_platforms(platforms)}\n" + "匹配到对应平台链接时,即使未开启自动解析也会自动下载" + ).send() + else: + await UniMessage.text(f"群 {group_id} 的自动策略已清空").send() diff --git a/hexi/plugins/nonebot_plugin_video_analysis/policy.py b/hexi/plugins/nonebot_plugin_video_analysis/policy.py new file mode 100644 index 0000000..e810d04 --- /dev/null +++ b/hexi/plugins/nonebot_plugin_video_analysis/policy.py @@ -0,0 +1,486 @@ +"""群策略模型与存储(list.json v3)。 + +策略以「群」为单位,字段全部可选,缺省值见 `Policy`: + + auto 自动解析 默认 False + auto_link 自动策略(命中平台即解析) 默认 [] + ban_link 禁用策略(命中平台不解析) 默认 [] + plan 存储策略 A/B/C 默认 C + upload_public 上传公网 默认 False + send_link 发送下载链接 默认 False(依赖 upload_public) + upload_group_file 上传群文件(并行通道) 默认 False + group_file_platforms 群文件限定平台 默认 [](空 = 全部平台) + +文件结构(v3):: + + { + "groups": {"<群号>": {}}, + "default": {}, # 私聊 + 群条目缺字段时的兜底 + "blacklist": ["", ...] # 全局用户黑名单 + } + +读路径全内存:`verify_user` 在每条消息的热路径上,首次 `load()` 之后不再读盘。 +写路径走 `asyncio.Lock` + `.tmp` 原子替换;Web 子应用与 bot 同进程同事件循环, +asyncio 锁即可覆盖两边并发写(Web 不在独立线程里跑写操作)。 + +换 ORM 时只需再实现一个同样接口的 Store,调用方零改动。 +""" + +from __future__ import annotations + +import asyncio +import json +import os +import re +from collections.abc import Iterable +from dataclasses import dataclass, field, replace +from pathlib import Path +from typing import Any + +from nonebot import logger + +# ─────────────────────────── 平台 ─────────────────────────── + +XHS = "小红书" +BILIBILI = "哔哩哔哩" +DOUYIN = "抖音" +YOUTUBE = "Youtube" +X = "X" + +#: 规范标签(存储值 = 展示值,命令/Web 输入同一个词) +PLATFORMS: tuple[str, ...] = (XHS, BILIBILI, DOUYIN, YOUTUBE, X) + +#: 平台 → 域名片段(判定 auto_link / ban_link 命中用) +PLATFORM_DOMAINS: dict[str, tuple[str, ...]] = { + XHS: ("xiaohongshu.com", "xhslink.com", "xhslink.cn"), + BILIBILI: ("bilibili.com", "b23.tv", "bili2233.cn"), + DOUYIN: ( + "douyin.com", + "iesdouyin.com", + "m.douyin.com", + "jingxuan.douyin.com", + ), + YOUTUBE: ("youtube.com", "youtu.be"), + X: ("x.com", "twitter.com"), +} + +#: 输入别名 → 规范标签(统一小写比较) +_PLATFORM_ALIASES: dict[str, str] = { + "xhs": XHS, + "小红书": XHS, + "bilibili": BILIBILI, + "b23": BILIBILI, + "b站": BILIBILI, + "哔哩哔哩": BILIBILI, + "douyin": DOUYIN, + "抖音": DOUYIN, + "yt": YOUTUBE, + "youtube": YOUTUBE, + "油管": YOUTUBE, + "x": X, + "twitter": X, + "推特": X, +} + +#: 分隔符:兼容手写 JSON 里的 "小红书,xhs 抖音" 这类写法 +_SPLIT_RE = re.compile(r"[,,、/\s]+") + + +def normalize_platform(raw: Any) -> str | None: + """用户输入 → 规范标签,未知返回 None。""" + if not isinstance(raw, str): + return None + return _PLATFORM_ALIASES.get(raw.strip().lower()) + + +def normalize_platforms(values: Any) -> list[str]: + """归一 + 去重 + 丢弃非法值,结果按 PLATFORMS 顺序排列。 + + 接受字符串(按分隔符拆)或列表。 + """ + if values is None: + items: Iterable[Any] = () + elif isinstance(values, str): + items = _SPLIT_RE.split(values) + elif isinstance(values, (list, tuple, set)): + items = list(values) + else: + return [] + + picked = {p for p in (normalize_platform(v) for v in items) if p} + return [p for p in PLATFORMS if p in picked] + + +def apply_platform_diff(current: Iterable[str], tokens: Iterable[str]) -> list[str] | None: + """按 `+平台` / `-平台` 增删;出现裸平台名时整体覆盖。 + + 全部 token 都带 +/- 时做增量,否则视为覆盖(命令 `视频策略 自动策略 …` 用)。 + 含未知平台(或覆盖时没有合法平台)返回 None,由调用方提示用法。 + """ + tokens = [t for t in tokens if str(t).strip()] + if not tokens: + return None + + if all(str(t)[:1] in "+-" for t in tokens): + result = normalize_platforms(current) + for token in tokens: + picked = normalize_platforms([str(token)[1:]]) + if not picked: + return None + if str(token)[0] == "+" and picked[0] not in result: + result.append(picked[0]) + elif str(token)[0] == "-" and picked[0] in result: + result.remove(picked[0]) + return normalize_platforms(result) + + picked = normalize_platforms(tokens) + return picked or None + + +def match_platform(url: str) -> str | None: + """URL → 平台规范标签;不属于任何已支持平台时返回 None。""" + for name, domains in PLATFORM_DOMAINS.items(): + if any(domain in url for domain in domains): + return name + return None + + +# ─────────────────────────── 策略 ─────────────────────────── + +PLANS: tuple[str, ...] = ("A", "B", "C") +DEFAULT_PLAN = "C" + +POLICY_FIELDS: tuple[str, ...] = ( + "auto", + "auto_link", + "ban_link", + "plan", + "upload_public", + "send_link", + "upload_group_file", + "group_file_platforms", +) + + +@dataclass +class Policy: + """单个群的策略(字段缺省即默认值,见类文档)。""" + + auto: bool = False + auto_link: list[str] = field(default_factory=list) + ban_link: list[str] = field(default_factory=list) + plan: str = DEFAULT_PLAN + upload_public: bool = False + send_link: bool = False + upload_group_file: bool = False + #: 群文件限定平台;空列表 = 所有平台都传群文件 + group_file_platforms: list[str] = field(default_factory=list) + + def __post_init__(self) -> None: + """构造即归一:直接 Policy(...) 传入平台别名/非法 plan 也不至于静默失效。""" + self.auto_link = normalize_platforms(self.auto_link) + self.ban_link = normalize_platforms(self.ban_link) + self.group_file_platforms = normalize_platforms(self.group_file_platforms) + if self.plan not in PLANS: + self.plan = DEFAULT_PLAN + + @property + def sends_link(self) -> bool: + """是否真的发下载链接:没有公网链接可发时恒为 False。""" + return self.send_link and self.upload_public + + def allows_group_file(self, platform: str | None) -> bool: + """这个平台的作品要不要传群文件。 + + 开关关着 → 不传;清单为空 → 全部平台都传;否则只认清单里的平台 + (平台识别不出来时按"不在清单里"处理)。 + """ + if not self.upload_group_file: + return False + if not self.group_file_platforms: + return True + return platform is not None and platform in self.group_file_platforms + + def banned(self, url: str) -> str | None: + """URL 命中的禁用平台(未命中返回 None)。""" + platform = match_platform(url) + return platform if platform and platform in self.ban_link else None + + def auto_matched(self, url: str) -> str | None: + """URL 命中的自动策略平台(未命中返回 None)。""" + platform = match_platform(url) + return platform if platform and platform in self.auto_link else None + + def to_dict(self) -> dict[str, Any]: + return { + "auto": self.auto, + "auto_link": list(self.auto_link), + "ban_link": list(self.ban_link), + "plan": self.plan, + "upload_public": self.upload_public, + "send_link": self.send_link, + "upload_group_file": self.upload_group_file, + "group_file_platforms": list(self.group_file_platforms), + } + + @classmethod + def from_dict(cls, raw: Any) -> Policy: + """脏数据归一:非法 plan 回退 C,平台别名转规范标签,多余键丢弃。""" + if not isinstance(raw, dict): + return cls() + plan = str(raw.get("plan", DEFAULT_PLAN)).strip().upper() + return cls( + auto=bool(raw.get("auto", False)), + auto_link=normalize_platforms(raw.get("auto_link")), + ban_link=normalize_platforms(raw.get("ban_link")), + plan=plan if plan in PLANS else DEFAULT_PLAN, + upload_public=bool(raw.get("upload_public", False)), + send_link=bool(raw.get("send_link", False)), + upload_group_file=bool(raw.get("upload_group_file", False)), + group_file_platforms=normalize_platforms(raw.get("group_file_platforms")), + ) + + +# ───────────────────────── 迁移 ───────────────────────── + + +def _v1_to_v2(raw: dict) -> dict: + """旧格式(平铺数组)→ group-centric。""" + white = raw.get("WHITE_LIST", []) + auto_list = raw.get("AUTO_ANALYSIS", []) + pa = raw.get("PLANA", []) + pb = raw.get("PLANB", []) + + groups: dict[str, dict] = {} + for gid in white: + entry: dict = {} + if gid in auto_list: + entry["auto"] = True + if gid in pa: + entry["plan"] = "A" + elif gid in pb: + entry["plan"] = "B" + groups[str(gid)] = entry + + return {"groups": groups, "blacklist": raw.get("BLACK_LIST", [])} + + +def migrate(raw: Any) -> tuple[dict, bool]: + """v1 / v2 → v3,返回 (数据, 是否发生迁移)。 + + v2 → v3 的关键一步:v2 的 `plan=B` 隐含"上传公网",解耦后给它显式补上 + `upload_public`,保证已有群行为不变。 + """ + if not isinstance(raw, dict): + return _empty_data(), False + + changed = False + if "groups" not in raw: + raw = _v1_to_v2(raw) + changed = True + + groups: dict[str, dict] = {} + for gid, entry in (raw.get("groups") or {}).items(): + if not isinstance(entry, dict): + changed = True + continue + policy = Policy.from_dict(entry) + if policy.plan == "B" and "upload_public" not in entry: + policy.upload_public = True + if entry != policy.to_dict(): + changed = True + groups[str(gid)] = policy.to_dict() + + if "default" not in raw: + changed = True + + data = { + "groups": groups, + "default": Policy.from_dict(raw.get("default")).to_dict(), + "blacklist": [str(x) for x in (raw.get("blacklist") or [])], + } + return data, changed + + +def _empty_data() -> dict: + return { + "groups": {}, + "default": Policy().to_dict(), + "blacklist": [], + } + + +# ─────────────────────────── 存储 ─────────────────────────── + + +class PolicyStore: + """list.json 读写:读全内存,写加锁 + 原子替换。""" + + def __init__(self, path: Path) -> None: + self.path = Path(path) + self._groups: dict[str, Policy] = {} + self._default = Policy() + self._blacklist: list[str] = [] + self._loaded = False + self._lock = asyncio.Lock() + + # ── 读 ────────────────────────────────────────────── + + def load(self) -> None: + """首次调用读盘 + 迁移(幂等,之后不再读盘)。""" + if self._loaded: + return + + raw: Any = {} + if self.path.exists(): + try: + raw = json.loads(self.path.read_text(encoding="utf-8")) + except Exception: + logger.exception(f"读取 {self.path.name} 失败,改用空配置") + + data, migrated = migrate(raw) + self._apply(data) + self._loaded = True + + if migrated: + self._write_sync(self._snapshot(data)) + logger.info(f"{self.path.name} 已迁移为 v3 格式(群 {len(self._groups)} 个)") + + def _apply(self, data: dict) -> None: + self._groups = { + str(gid): Policy.from_dict(entry) + for gid, entry in (data.get("groups") or {}).items() + } + self._default = Policy.from_dict(data.get("default")) + self._blacklist = [str(x) for x in (data.get("blacklist") or [])] + + def is_whitelisted(self, group_id: Any) -> bool: + """群是否在白名单里(白名单即 groups 的键,仍是准入门槛)。""" + self.load() + return str(group_id) in self._groups + + def get(self, group_id: Any) -> Policy: + """群策略:未配置的群回落到 default 节。""" + self.load() + return self._groups.get(str(group_id), self._default) + + def all_groups(self) -> dict[str, Policy]: + self.load() + return dict(self._groups) + + def default_policy(self) -> Policy: + self.load() + return self._default + + def blacklist(self) -> list[str]: + self.load() + return list(self._blacklist) + + def is_blacklisted(self, user_id: Any) -> bool: + self.load() + return str(user_id) in self._blacklist + + # ── 写 ────────────────────────────────────────────── + + def _snapshot(self, data: dict | None = None) -> dict: + """在事件循环线程内构造完整快照,写线程不再触碰共享状态。""" + if data is not None: + return data + return { + "groups": {gid: p.to_dict() for gid, p in self._groups.items()}, + "default": self._default.to_dict(), + "blacklist": list(self._blacklist), + } + + def _write_sync(self, data: dict) -> None: + tmp = self.path.with_name(self.path.name + ".tmp") + tmp.parent.mkdir(parents=True, exist_ok=True) + tmp.write_text( + json.dumps(data, ensure_ascii=False, indent=2), encoding="utf-8" + ) + os.replace(tmp, self.path) + + async def _save(self) -> None: + snapshot = self._snapshot() + async with self._lock: + await asyncio.to_thread(self._write_sync, snapshot) + + async def update_group( + self, group_id: Any, *, create: bool = False, **changes: Any + ) -> Policy | None: + """更新群策略的部分字段;群不存在且 create=False 时返回 None。 + + changes 传最终值(列表类字段由调用方算好新列表)。 + """ + self.load() + gid = str(group_id) + current = self._groups.get(gid) + if current is None: + if not create: + return None + current = Policy() + + valid = {k: v for k, v in changes.items() if k in POLICY_FIELDS} + policy = replace(current, **valid) if valid else current + policy = Policy.from_dict(policy.to_dict()) # 归一化(平台别名/非法 plan) + self._groups[gid] = policy + await self._save() + return policy + + async def set_group(self, group_id: Any, policy: Policy) -> Policy: + """整体覆盖群策略(Web 编辑用)。""" + self.load() + gid = str(group_id) + policy = Policy.from_dict(policy.to_dict()) + self._groups[gid] = policy + await self._save() + return policy + + async def remove_group(self, group_id: Any) -> bool: + self.load() + gid = str(group_id) + if gid not in self._groups: + return False + del self._groups[gid] + await self._save() + return True + + async def set_default(self, policy: Policy) -> Policy: + self.load() + self._default = Policy.from_dict(policy.to_dict()) + await self._save() + return self._default + + async def add_blacklist(self, user_id: Any) -> bool: + self.load() + uid = str(user_id) + if uid in self._blacklist: + return False + self._blacklist.append(uid) + await self._save() + return True + + async def set_blacklist(self, values: Iterable[Any]) -> list[str]: + """整体替换黑名单(去重保序,一次落盘)。""" + self.load() + cleaned: list[str] = [] + for value in values: + uid = str(value).strip() + if uid and uid not in cleaned: + cleaned.append(uid) + self._blacklist = cleaned + await self._save() + return list(self._blacklist) + + async def remove_blacklist(self, user_id: Any) -> bool: + self.load() + uid = str(user_id) + if uid not in self._blacklist: + return False + self._blacklist.remove(uid) + await self._save() + return True + + +#: 单例(插件自己的 data/ 目录,与历史 list.json 同路径,原地迁移) +STORE = PolicyStore(Path(__file__).resolve().parent / "data" / "list.json") diff --git a/hexi/plugins/nonebot_plugin_video_analysis/services/fetchers/bilibili_content.py b/hexi/plugins/nonebot_plugin_video_analysis/services/fetchers/bilibili_content.py index ec79342..34f9782 100644 --- a/hexi/plugins/nonebot_plugin_video_analysis/services/fetchers/bilibili_content.py +++ b/hexi/plugins/nonebot_plugin_video_analysis/services/fetchers/bilibili_content.py @@ -14,15 +14,19 @@ """ import re -import tempfile -from datetime import datetime from pathlib import Path from typing import Optional, Union from nonebot import logger from ...models import ContentFetchError -from ...utils import get_data_dir, parse_netscape_cookies, slugify +from ...utils import ( + build_author_dir, + build_work_stem, + get_data_dir, + get_temp_root, + parse_netscape_cookies, +) DATA_DIR = get_data_dir() @@ -81,10 +85,10 @@ async def _parse(url: str): # 4. 文章动态 → 转 opus if await dynamic.is_article(): - return await _parse_opus(dynamic.turn_to_opus(), "文章") + return await _parse_opus(dynamic.turn_to_opus(), "文章", url) info = await dynamic.get_info() - return await _parse_dynamic_info(info) + return await _parse_dynamic_info(info, url) async def _parse_article(read_id: int) -> tuple[str, Union[Path, list[Path]]]: @@ -94,10 +98,12 @@ async def _parse_article(read_id: int) -> tuple[str, Union[Path, list[Path]]]: # 文章接口对匿名请求风控更严(-509),必须带凭证 article = Article(read_id, _build_credential()) opus = await article.turn_to_opus() - return await _parse_opus(opus, "文章") + return await _parse_opus(opus, "文章", f"cv{read_id}") -async def _parse_opus(opus, kind: str) -> tuple[str, Union[Path, list[Path]]]: +async def _parse_opus( + opus, kind: str, source: str = "" +) -> tuple[str, Union[Path, list[Path]]]: """图文动态/专栏解析(opus 接口返回 dict,直接访问)""" info = await opus.get_info() item = info.get("item") or {} @@ -107,10 +113,12 @@ async def _parse_opus(opus, kind: str) -> tuple[str, Union[Path, list[Path]]]: images: list[str] = [] texts: list[str] = [] author = "" + author_id = "" for module in item.get("modules") or []: if module.get("module_type") == "MODULE_TYPE_AUTHOR": author_info = module.get("module_author") or {} author = author_info.get("name", "") + author_id = str(author_info.get("mid") or "") elif module.get("module_type") == "MODULE_TYPE_CONTENT": content = module.get("module_content") or {} for para in content.get("paragraphs") or []: @@ -127,16 +135,20 @@ async def _parse_opus(opus, kind: str) -> tuple[str, Union[Path, list[Path]]]: if not images: return text or f"B站{kind}", [] - file_name = _build_file_name(author, text or f"B站{kind}", kind) - file_paths = await _download_images(images, file_name) + rel_stem = _build_rel_stem(author, author_id, text or f"B站{kind}", source) + file_paths = await _download_images(images, rel_stem) return text, file_paths -async def _parse_dynamic_info(info: dict) -> tuple[str, Union[Path, list[Path]]]: +async def _parse_dynamic_info( + info: dict, source: str = "" +) -> tuple[str, Union[Path, list[Path]]]: """动态解析(图文 / 视频 / 纯文字)""" item = info.get("item") or {} modules = item.get("modules") or {} - author = ((modules.get("module_author") or {}).get("name")) or "B站用户" + module_author = modules.get("module_author") or {} + author = module_author.get("name") or "B站用户" + author_id = str(module_author.get("mid") or "") module_dynamic = modules.get("module_dynamic") or {} major = module_dynamic.get("major") or {} major_type = major.get("type", "") @@ -151,7 +163,13 @@ async def _parse_dynamic_info(info: dict) -> tuple[str, Union[Path, list[Path]]] title = archive.get("title") or desc or "B站视频动态" logger.info(f"B站视频动态: bvid={bvid} 标题={title[:40]}") - video_path = await download_video(f"https://www.bilibili.com/video/{bvid}") + # 动态数据里已有 up 的昵称/mid,传下去才能和同一位 up 的图文 + # 落在同一个作者目录(否则要赌 yt-dlp 返回的 id 对得上) + video_path, _ = await download_video( + f"https://www.bilibili.com/video/{bvid}", + author=author, + author_id=author_id, + ) if video_path: return title, video_path raise ContentFetchError(f"视频动态下载失败: {bvid}") @@ -173,8 +191,8 @@ async def _parse_dynamic_info(info: dict) -> tuple[str, Union[Path, list[Path]]] images = [u for u in images if u] if images: - file_name = _build_file_name(author, title, "动态") - file_paths = await _download_images(images, file_name) + rel_stem = _build_rel_stem(author, author_id, title, source) + file_paths = await _download_images(images, rel_stem) logger.info(f"B站图文动态: 作者={author}, 标题={title[:40]}, 图片={len(images)} 张") return title, file_paths @@ -197,14 +215,17 @@ def _extract_text(nodes: list) -> str: return "".join(parts) -def _build_file_name(author: str, title: str, kind: str) -> str: - """构建文件名 stem: {作者}_{标题}_{类型}_{时间}""" - slug_author = slugify(author) - slug_title = slugify(title or "", max_length=15) - if not slug_title: - slug_title = datetime.now().strftime("%H%M%S") - time_suffix = datetime.now().strftime("%H%M%S") - return f"{slug_author}_{slug_title}_{kind}_{time_suffix}" +def _build_rel_stem( + author: str, author_id: str, title: str, source: str = "" +) -> str: + """相对平台根的路径词干:`{作者}_{mid}/{作品名}` + + 昵称/mid 都拿不到时用 source(动态链接)当来源码(见 utils.build_author_dir)。 + """ + return ( + f"{build_author_dir(author, author_id, source=source)}" + f"/{build_work_stem(title)}" + ) def _build_credential(): @@ -227,20 +248,21 @@ def _build_credential(): ) -async def _download_images(image_urls: list[str], file_name: str) -> list[Path]: - """并发下载图片(复用抖音图文的下载流程)""" - import httpx +async def _download_images(image_urls: list[str], rel_stem: str) -> list[Path]: + """并发下载图片(复用抖音图文的下载流程) + + 落 hexi/data/temp/bilibili(原先落系统 temp,cleanup 扫不到、永不清理)。 + """ from .douyin_api import _process_note_with_parsed - tmp_root = Path(tempfile.gettempdir()) / "bilibili" - tmp_root.mkdir(parents=True, exist_ok=True) + tmp_root = get_temp_root("bilibili") headers = { "Referer": BILI_REFERER, "User-Agent": BILI_UA, } return await _process_note_with_parsed( - [[u] for u in image_urls], None, tmp_root, file_name, headers + [[u] for u in image_urls], None, tmp_root, rel_stem, headers ) diff --git a/hexi/plugins/nonebot_plugin_video_analysis/services/fetchers/douyin_api.py b/hexi/plugins/nonebot_plugin_video_analysis/services/fetchers/douyin_api.py index 3547c0a..086f020 100644 --- a/hexi/plugins/nonebot_plugin_video_analysis/services/fetchers/douyin_api.py +++ b/hexi/plugins/nonebot_plugin_video_analysis/services/fetchers/douyin_api.py @@ -10,11 +10,9 @@ from nonebot import logger from playwright.async_api import async_playwright from ...models import DouyinFetchError -from ...utils import ensure_unique_path, get_temp_root +from ...utils import get_temp_root, unique_media_path from .douyin_parser import ( - ParsedDouyinContent, extract_trailing_digits, - is_animated_note, parse_animated_note_videos, parse_douyin_response, parse_note_images, @@ -194,9 +192,9 @@ async def fetch_douyin_content( if parsed.media_type == "视频": content = await _process_video( - api_response, tmp_root, parsed.file_name, aweme_id, headers + api_response, tmp_root, parsed.rel_stem, aweme_id, headers ) - return parsed.file_name, content + return parsed.raw_title, content elif parsed.media_type == "图片": # 先解析 images 列表,区分纯动图和图文/图+视频 @@ -209,15 +207,15 @@ async def fetch_douyin_content( images_urls, video_url, tmp_root, - parsed.file_name, + parsed.rel_stem, headers, ) else: # 纯动图(所有项都是视频) content = await _process_animated_note( - api_response, tmp_root, parsed.file_name, headers + api_response, tmp_root, parsed.rel_stem, headers ) - return parsed.file_name, content + return parsed.raw_title, content return None, None @@ -228,11 +226,14 @@ async def fetch_douyin_content( async def _process_video( api_response: dict, tmp_root: Path, - file_name: str, + rel_stem: str, aweme_id: str, headers: Dict[str, str], ) -> Path: - """处理视频内容,返回本地文件路径""" + """处理视频内容,返回本地文件路径 + + rel_stem 是相对平台根的路径词干 `{作者目录}/{作品名}`(见 ParsedDouyinContent.rel_stem)。 + """ groups = parse_video_urls(api_response) best_group = None @@ -251,7 +252,7 @@ async def _process_video( best = max(full, key=lambda x: x["br"]) logger.info(f"选择码率: {best['br']} - {best['url'][:60]}...") - output_path = ensure_unique_path(tmp_root / f"{file_name}.mp4") + output_path = unique_media_path(tmp_root / f"{rel_stem}.mp4") async with httpx.AsyncClient(headers=headers) as client: async with client.stream("GET", best["url"]) as resp: resp.raise_for_status() @@ -271,9 +272,10 @@ async def _process_video( logger.info(f"选择视频码率: {video['br']}") logger.info(f"选择音频码率: {audio['br']}") - video_path = tmp_root / f"{file_name}_v.mp4" - audio_path = tmp_root / f"{file_name}_a.mp4" - output_path = ensure_unique_path(tmp_root / f"{file_name}.mp4") + # 先定下产物名(顺带建好作者目录),分轨中间文件与产物同目录 + output_path = unique_media_path(tmp_root / f"{rel_stem}.mp4") + video_path = output_path.parent / f"{output_path.stem}_v.mp4" + audio_path = output_path.parent / f"{output_path.stem}_a.mp4" async with httpx.AsyncClient(headers=headers) as client: logger.info("开始下载视频...") @@ -291,7 +293,9 @@ async def _process_video( f.write(chunk) logger.info("合并视频和音频...") - merge_video_audio(video_path, audio_path, output_path) + # ffmpeg 是同步子进程,直接 await 不了:不丢线程池会卡住整个事件循环 + # (合并期间 Web 轮询、群消息全都停摆) + await asyncio.to_thread(merge_video_audio, video_path, audio_path, output_path) video_path.unlink() audio_path.unlink() @@ -306,11 +310,14 @@ async def _process_note_with_parsed( images_urls: List[List[str]], video_url: Optional[str], tmp_root: Path, - file_name: str, + rel_stem: str, headers: Dict[str, str], ) -> List[Path]: - """根据已解析的图片/视频 URL 列表,并行下载""" - note_dir = ensure_unique_path(tmp_root / file_name) + """根据已解析的图片/视频 URL 列表,并行下载 + + 一个作品一个目录:`{平台根}/{作者目录}/{作品名}[_{短码}]/001.jpg…` + """ + note_dir = unique_media_path(tmp_root / rel_stem) note_dir.mkdir(parents=True, exist_ok=True) logger.info(f"图文保存目录: {note_dir}") @@ -353,7 +360,7 @@ async def _process_note( api_response: dict, api_response_favorite: dict, tmp_root: Path, - file_name: str, + rel_stem: str, aweme_id: str, headers: Dict[str, str], ) -> List[Path]: @@ -367,7 +374,7 @@ async def _process_note( raise DouyinFetchError("未找到图文链接") return await _process_note_with_parsed( - images_urls, video_url, tmp_root, file_name, headers + images_urls, video_url, tmp_root, rel_stem, headers ) @@ -377,14 +384,14 @@ async def _process_note( async def _process_animated_note( api_response: dict, tmp_root: Path, - file_name: str, + rel_stem: str, headers: Dict[str, str], ) -> List[Path]: """处理动图内容(media_type=42),并行下载所有无声 mp4 视频""" video_urls = parse_animated_note_videos(api_response) logger.info(f"解析到的动图视频链接: {video_urls}") - note_dir = ensure_unique_path(tmp_root / file_name) + note_dir = unique_media_path(tmp_root / rel_stem) note_dir.mkdir(parents=True, exist_ok=True) logger.info(f"动图保存目录: {note_dir}") diff --git a/hexi/plugins/nonebot_plugin_video_analysis/services/fetchers/douyin_parser.py b/hexi/plugins/nonebot_plugin_video_analysis/services/fetchers/douyin_parser.py index c8c0758..e5594c7 100644 --- a/hexi/plugins/nonebot_plugin_video_analysis/services/fetchers/douyin_parser.py +++ b/hexi/plugins/nonebot_plugin_video_analysis/services/fetchers/douyin_parser.py @@ -4,13 +4,12 @@ import json import re import urllib.parse from dataclasses import dataclass -from datetime import datetime from typing import Dict, List, Optional from nonebot import logger from ...models import DouyinFetchError -from ...utils import slugify +from ...utils import build_author_dir, build_work_stem # ============================= 数据结构 ============================= @@ -22,7 +21,13 @@ class ParsedDouyinContent: raw_title: str raw_nickname: str media_type: str # "视频" | "图片" - file_name: str # 构建好的文件名 stem + file_name: str # 作品名 stem(落盘文件名/多图子目录名) + author_dir: str = "" # 作者目录 `{昵称}_{uid}`(落盘与 S3 key 的首层) + + @property + def rel_stem(self) -> str: + """相对平台根目录的路径词干:`{作者目录}/{作品名}`""" + return f"{self.author_dir}/{self.file_name}" if self.author_dir else self.file_name # ============================= URL 工具 ============================= @@ -109,6 +114,29 @@ def extract_author_nickname(api_response: dict) -> str: return nickname +def extract_author_id(api_response: dict) -> str: + """从 API 响应提取作者稳定 id,优先级: uid > unique_id(抖音号) > sec_uid + + SSR 路径的 `aweme_list[0].author.uid` 与 API 路径的 + `aweme_detail.author.uid` 走同一套取值;都拿不到返回空串, + 作者目录退化成只用昵称(见 utils.build_author_dir)。 + """ + aweme_detail = api_response.get("aweme_detail") or {} + author = aweme_detail.get("author") or {} + if not author: + aweme_list = api_response.get("aweme_list") or [] + if aweme_list and isinstance(aweme_list, list): + author = (aweme_list[0] or {}).get("author") or {} + + for key in ("uid", "unique_id", "sec_uid"): + value = str(author.get(key) or "").strip() + if value and value != "0": + logger.info(f"RAW作者id({key}):{value}") + return value + logger.info("未获取到作者 id,作者目录只用昵称") + return "" + + def detect_media_type(referer_url: str | None) -> str | None: """根据页面 URL 检测媒体类型(视频/图片)""" if referer_url is None: @@ -326,36 +354,6 @@ def parse_ssr_page(html: str) -> Optional[dict]: return {"aweme_list": [item]} -# ============================= 文件名构建 ============================= - - -def build_file_name( - raw_title: str, - raw_nickname: str, - media_type: str, -) -> str: - """ - 构建文件名 stem,格式: {作者}_{标题}_{类型}_{时间戳} - - 昵称不限长,标题最多 15 字符(slugify 后),末尾 HHMMSS 防覆盖。 - """ - slug_nickname = slugify(raw_nickname) - - if raw_title: - slug_title = slugify(raw_title, max_length=15) - else: - slug_title = "" - - if not slug_title: - slug_title = datetime.now().strftime("%H%M%S") - logger.info(f"标题为空,使用短时间戳: {slug_title}") - - slug_type = slugify(media_type) - time_suffix = datetime.now().strftime("%H%M%S") - - return f"{slug_nickname}_{slug_title}_{slug_type}_{time_suffix}" - - # ============================= 动图检测 ============================= @@ -410,11 +408,21 @@ def parse_douyin_response( raw_title = extract_title_from_api(api_response) raw_nickname = extract_author_nickname(api_response) - file_name = build_file_name(raw_title, raw_nickname, media_type) + raw_author_id = extract_author_id(api_response) + # 昵称/uid 都拿不到时的来源码:优先作品链接,其次响应里的作品 id + aweme_id = str( + (api_response.get("aweme_detail") or {}).get("aweme_id") + or ((api_response.get("aweme_list") or [{}])[0] or {}).get("aweme_id") + or "" + ) + author_dir = build_author_dir( + raw_nickname, raw_author_id, source=referer_url or aweme_id + ) + file_name = build_work_stem(raw_title) logger.info( - f"内容标题: {raw_title}, 作者: {raw_nickname}, " - f"类型: {media_type}, 文件名: {file_name}" + f"内容标题: {raw_title}, 作者: {raw_nickname}({raw_author_id}), " + f"类型: {media_type}, 落盘路径: {author_dir}/{file_name}" ) return ParsedDouyinContent( @@ -422,4 +430,5 @@ def parse_douyin_response( raw_nickname=raw_nickname, media_type=media_type, file_name=file_name, + author_dir=author_dir, ) diff --git a/hexi/plugins/nonebot_plugin_video_analysis/services/fetchers/douyin_ssr.py b/hexi/plugins/nonebot_plugin_video_analysis/services/fetchers/douyin_ssr.py index e805852..61f88e5 100644 --- a/hexi/plugins/nonebot_plugin_video_analysis/services/fetchers/douyin_ssr.py +++ b/hexi/plugins/nonebot_plugin_video_analysis/services/fetchers/douyin_ssr.py @@ -85,13 +85,13 @@ async def fetch_douyin_note_ssr( images_urls, video_url = parse_note_images(api_response, None, vid) if images_urls: file_paths = await _process_note_with_parsed( - images_urls, video_url, tmp_root, parsed.file_name, headers + images_urls, video_url, tmp_root, parsed.rel_stem, headers ) else: file_paths = await _process_animated_note( - api_response, tmp_root, parsed.file_name, headers + api_response, tmp_root, parsed.rel_stem, headers ) - return parsed.file_name, file_paths + return parsed.raw_title, file_paths async def _fetch_ssr_api_response( diff --git a/hexi/plugins/nonebot_plugin_video_analysis/services/fetchers/rednote_content.py b/hexi/plugins/nonebot_plugin_video_analysis/services/fetchers/rednote_content.py index 9f33b1c..49028c1 100644 --- a/hexi/plugins/nonebot_plugin_video_analysis/services/fetchers/rednote_content.py +++ b/hexi/plugins/nonebot_plugin_video_analysis/services/fetchers/rednote_content.py @@ -19,16 +19,21 @@ import asyncio import json import re -import tempfile import urllib.parse -from datetime import datetime from pathlib import Path from typing import Optional, Union from nonebot import logger from ...models import ContentFetchError -from ...utils import get_data_dir, get_temp_root, parse_netscape_cookies, slugify +from ...utils import ( + build_author_dir, + build_work_stem, + get_data_dir, + get_temp_root, + parse_netscape_cookies, + unique_media_path, +) DATA_DIR = get_data_dir() @@ -214,6 +219,12 @@ async def _build_result(note: dict) -> tuple[str, Union[Path, list[Path]]]: desc = note.get("desc") or "" nickname = ((note.get("user") or {}).get("nickname")) or "小红书用户" text = title or desc or "小红书笔记" + rel_stem = _build_rel_stem( + nickname, + _extract_author_id(note), + text, + source=str(note.get("noteId") or note.get("id") or ""), + ) # 1. 视频笔记 → 无水印原片优先 if note.get("type") == "video" and note.get("video"): @@ -223,8 +234,7 @@ async def _build_result(note: dict) -> tuple[str, Union[Path, list[Path]]]: if okey: video_url = f"https://sns-video-bd.xhscdn.com/{okey}" logger.info(f"小红书视频: 无水印原片 originVideoKey={okey[:30]}...") - file_name = _build_file_name(nickname, text, "视频") - video_path = await _download_video(video_url, file_name) + video_path = await _download_video(video_url, rel_stem) return text, video_path # 1b. 无 originVideoKey(国内站数据)→ 从 stream 分组选无水印原片 @@ -251,8 +261,7 @@ async def _build_result(note: dict) -> tuple[str, Union[Path, list[Path]]]: f"{best.get('width')}x{best.get('height')} {best.get('fps')}fps " f"size={best.get('size')} duration={duration}ms" ) - file_name = _build_file_name(nickname, text, "视频") - video_path = await _download_video(video_url, file_name) + video_path = await _download_video(video_url, rel_stem) return text, video_path raise ContentFetchError("小红书视频流解析失败") @@ -266,25 +275,36 @@ async def _build_result(note: dict) -> tuple[str, Union[Path, list[Path]]]: logger.info(f"小红书文字笔记: {text[:30]}") return text, [] - file_name = _build_file_name(nickname, text, "笔记") - file_paths = await _download_images(images, file_name) + file_paths = await _download_images(images, rel_stem) logger.info(f"小红书图文笔记: 作者={nickname}, 图片={len(images)} 张") return text, file_paths -def _build_file_name(nickname: str, title: str, kind: str) -> str: - """构建文件名 stem: {作者}_{标题}_{类型}_{时间}""" - slug_nickname = slugify(nickname) - slug_title = slugify(title or "", max_length=15) - if not slug_title: - slug_title = datetime.now().strftime("%H%M%S") - time_suffix = datetime.now().strftime("%H%M%S") - return f"{slug_nickname}_{slug_title}_{kind}_{time_suffix}" +def _extract_author_id(note: dict) -> str: + """小红书作者稳定 id(页面数据字段未实测,逐个兜底;取不到返回空串)""" + user = note.get("user") or {} + for key in ("userId", "user_id", "id"): + value = user.get(key) + if isinstance(value, (str, int)) and str(value).strip() not in ("", "0"): + return str(value).strip() + return "" -async def _download_images(image_urls: list[str], file_name: str) -> list[Path]: +def _build_rel_stem( + nickname: str, author_id: str, title: str, source: str = "" +) -> str: + """相对平台根的路径词干:`{作者}_{userId}/{作品名}` + + 昵称/作者 id 都拿不到时用 source(笔记 id)当来源码(见 utils.build_author_dir)。 + """ + return ( + f"{build_author_dir(nickname, author_id, source=source)}" + f"/{build_work_stem(title)}" + ) + + +async def _download_images(image_urls: list[str], rel_stem: str) -> list[Path]: """并发下载图片(复用抖音图文的下载流程)""" - import httpx from .douyin_api import _process_note_with_parsed @@ -294,11 +314,11 @@ async def _download_images(image_urls: list[str], file_name: str) -> list[Path]: "User-Agent": REDNOTE_UA, } return await _process_note_with_parsed( - [[u] for u in image_urls], None, tmp_root, file_name, headers + [[u] for u in image_urls], None, tmp_root, rel_stem, headers ) -async def _download_video(video_url: str, file_name: str) -> Path: +async def _download_video(video_url: str, rel_stem: str) -> Path: """流式下载视频 注意:sns-video-bd(无水印原片)不带 Referer 或带 xiaohongshu.com @@ -307,7 +327,7 @@ async def _download_video(video_url: str, file_name: str) -> Path: import httpx tmp_root = get_temp_root("xiaohongshu") - output_path = tmp_root / f"{file_name}.mp4" + output_path = unique_media_path(tmp_root / f"{rel_stem}.mp4") headers = {"User-Agent": REDNOTE_UA} async with httpx.AsyncClient(headers=headers, timeout=300) as client: async with client.stream("GET", video_url) as resp: diff --git a/hexi/plugins/nonebot_plugin_video_analysis/services/fetchers/video_downloader.py b/hexi/plugins/nonebot_plugin_video_analysis/services/fetchers/video_downloader.py index 0ee3231..cae3a90 100644 --- a/hexi/plugins/nonebot_plugin_video_analysis/services/fetchers/video_downloader.py +++ b/hexi/plugins/nonebot_plugin_video_analysis/services/fetchers/video_downloader.py @@ -3,9 +3,9 @@ import asyncio import os import re +import shutil import sys import tempfile -from datetime import datetime from pathlib import Path from typing import Optional @@ -14,7 +14,14 @@ from nonebot import logger from yt_dlp import YoutubeDL from yt_dlp.utils import DownloadError -from ...utils import get_data_dir, get_temp_root, slugify, ensure_unique_path +from ...utils import ( + build_author_dir, + build_work_stem, + get_data_dir, + get_temp_root, + slugify, + unique_media_path, +) def detect_platform(url: str) -> str: @@ -41,6 +48,16 @@ def extract_uploader(info: dict) -> Optional[str]: ) +def extract_uploader_id(info: dict) -> str: + """作者稳定 id:channel_id > uploader_id(@handle / B站 mid),拿不到返回空串 + + 不用 `id`(那是作品 id,会把同一作者的作品拆到不同目录)。 + """ + if not info: + return "" + return str(info.get("channel_id") or info.get("uploader_id") or "").strip() + + def get_ffmpeg_path() -> str: scripts_dir = os.path.dirname(sys.executable) ffmpeg_path = os.path.join(scripts_dir, "ffmpeg.exe") @@ -78,19 +95,36 @@ async def _retry_download( logger.error( f"yt-dlp 重试 {max_retries} 次后仍失败: {str(e)[:120]}" ) - except Exception as e: + except Exception: # 非 DownloadError(如 OSError)不重试,直接抛出 raise raise last_error # type: ignore[misc] -async def download_video(url: str) -> Optional[Path]: - """下载视频,支持直链和 yt-dlp""" +async def download_video( + url: str, + *, + author: Optional[str] = None, + author_id: Optional[str] = None, +) -> tuple[Optional[Path], str]: + """下载视频,支持直链和 yt-dlp + + author / author_id 可由调用方覆盖(如 B站 视频动态已从动态数据里拿到 mid, + 传进来才能和同一位 up 的图文落在同一个作者目录)。 + + Returns: + (本地文件, 作品标题) — 失败时 (None, "")。 + 落盘位置:`temp/{平台}/{作者}_{作者id}/{作品名}[_{短码}].ext` + (直链拿不到作者信息,统一进 `未知作者/`)。 + """ + platform = detect_platform(url) + temp_root = get_temp_root(platform) # ---------- 1. 直链探测 ---------- + # 不含 m3u8:HLS 播放列表直下只会得到一个文本文件,交给 yt-dlp 处理 direct_media_ext = re.search( - r"\.(mp4|m3u8|ts|webm|mov|flv)(?:$|\?)", url, re.IGNORECASE + r"\.(mp4|ts|webm|mov|flv)(?:$|\?)", url, re.IGNORECASE ) is_direct = bool(direct_media_ext) @@ -105,39 +139,35 @@ async def download_video(url: str) -> Optional[Path]: is_direct = False if is_direct: - temp_dir = tempfile.mkdtemp(prefix="direct_ytcache_", dir=get_temp_root("ytcache")) ext = "mp4" m = re.search(r"\.([a-zA-Z0-9]{2,5})(?:$|\?)", url) if m and len(m.group(1)) <= 5: ext = m.group(1) url_stem = Path(url.split("?")[0]).stem or "video" - slug_stem = slugify(url_stem, max_length=15) - if not slug_stem: - slug_stem = datetime.now().strftime("%H%M%S") - time_suffix = datetime.now().strftime("%H%M%S") - new_name = f"{slug_stem}_视频_{time_suffix}.{ext}" - filename = os.path.join(temp_dir, new_name) + slug_stem = slugify(url_stem, max_length=15) or "视频" + # 直链拿不到作者信息 → `未知作者_{来源短码}`(同一链接稳定、不同链接不撞) + author_dir = build_author_dir(author, author_id, source=url) + final_path = unique_media_path(temp_root / author_dir / f"{slug_stem}.{ext}") try: async with AsyncClient(follow_redirects=True, timeout=300) as client: async with client.stream("GET", url) as resp: resp.raise_for_status() - with open(filename, "wb") as fh: + with open(final_path, "wb") as fh: async for chunk in resp.aiter_bytes(chunk_size=8192): fh.write(chunk) - final_path = ensure_unique_path(Path(filename)) logger.info(f"直接下载完成: {final_path}") - return final_path + return final_path, "" except Exception: logger.exception("直接下载失败,回退 yt-dlp") - if os.path.exists(filename): - os.remove(filename) + final_path.unlink(missing_ok=True) # ---------- 2. yt-dlp 下载 ---------- - platform = detect_platform(url) - temp_dir = tempfile.mkdtemp(prefix="ytcache_", dir=get_temp_root("ytcache")) + # 先下到 scratch 目录(outtmpl 必须在拿到 info 之前给定),拿到 info 后再 + # 按作者归位到 temp/{平台}/{作者}_{作者id}/ + temp_dir = tempfile.mkdtemp(prefix="_dl_", dir=temp_root) output_path = os.path.join(temp_dir, "%(title).80s.%(ext)s") base_opts = { @@ -188,7 +218,10 @@ async def download_video(url: str) -> Optional[Path]: "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,image/webp,*/*;q=0.8", "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8", } - base_opts["extractor_args"] = {"twitter": {"api": ["syndication"]}} + # 不要强制 twitter:api=syndication:该端点为未登录视角,对敏感/受限推文 + # 只返回 tombstone(无 mediaDetails),且会无条件覆盖已登录 GraphQL 的结果, + # 表现为 "No video could be found in this tweet"。默认走 GraphQL + cookies, + # 遇 429 yt-dlp 会自行回退 syndication。 elif platform == "youtube": base_opts["http_headers"] = { "User-Agent": ua, @@ -206,54 +239,53 @@ async def download_video(url: str) -> Optional[Path]: info = await _retry_download(loop, url, base_opts) except Exception: logger.exception("yt-dlp 下载失败") - # YouTube: cookies 可能触发 bot 检测导致只返回图片无视频格式 - # 回退无 cookie 模式重试 if platform == "youtube" and "cookiefile" in base_opts: + # YouTube: cookies 可能触发 bot 检测导致只返回图片无视频格式 + # 回退无 cookie 模式重试 logger.info("YouTube 回退无 cookies 模式重试...") base_opts.pop("cookiefile", None) base_opts.pop("http_headers", None) - # 清理失败残留 - for f in Path(temp_dir).glob("*.*"): - try: - f.unlink() - except Exception: - pass - try: - info = await _retry_download(loop, url, base_opts, max_retries=2) - except Exception: - logger.exception("yt-dlp 无 cookies 重试也失败") - return None + elif platform == "twitter": + # X 登录态失效(auth_token 过期)时 GraphQL 会直接拒绝请求; + # 退回未登录的 syndication 端点,公开推文仍可下载(敏感推文会失败) + logger.info("Twitter 回退 syndication 端点重试...") + base_opts["extractor_args"] = {"twitter": {"api": ["syndication"]}} else: - return None + return None, "" + + # 清理失败残留 + for f in Path(temp_dir).glob("*.*"): + try: + f.unlink() + except Exception: + pass + try: + info = await _retry_download(loop, url, base_opts, max_retries=2) + except Exception: + logger.exception("yt-dlp 回退重试也失败") + return None, "" files = list(Path(temp_dir).glob("*.*")) if not files: - return None + return None, "" original_file = files[0] - # 构建新文件名 - uploader = extract_uploader(info or {}) + # 归位到作者目录:{作者}_{作者id}/{作品名}[_{短码}].ext + uploader = author or extract_uploader(info or {}) + uploader_id = author_id or extract_uploader_id(info or {}) title = ((info or {}).get("title") or "").strip() - slug_title = slugify(title, max_length=15) if title else "" - if not slug_title: - slug_title = datetime.now().strftime("%H%M%S") - - time_suffix = datetime.now().strftime("%H%M%S") - if uploader: - slug_uploader = slugify(str(uploader)) - new_stem = f"{slug_uploader}_{slug_title}_视频_{time_suffix}" - else: - new_stem = f"{slug_title}_视频_{time_suffix}" - - new_path = ensure_unique_path( - original_file.with_name(f"{new_stem}{original_file.suffix}") + author_dir = build_author_dir(uploader, uploader_id, source=url) + new_path = unique_media_path( + temp_root / author_dir / f"{build_work_stem(title)}{original_file.suffix}" ) - original_file.rename(new_path) + shutil.move(str(original_file), str(new_path)) + # 下载用的 scratch 目录已空,顺手收掉(cleanup 不删目录) + shutil.rmtree(temp_dir, ignore_errors=True) logger.info( f"yt-dlp 下载完成, 标题: {title}, " - f"作者: {uploader}, 重命名: {new_path}" + f"作者: {uploader}({uploader_id}), 落盘: {new_path}" ) - return new_path + return new_path, title diff --git a/hexi/plugins/nonebot_plugin_video_analysis/services/storage/group_file.py b/hexi/plugins/nonebot_plugin_video_analysis/services/storage/group_file.py new file mode 100644 index 0000000..8aa275d --- /dev/null +++ b/hexi/plugins/nonebot_plugin_video_analysis/services/storage/group_file.py @@ -0,0 +1,271 @@ +"""群文件上传 —— 与消息发送并行的第二条投递通道。 + +策略开关见 `policy.Policy.upload_group_file`:开着的时候,媒体照常发到群里, +同时另传一份到群文件(群友可随时下载、不占聊天记录)。 + +投递前会按全局配置打包(`video_analysis_group_file_zip`)成一个 zip: +配了解压密码(`video_analysis_group_file_password`)就用 AES-256 加密, +**密码设置了但 pyzipper 不可用时直接放弃上传,绝不退化成传明文**。 +(无密码的普通 zip 走标准库,零依赖。) + +失败只记日志(bot 没有群文件权限、超出群文件大小上限等都属于预期内的失败), +不影响消息发送链路;上传成功后不额外发消息,避免刷屏。 +""" + +from __future__ import annotations + +import asyncio +import re +import zipfile +from collections.abc import Iterable +from datetime import datetime +from pathlib import Path +from typing import TYPE_CHECKING, Any + +from nonebot import get_bot, logger + +if TYPE_CHECKING: # 仅类型检查:运行期不导入,单测可裸加载本模块 + from ..policy import Policy + +try: # 缺失时只影响「加密打包」这一路,见 build_archive + import pyzipper +except ImportError: # pragma: no cover + pyzipper = None # type: ignore[assignment] + + +def encryption_available() -> bool: + """加密打包是否可用(pyzipper 已安装)。""" + return pyzipper is not None + + +def build_archive( + files: Iterable[Path | str], + title: str = "", + *, + password: str = "", + out_dir: Path | None = None, + rel_dir: str = "", +) -> Path: + """把文件打包成一个 zip,返回产物路径;password 非空则 AES-256 加密。 + + out_dir 缺省落 `hexi/data/temp/archive/{作者目录}`(rel_dir 由调用方传入, + 与源媒体同一套目录结构,见 sender.media_rel_dir_of),由 cleanup 按天清理。 + 加密需要 pyzipper,缺失时抛 RuntimeError —— 调用方应当**放弃上传**。 + """ + paths = [Path(f) for f in files] + if not paths: + raise ValueError("没有可打包的文件") + missing = [p for p in paths if not p.exists()] + if missing: + raise FileNotFoundError(f"待打包文件不存在: {missing[0]}") + if password and pyzipper is None: + raise RuntimeError("配置了解压密码,但 pyzipper 未安装,无法加密打包") + + archive = _resolve_target(out_dir, title, rel_dir) + + if password: + with pyzipper.AESZipFile( + archive, + "w", + compression=pyzipper.ZIP_DEFLATED, + encryption=pyzipper.WZ_AES, + ) as zf: + zf.setpassword(password.encode("utf-8")) + _write_members(zf, paths) + else: + with zipfile.ZipFile(archive, "w", zipfile.ZIP_DEFLATED) as zf: + _write_members(zf, paths) + + logger.info( + f"群文件打包完成: {archive.name}({len(paths)} 个文件" + f"{',已加密' if password else ''})" + ) + return archive + + +def _default_archive_dir() -> Path: + """默认落 temp/archive(延迟导入 utils:单测裸加载本模块时没有包上下文)。""" + from ...utils import get_temp_root + + return get_temp_root("archive") + + +def _safe_stem(title: str, max_length: int = 15) -> str: + """标题 → 安全的文件名片段。 + + 原始标题(抖音文案、YouTube 标题)可能带 `/`、换行、控制字符与 `#话题`, + 这些进不了文件名,统一换成 `_` 后截断。 + """ + cleaned = re.sub(r"#\S+", "", title) + cleaned = re.sub(r'[\\/:*?"<>|\s]+', "_", cleaned.strip()) + cleaned = re.sub(r"_+", "_", cleaned).strip("_") + return cleaned[:max_length].strip("_") + + +def _resolve_target(out_dir: Path | None, title: str, rel_dir: str = "") -> Path: + """产物最终路径:默认 temp/archive/{rel_dir},重名自动加序号。 + + 空标题 / 占位符 / 已带「群文件」前缀的一律回退成默认名。 + """ + base = Path(out_dir) if out_dir is not None else _default_archive_dir() + if rel_dir: + base = base / rel_dir + base.mkdir(parents=True, exist_ok=True) + + stamp = f"{datetime.now():%H%M%S}" + stem = _safe_stem(title) if title else "" + if not stem or stem in ("title", "video") or stem.startswith("群文件"): + return _unique_path(base / f"群文件_{stamp}.zip") + return _unique_path(base / f"{stem}_群文件_{stamp}.zip") + + +def _unique_path(path: Path) -> Path: + """重名追加 _2/_3…(同 utils.ensure_unique_path 的语义,就地实现)。""" + if not path.exists(): + return path + index = 2 + while True: + candidate = path.with_name(f"{path.stem}_{index}{path.suffix}") + if not candidate.exists(): + return candidate + index += 1 + + +def _write_members(zf: Any, paths: list[Path]) -> None: + """写入成员:同名文件自动加序号,避免在包里互相覆盖。""" + used: set[str] = set() + for path in paths: + name = path.name + if name in used: + index = 2 + while f"{path.stem}_{index}{path.suffix}" in used: + index += 1 + name = f"{path.stem}_{index}{path.suffix}" + used.add(name) + zf.write(path, arcname=name) + + +def _file_uri(path: Path) -> str: + """本地文件 → `file:///D:/a/b.zip`(正斜杠,中文/空格不转义)。 + + 与图片/视频消息发给 NapCat 的形式一致;直接传 Windows 反斜杠路径会被 + 它的 realpath 判成 ENOENT(见 upload_group_file 的降级说明)。 + """ + return "file:///" + str(path.resolve()).replace("\\", "/") + + +async def _call_upload( + file_value: str, group_id: int, name: str, folder_id: str | None = None +) -> None: + params: dict = {"group_id": group_id, "file": file_value, "name": name} + if folder_id: + params["folder"] = folder_id + await get_bot().call_api("upload_group_file", **params) + + +def _s3_url(path: Path, policy: "Policy | None") -> str: + """把文件传到局域网 S3 换预签名链接(延迟导入 s3,便于单测裸加载本模块)。""" + from .s3 import upload_with_plan + + url, _ = upload_with_plan(path, policy=policy) + return url + + +async def upload_group_file( + file_path: Path | str, + group_id: int, + *, + folder_id: str | None = None, + policy: Policy | None = None, + name: str | None = None, +) -> bool: + """上传单个文件到群文件,返回是否成功。 + + 两级投递,**S3 链接优先、本地直传兜底**: + + 本环境实测 NapCat 读不到 bot 进程写的本地文件(`realpath ... ENOENT`, + 媒体消息的本地直发 55 次全失败、换 S3 链接后次次成功),所以直传只留作 + S3 不可用时的兜底。policy 决定 S3 落到哪个桶(plan)。 + + 本地直传用 `file:///D:/…` 形式(正斜杠、不转义中文)——图片/视频消息就是 + 这么发的,同机部署时能work。 + + name 可覆盖群文件列表里显示的名字(多图作品的成员是 001.jpg,调用方会补上 + 作品名前缀,免得多张图在群文件里全叫 001.jpg)。 + """ + path = Path(file_path) + if not path.exists(): + logger.warning(f"群文件上传:文件不存在 {path}") + return False + + display_name = name or path.name + + # 1) 局域网 S3 预签名链接(boto3 同步,丢线程池) + url = await asyncio.to_thread(_s3_url, path, policy) + if url: + try: + await _call_upload(url, group_id, display_name, folder_id) + except Exception as e: + logger.warning( + f"群文件 S3 链接上传失败(群 {group_id} / {display_name}): " + f"{str(e)[:160]};改用本地路径重试" + ) + else: + logger.info(f"群文件上传成功(群 {group_id},S3 链接): {display_name}") + return True + else: + logger.warning("群文件:S3 中转没拿到链接,改用本地路径直传") + + # 2) 兜底:file:/// 直传本地文件 + try: + await _call_upload(_file_uri(path), group_id, display_name, folder_id) + except Exception as e: + logger.warning( + f"群文件上传失败(群 {group_id} / {display_name}): {str(e)[:160]}" + ) + return False + + logger.info(f"群文件上传成功(群 {group_id},本地直传): {display_name}") + return True + + +async def upload_group_files( + files: Iterable[Path | str], + group_id: int, + *, + title: str = "", + zip_files: bool = True, + password: str = "", + policy: Policy | None = None, + rel_dir: str = "", +) -> bool: + """群文件投递入口:按配置打包(可加密)后传一个包,或逐个传原文件。 + + rel_dir 是源媒体所在的 `{作者}_{作者id}[/{作品名}]` 子目录,打包产物落到 + `temp/archive/{rel_dir}`(由调用方从文件路径推出来,见 sender)。 + """ + paths = [Path(f) for f in files] + if not paths: + return False + + if not zip_files: + # 多图作品逐个传时,成员名是 001.jpg…,补上作品名前缀便于在群文件里辨认 + work = Path(rel_dir).name if rel_dir else "" + prefix = f"{work}_" if work and len(paths) > 1 else "" + results = [ + await upload_group_file( + p, group_id, policy=policy, name=f"{prefix}{p.name}" + ) + for p in paths + ] + return any(results) + + try: + archive = await asyncio.to_thread( + build_archive, paths, title, password=password, rel_dir=rel_dir + ) + except Exception as e: + logger.error(f"群文件打包失败,跳过本次群文件上传: {e}") + return False + + return await upload_group_file(archive, group_id, policy=policy) diff --git a/hexi/plugins/nonebot_plugin_video_analysis/services/storage/s3.py b/hexi/plugins/nonebot_plugin_video_analysis/services/storage/s3.py index 5d2e4a3..f3f86f9 100644 --- a/hexi/plugins/nonebot_plugin_video_analysis/services/storage/s3.py +++ b/hexi/plugins/nonebot_plugin_video_analysis/services/storage/s3.py @@ -1,7 +1,6 @@ """统一 S3 存储模块 — 合并本地局域网 S3 和公网 MinIO""" from pathlib import Path -from time import strftime, localtime import boto3 from botocore.config import Config @@ -9,6 +8,9 @@ from nonebot import logger from hexi.web_hub.web_config import get_effective_value +from ...policy import Policy +from ...utils import media_key_of + # 该插件模块名,用于读取统一配置值库(Web 修改后生效) _PLUGIN_ID = "hexi.plugins.nonebot_plugin_video_analysis" @@ -171,11 +173,13 @@ def upload_to_public_s3(file_path: str | Path) -> tuple[str, str]: """ 上传到公网 MinIO,返回 (公网URL, file_key) - 私聊场景使用,生成可公网访问的链接 + 私聊场景使用,生成可公网访问的链接。 + + key 按本地落盘结构推导:`{作者}_{作者id}/{作品名}[_{短码}].ext` + (不再用 `{年月日}/` 前缀,见 utils.media_key_of)。 """ file_path = Path(file_path) - current_day = strftime("%Y-%m-%d", localtime()) - file_key = f"{current_day}/{file_path.name}" + file_key = media_key_of(file_path) try: url = _get_public_s3().upload_public(str(file_path), file_key) return url, file_key @@ -184,22 +188,19 @@ def upload_to_public_s3(file_path: str | Path) -> tuple[str, str]: return "", "" -def upload_to_local_s3( - title: str, image_post: bool, file_path: str | Path, plan: str | None = None -) -> str: +def upload_to_local_s3(file_path: str | Path, plan: str | None = None) -> str: """ 上传到局域网 S3,返回预签名 URL plan="A" → PLANA 桶 plan="B" → PLANB 桶 None → PLANC 桶(默认) + + key 与公网一致:`{作者}_{作者id}/{作品名}[_{短码}].ext` + (不再有 `{年月日}/` 前缀,也不再重复套一层标题目录)。 """ file_path = Path(file_path) - current_day = strftime("%Y-%m-%d", localtime()) - if image_post: - file_key = f"{current_day}/{title}/{file_path.name}" - else: - file_key = f"{current_day}/{file_path.name}" + file_key = media_key_of(file_path) if plan == "A": client = _get_local_s3_plana() @@ -224,35 +225,24 @@ def delete_from_public_s3(file_key: str) -> bool: def upload_with_plan( file_path: str | Path, *, - plan: str | None = None, - is_private: bool = False, - title: str = "", - image_post: bool = False, + policy: Policy | None = None, ) -> tuple[str, str | None]: """ - 统一上传入口:按 plan 路由到对应的本地桶,并按需上传公网 + 统一上传入口:按策略路由本地桶,并按需上传公网 - plan="A" → PLANA(仅本地) - plan="B" → PLANB + 公网 - 私聊 → PLANB + 公网 - 默认 → PLANC(仅本地) + policy.plan → PLANA / PLANB / PLANC(默认 C) + policy.upload_public → 是否额外上传公网(决定能否发下载链接) + + key 由文件路径推导(utils.media_key_of),调用方不再传标题。 Returns: (local_url, public_url_or_none) """ - # 本地上传 - if plan == "A": - local_url = upload_to_local_s3(title, image_post, file_path, plan="A") - elif plan == "B": - local_url = upload_to_local_s3(title, image_post, file_path, plan="B") - elif is_private: - local_url = upload_to_local_s3(title, image_post, file_path, plan="B") - else: - local_url = upload_to_local_s3(title, image_post, file_path) # PLANC + policy = policy or Policy() + local_url = upload_to_local_s3(file_path, plan=policy.plan) - # 公网上传(仅 PLANB 或私聊) public_url = None - if plan == "B" or is_private: + if policy.upload_public: public_url, _ = upload_to_public_s3(file_path) return local_url, public_url diff --git a/hexi/plugins/nonebot_plugin_video_analysis/services/web_jobs.py b/hexi/plugins/nonebot_plugin_video_analysis/services/web_jobs.py new file mode 100644 index 0000000..684f61a --- /dev/null +++ b/hexi/plugins/nonebot_plugin_video_analysis/services/web_jobs.py @@ -0,0 +1,508 @@ +"""Web 管理台的「链接解析 + 预览」任务层(供 web_hub.py 调用)。 + +群消息链路(`handlers/entry.py::dispatch_url`)从解析到投递都绑在 event/bot 上, +Web 上下文里第一次 `UniMessage.send()` 就会抛 SerializeFailed,所以这里**只复用 +纯函数层**:fetchers 的解析/下载 + `storage/s3.py::upload_with_plan`, +全程不发消息、不碰 event、不读群策略。 + +存储按**默认策略**走(`STORE.default_policy()` 的 plan / upload_public): +产出什么链接就返回什么链接 —— 开了 `upload_public` 才有公网链接,否则只有 +局域网 S3 的预签名链接(1 小时过期,靠 `refresh()` 重传换新)。 + +任务表在内存里(进程重启即清空,前端按"查不到就算了"处理): + + submit(urls, force=False) 新建任务;同 URL 已有 queued/running 任务时**复用** + list_jobs() 全部任务,新的在前 + refresh(job_id) 对已落盘文件重跑上传,换一批新链接 + +抖音每次都新起一个 Chrome,所以并发闸门固定 `Semaphore(2)`;单任务 240s 超时兜底。 +单文件上传失败只标该文件(`files[i].error`),不整体判失败 —— 多图作品挂一张 +不该让整条任务失败,缺的那个文件点「刷新链接」还能补回来。 + +进度只有两档(解析中 → 上传中):解析和下载都在 fetcher 内部完成,中间没有可挂的钩子。 +""" + +from __future__ import annotations + +import asyncio +import re +import subprocess +import time +import uuid +from collections.abc import Awaitable, Callable, Iterable +from dataclasses import dataclass, field +from pathlib import Path +from typing import TYPE_CHECKING, Any, Optional + +from nonebot import logger + +if TYPE_CHECKING: # 仅类型检查:运行期不导入,单测可裸加载本模块 + from ..policy import Policy + +# ─────────────────────────── 常量 ─────────────────────────── + +#: URL 的终止符:空白 + 中文标点(全角括号/引号/破折号也要断) +URL_STOP = ",。!?、;:()【】《》「」『』…—“”‘’" +#: 从整段文本里挑 http(s) 链接。形制同 handlers/entry.py::URL_PATTERN, +#: 但把中文标点也当成终止符 —— `链接A,链接B` 中间没有空格也能各自成条 +#: (`\S+` 会把它们连同后面的中文连成一个词,整段粘进来时更容易踩到)。 +URL_PATTERN = re.compile(r"https?://[^" + re.escape(URL_STOP) + r"\s]+") +#: 兜底再洗一遍收尾标点(entry.py::dispatch_url 同款) +URL_TRAILING = URL_STOP + +#: 任务表上限与终态任务的保留时长(秒) +MAX_JOBS = 100 +TERMINAL_TTL = 30 * 60 +#: 单任务超时(含下载,不含排队等闸门的时间) +JOB_TIMEOUT = 240 +#: 并发闸门:抖音每次都新起 Chrome,必须限流 +MAX_CONCURRENCY = 2 + +VIDEO_SUFFIXES = {".mp4", ".webm", ".mov", ".flv", ".mkv", ".ts", ".m4v"} +IMAGE_SUFFIXES = {".jpg", ".jpeg", ".png", ".gif", ".webp", ".bmp", ".avif"} + +#: 视频封面(关键帧):宽度压到 480 够卡片用,别把 4K 原帧传上云 +POSTER_WIDTH = 480 +POSTER_SUFFIX = "_封面.jpg" +POSTER_TIMEOUT = 20 + +#: 状态机:queued → running → done / failed +ACTIVE_STATUSES = ("queued", "running") + +# ─────────────────────────── 纯函数 ─────────────────────────── + + +def extract_urls(text: str) -> list[str]: + """整段文本 → 去重后的 URL 列表(保序,去掉尾随中文标点)。 + + 多行粘贴时一行一条;URL 后面跟的"。"这类标点不能进链接。 + """ + urls: list[str] = [] + for raw in URL_PATTERN.findall(text or ""): + url = raw.rstrip(URL_TRAILING) + if url and url not in urls: + urls.append(url) + return urls + + +def kind_of(path: Path | str) -> str: + """文件类型(前端据此决定内联预览方式)。""" + suffix = Path(path).suffix.lower() + if suffix in VIDEO_SUFFIXES: + return "video" + if suffix in IMAGE_SUFFIXES: + return "image" + return "file" + + +# ─────────────────────────── 依赖注入点 ─────────────────────────── + +#: (标题, 落盘文件列表, 是否图文作品) +FetchResult = tuple[Optional[str], list[Path], bool] +Fetcher = Callable[[str], Awaitable[FetchResult]] +#: (局域网链接, 公网链接) —— 拿不到时返回 "" / None +Uploader = Callable[[Path, "Optional[Policy]"], Awaitable[tuple[str, Optional[str]]]] +#: 视频 → 封面图(抽关键帧);抽不出来返回 None +PosterMaker = Callable[[Path], Awaitable[Optional[Path]]] + + +def ffmpeg_path() -> str: + """ffmpeg 可执行文件(解析逻辑复用 video_downloader;单测可替换)""" + from .fetchers.video_downloader import get_ffmpeg_path + + return get_ffmpeg_path() + + +async def make_poster(video: Path) -> Optional[Path]: + """抽一帧当视频封面,落同目录 `{作品名}_封面.jpg`;失败返回 None(不影响任务)。 + + 先取第 1 秒(躲开黑场/淡入),整段不到 1 秒的视频退回第 0 秒再试一次。 + """ + out = video.with_name(f"{video.stem}{POSTER_SUFFIX}") + for seek in ("1", "0"): + cmd = [ + ffmpeg_path(), + "-ss", + seek, + "-i", + str(video), + "-frames:v", + "1", + "-vf", + f"scale={POSTER_WIDTH}:-2", + "-q:v", + "4", + "-y", + str(out), + ] + try: + # 同步子进程必须丢线程池,否则抽帧期间整个事件循环都停摆 + await asyncio.to_thread( + subprocess.run, cmd, capture_output=True, timeout=POSTER_TIMEOUT + ) + except Exception: # noqa: BLE001 — 封面失败不该拖累任务 + logger.warning(f"抽关键帧失败:{video.name}") + return None + if out.exists() and out.stat().st_size > 0: + return out + return None + + +def platform_of(url: str) -> Optional[str]: + """URL → 平台规范标签(延迟导入 policy,便于单测裸加载本模块)。""" + from ..policy import match_platform + + return match_platform(url) + + +def default_policy() -> "Optional[Policy]": + """Web 解析统一按默认策略走(不是某个群的策略)。""" + from ..policy import STORE + + return STORE.default_policy() + + +async def fetch_media(url: str) -> FetchResult: + """按平台解析并下载媒体(纯函数层:不发消息、不碰 event、不上传)。 + + 分派规则与 `handlers/entry.py::dispatch_url` 一致:抖音 → parse_douyin; + B站动态/专栏 → fetch_bilibili_content;其余 B站链接 → download_video(yt-dlp); + 小红书 → fetch_rednote_content;其它平台 → download_video。 + b23 / xhslink / 抖音短链的重定向在各 fetcher 内部自己处理。 + """ + from ..handlers.douyin import parse_douyin + from ..policy import BILIBILI, DOUYIN, XHS + from .fetchers.bilibili_content import fetch_bilibili_content + from .fetchers.rednote_content import fetch_rednote_content + from .fetchers.video_downloader import download_video + + platform = platform_of(url) + title: Optional[str] = None + parsed: Path | list[Path] | None = None + image_post = False + + if platform == DOUYIN: + title, parsed, image_post = await parse_douyin(url) + elif platform == XHS: + title, parsed = await fetch_rednote_content(url) + image_post = isinstance(parsed, list) + elif platform == BILIBILI and any( + kw in url + for kw in ( + "bilibili.com/opus", + "bilibili.com/dynamic", + "t.bilibili.com", + "bilibili.com/read", + ) + ): + title, parsed = await fetch_bilibili_content(url) + image_post = isinstance(parsed, list) + else: + video_file, title = await download_video(url) + parsed = video_file + + if isinstance(parsed, list): + files = [Path(p) for p in parsed] + else: + files = [Path(parsed)] if parsed else [] + return title, files, image_post + + +async def upload_file( + path: Path, policy: "Optional[Policy]" +) -> tuple[str, Optional[str]]: + """上传单个文件 → (局域网链接, 公网链接)。 + + boto3 是同步的,必须丢线程池:直接在协程里跑会卡死整个事件循环 + (Web 轮询、群消息全都停摆)。 + """ + from .storage.s3 import upload_with_plan + + return await asyncio.to_thread(upload_with_plan, path, policy=policy) + + +# ─────────────────────────── 任务模型 ─────────────────────────── + + +@dataclass +class WebJob: + """一条解析任务;`to_dict()` 就是 API 的返回体。""" + + id: str + url: str + platform: Optional[str] = None + title: str = "" + status: str = "queued" + #: 运行中的进度文案:解析中 / 上传中(见模块文档:解析与下载不细分) + stage: str = "" + #: 每个文件:{name, kind, size, local_url, public_url, error} + files: list[dict[str, Any]] = field(default_factory=list) + #: 卡片封面:有图取第一张图,全是视频则抽第一帧(见 `_resolve_cover`) + cover_url: str = "" + error: str = "" + created_at: float = field(default_factory=time.time) + updated_at: float = field(default_factory=time.time) + #: 落盘的源文件(不出 API;`refresh()` 重传要用) + paths: list[Path] = field(default_factory=list, repr=False) + #: 抽出来的封面文件(不出 API;`refresh()` 重传要用) + poster_path: Optional[Path] = field(default=None, repr=False) + + @property + def active(self) -> bool: + """还在排队或运行(终态任务才会被 prune 掉)。""" + return self.status in ACTIVE_STATUSES + + def to_dict(self) -> dict[str, Any]: + return { + "id": self.id, + "url": self.url, + "platform": self.platform, + "title": self.title, + "status": self.status, + "stage": self.stage, + "files": self.files, + "cover_url": self.cover_url, + "error": self.error, + "created_at": self.created_at, + "updated_at": self.updated_at, + } + + +class JobManager: + """内存任务表:上限 MAX_JOBS 条,终态任务 TTL 30 分钟后自动清掉。 + + fetch / upload / poster / policy 四个依赖可注入(单测塞假实现,不碰网络与磁盘)。 + """ + + def __init__( + self, + *, + fetch: Optional[Fetcher] = None, + upload: Optional[Uploader] = None, + poster: Optional[PosterMaker] = None, + policy: Optional[Callable[[], "Optional[Policy]"]] = None, + ) -> None: + self._jobs: dict[str, WebJob] = {} + self._tasks: dict[str, asyncio.Task[None]] = {} + self._sem: Optional[asyncio.Semaphore] = None + self._fetch: Fetcher = fetch or fetch_media + self._upload: Uploader = upload or upload_file + self._poster: PosterMaker = poster or make_poster + self._policy: Callable[[], "Optional[Policy]"] = policy or default_policy + + # ── 查询 ────────────────────────────────────────────── + + def list_jobs(self) -> list[WebJob]: + """全部任务,新的在前。""" + self._prune() + return sorted(self._jobs.values(), key=lambda j: j.created_at, reverse=True) + + def get(self, job_id: str) -> Optional[WebJob]: + return self._jobs.get(job_id) + + def _find_active(self, url: str) -> Optional[WebJob]: + for job in self._jobs.values(): + if job.url == url and job.active: + return job + return None + + # ── 提交与执行 ──────────────────────────────────────── + + async def submit(self, urls: Iterable[str], *, force: bool = False) -> list[WebJob]: + """提交一批 URL;同 URL 已有在跑的任务时直接复用(force=True 强行新建)。 + + 复用不只是省一次解析:抖音双开浏览器毫无意义,而且两个任务同时往 + `unique_media_path` 的同一个路径写会撞车(它是 exists → 改名的写法)。 + """ + self._prune() + jobs: list[WebJob] = [] + for url in urls: + job = None if force else self._find_active(url) + if job is None: + job = WebJob( + id=uuid.uuid4().hex[:12], url=url, platform=platform_of(url) + ) + self._jobs[job.id] = job + self._spawn(job) + jobs.append(job) + return jobs + + async def refresh(self, job_id: str) -> Optional[WebJob]: + """对已落盘文件重跑上传,换一批新的预签名链接(旧的 1 小时过期)。 + + 任务不存在 / 还没有文件时返回 None,由调用方给提示。 + """ + job = self._jobs.get(job_id) + if job is None or not job.files: + return None + policy = self._policy() + job.files = [await self._upload_one(path, policy) for path in job.paths] + job.cover_url = await self._resolve_cover(job, policy) + job.updated_at = time.time() + return job + + # ── 清理 ────────────────────────────────────────────── + + def remove(self, job_id: str) -> bool: + """删掉一条任务;还在跑的一并取消(Chrome 那头由 playwright 自己收尾)。""" + job = self._jobs.pop(job_id, None) + if job is None: + return False + task = self._tasks.pop(job_id, None) + if task is not None and not task.done(): + task.cancel() + return True + + def clear_finished(self) -> int: + """清掉所有终态任务(排队/运行中的不动),返回清掉几条。""" + finished = [jid for jid, job in self._jobs.items() if not job.active] + for job_id in finished: + self._jobs.pop(job_id, None) + return len(finished) + + def _spawn(self, job: WebJob) -> None: + task = asyncio.create_task(self._run(job)) + self._tasks[job.id] = task + task.add_done_callback(lambda _t: self._tasks.pop(job.id, None)) + + async def _run(self, job: WebJob) -> None: + """排队等闸门 → 执行;任何异常都落到任务状态里,不外抛。""" + try: + async with self._gate(): + await asyncio.wait_for(self._execute(job), JOB_TIMEOUT) + except asyncio.CancelledError: + raise + except asyncio.TimeoutError: # 3.11+ 就是内置 TimeoutError,wait_for 抛的 + self._update(job, status="failed", error=f"解析超时(超过 {JOB_TIMEOUT}s)") + except Exception as e: # noqa: BLE001 — 任务边界,失败即任务状态 + logger.exception(f"Web 解析任务失败:{job.url}") + self._update(job, status="failed", error=str(e) or type(e).__name__) + + async def _execute(self, job: WebJob) -> None: + self._update(job, status="running", stage="解析中") + policy = self._policy() + + title, files, _ = await self._fetch(job.url) + job.title = title or job.url + if not files: + raise RuntimeError("无法解析到媒体(链接失效 / 风控 / cookies 过期)") + + job.paths = list(files) + self._update(job, stage="上传中") + + entries: list[dict[str, Any]] = [] + for path in files: + entries.append(await self._upload_one(path, policy)) + job.files = entries # 落一个刷一个,前端能看着进度出图 + job.cover_url = await self._resolve_cover(job, policy) + self._update(job, status="done", stage="") + + async def _resolve_cover(self, job: WebJob, policy: "Optional[Policy]") -> str: + """任务封面:有图就用第一张图;全是视频就抽第一帧上传当封面。 + + 抽帧/上传失败都只返回空串 —— 卡片那边退化成占位块,不影响任务本身。 + """ + for entry in job.files: + if entry["kind"] == "image": + url = entry["public_url"] or entry["local_url"] + if url: + return url + + if job.poster_path and job.poster_path.exists(): + entry = await self._upload_one(job.poster_path, policy) + return entry["public_url"] or entry["local_url"] + + for path in job.paths: + if kind_of(path) != "video" or not path.exists(): + continue + poster = await self._poster(path) + if poster is None: + return "" + job.poster_path = poster + entry = await self._upload_one(poster, policy) + return entry["public_url"] or entry["local_url"] + return "" + + async def _upload_one( + self, path: Path, policy: "Optional[Policy]" + ) -> dict[str, Any]: + """上传单个文件 → files 里的一项;失败只标这项。""" + if not path.exists(): + return _file_entry(path, error="本地文件不存在(可能已被 temp 清理)") + + try: + local_url, public_url = await self._upload(path, policy) + except Exception as e: # noqa: BLE001 — 单文件失败不拖累整条任务 + logger.exception(f"Web 任务上传失败:{path}") + return _file_entry(path, error=f"上传失败:{str(e)[:120]}") + + entry = _file_entry(path, local_url=local_url, public_url=public_url) + if not entry["local_url"] and not entry["public_url"]: + entry["error"] = "上传失败(S3 没返回链接)" + return entry + + # ── 内部工具 ────────────────────────────────────────── + + def _gate(self) -> asyncio.Semaphore: + """惰性建闸门:构造必须发生在跑着的事件循环里。""" + if self._sem is None: + self._sem = asyncio.Semaphore(MAX_CONCURRENCY) + return self._sem + + def _update( + self, + job: WebJob, + *, + status: Optional[str] = None, + stage: Optional[str] = None, + error: Optional[str] = None, + ) -> None: + if status is not None: + job.status = status + if stage is not None: + job.stage = stage + if error is not None: + job.error = error + job.updated_at = time.time() + + def _prune(self) -> None: + """先清超龄的终态任务,再按上限砍掉最旧的终态任务(在跑的不动)。""" + now = time.time() + for job_id, job in list(self._jobs.items()): + if not job.active and now - job.updated_at > TERMINAL_TTL: + self._jobs.pop(job_id, None) + + overflow = len(self._jobs) - MAX_JOBS + if overflow <= 0: + return + finished = sorted( + (j for j in self._jobs.values() if not j.active), + key=lambda j: j.created_at, + ) + for job in finished[:overflow]: + self._jobs.pop(job.id, None) + + +def _file_entry( + path: Path, + *, + local_url: str = "", + public_url: Optional[str] = None, + error: str = "", +) -> dict[str, Any]: + try: + size = path.stat().st_size + except OSError: + size = 0 + return { + "name": path.name, + "kind": kind_of(path), + "size": size, + "local_url": local_url or "", + "public_url": public_url or "", + "error": error, + } + + +#: 全局单例(web_hub.py 的路由共用) +JOBS = JobManager() diff --git a/hexi/plugins/nonebot_plugin_video_analysis/utils.py b/hexi/plugins/nonebot_plugin_video_analysis/utils.py index d6e22dd..1bda10a 100644 --- a/hexi/plugins/nonebot_plugin_video_analysis/utils.py +++ b/hexi/plugins/nonebot_plugin_video_analysis/utils.py @@ -1,8 +1,9 @@ +import hashlib import re +import time import unicodedata from pathlib import Path -from time import strftime, localtime -from typing import List +from typing import List, Optional, Tuple def get_data_dir() -> Path: @@ -72,6 +73,9 @@ def ensure_unique_path(base_path: Path) -> Path: """ 确保路径不冲突:如已存在则追加 _2, _3... 后缀 适用于文件和目录 + + 注:媒体落盘已统一走 `unique_media_path`(同名加 4 位短码),本函数仅作 + 通用兜底保留。 """ if not base_path.exists(): return base_path @@ -88,52 +92,139 @@ def ensure_unique_path(base_path: Path) -> Path: counter += 1 -def clean_filename(filename: str, max_length: int = 120) -> str: +_B36_ALPHABET = "0123456789abcdefghijklmnopqrstuvwxyz" + +#: 同名短码的模数 —— base36 四位(36**4 ≈ 168 万秒 ≈ 19.4 天一轮) +_B36_MOD = 36**4 + + +def _to_base36(value: int, width: int = 4) -> str: + """整数 → 定宽 base36(不足左侧补 0)""" + if value <= 0: + return "0" * width + digits = [] + while value: + value, rem = divmod(value, 36) + digits.append(_B36_ALPHABET[rem]) + return "".join(reversed(digits)).rjust(width, "0") + + +def short_time_code(at: Optional[float] = None) -> str: + """4 位 base36 短码(同名兜底用,不可读时间,仅作区分码)""" + return _to_base36(int(time.time() if at is None else at) % _B36_MOD) + + +#: 各平台把"没有 id"写成过这些值,别让它们进目录名 +_JUNK_AUTHOR_IDS = {"", "0", "none", "null", "na", "nan", "undefined"} + +#: 抓取层拿不到昵称时的兜底串(slugify 后)—— 它们等于"没有作者信息" +_PLACEHOLDER_AUTHORS = {"未知作者", "小红书用户", "b站用户"} + + +def _clean_author_id(author_id: Optional[str]) -> str: + """作者 id 归一:剔除占位值(yt-dlp 缺字段常给 `NA`),再 slugify。""" + raw = str(author_id or "").strip() + if raw.lower() in _JUNK_AUTHOR_IDS: + return "" + return slugify(raw, max_length=40) + + +def short_source_code(source: str) -> str: + """来源串(作品 URL/id)→ 4 位 base36 短码。 + + 与 `short_time_code` 的区别:同一个来源永远得到同一个码(md5 取摘要, + 不能用内置 `hash()` —— 它有随机盐,重启后目录名会变)。 """ - 清理文件名并添加时间前缀 - 支持多扩展名,如 .tar.gz + digest = hashlib.md5(source.encode("utf-8")).digest() + return _to_base36(int.from_bytes(digest[:4], "big") % _B36_MOD) + + +def build_author_dir( + author: Optional[str], + author_id: Optional[str] = None, + *, + source: Optional[str] = None, + at: Optional[float] = None, +) -> str: + """`{作者}_{作者id}` 作者目录名(拿不到 id 时追加码值避免同名混目录)。 + + - 昵称 + 作者 id → `{昵称}_{id}`(昵称上限 30、id 上限 40;**不能**把 id 截到 + 20:`MS4wLjABAAAA…` 这类 sec_uid 公共前缀就有 17 字符,再截断必撞车) + - 只有昵称 → `{昵称}_{4 位时间短码}`:没有 id 就分不清同名作者,宁可不聚合 + (同一作者的不同作品会各成一个目录)也不能把两个人混进同一个目录 + - 连昵称都没有 → `未知作者_{来源短码}`(`source` 给作品 URL/id,同一来源 + 稳定、不同来源不撞);连 source 都没有 → 退化成时间短码 """ + slug_author = slugify(author or "", max_length=30) + if slug_author in _PLACEHOLDER_AUTHORS: + slug_author = "" + slug_id = _clean_author_id(author_id) - current_time = strftime("%H-%M-%S", localtime()) + if slug_author and slug_id: + # 某些站点上传者名就是 handle(X 的 @someone),别产出 someone_someone + if slug_author == slug_id: + return slug_author + return f"{slug_author}_{slug_id}" + if slug_id: + return f"{slug_author or '未知作者'}_{slug_id}" + if slug_author: + return f"{slug_author}_{short_time_code(at)}" + if source: + return f"未知作者_{short_source_code(source)}" + return f"未知作者_{short_time_code(at)}" - p = Path(filename) - # 主文件名 - name = p.stem +def build_work_stem(title: Optional[str]) -> str: + """作品名做文件名/子目录名:slugify(沿用 15 字上限),空则 `作品`""" + return slugify(title or "", max_length=15) or "作品" - # 完整扩展名 (.tar.gz) - ext = "".join(p.suffixes) - # Unicode 标准化 - name = unicodedata.normalize("NFKC", name) +def unique_media_path(path: Path, *, at: Optional[float] = None) -> Path: + """同名才加 4 位短码:`{stem}_{码}{后缀}`,仍撞则再叠 `_2/_3…` - # 去掉 #tag - name = re.sub(r"#\S+", "", name) + 文件与目录通用(目录无后缀)。命名发生在落盘前,因此"不存在"即直接采用; + 顺带确保父目录存在(作者目录是按需创建的)。 + """ + if path.exists(): + code = short_time_code(at) + candidate = path.with_name(f"{path.stem}_{code}{path.suffix}") + index = 2 + while candidate.exists(): + candidate = path.with_name(f"{path.stem}_{code}_{index}{path.suffix}") + index += 1 + path = candidate - # 非法字符替换 - name = re.sub(r'[\\/:*?"<>|]', "_", name) + path.parent.mkdir(parents=True, exist_ok=True) + return path - # 中英文标点 - name = re.sub(r"[&'\"。,:?!《》【】|]", "_", name) - # 空白 -> _ - name = re.sub(r"\s+", "_", name) +def _temp_rel_parts(file_path: Path | str) -> Tuple[str, ...]: + """相对 temp 根拆路径:`{平台}/{作者目录}/…`(归档目录同理) - # 只保留:中文、字母、数字、_ - name = re.sub(r"[^\w一-鿿_]", "", name) + 不在 temp 下、或没到"平台 + 作者目录"这一层(老数据 / 第三方产物)→ 空元组, + 调用方退化为只用文件名。 + """ + try: + rel = Path(file_path).resolve().relative_to(get_temp_root().resolve()) + except (ValueError, OSError): + return () + parts = rel.parts + return parts[1:] if len(parts) >= 3 else () - # 合并 _ - name = re.sub(r"_+", "_", name) - # 去首尾 _ - name = name.strip("_") +def media_key_of(file_path: Path | str) -> str: + """媒体文件的 S3 对象 key:`{作者}_{作者id}/{作品名}[_{码}].后缀` - # 长度控制 - max_name_length = max_length - len(ext) - len(current_time) - 1 - if len(name) > max_name_length: - name = name[:max_name_length].rstrip("_") + 由本地路径反推(去掉平台层),保证桶里和 temp 里结构一致。 + """ + parts = _temp_rel_parts(file_path) + return "/".join(parts) if parts else Path(file_path).name - return f"{current_time}_{name}{ext}" + +def media_rel_dir_of(file_path: Path | str) -> str: + """媒体文件所在的作者/作品子目录(相对 temp 根、去掉平台层),供归档复用""" + parts = _temp_rel_parts(file_path) + return "/".join(parts[:-1]) if len(parts) > 1 else "" def parse_netscape_cookies(file_path: str) -> List[dict]: diff --git a/hexi/plugins/nonebot_plugin_video_analysis/web_hub.py b/hexi/plugins/nonebot_plugin_video_analysis/web_hub.py new file mode 100644 index 0000000..3d14b33 --- /dev/null +++ b/hexi/plugins/nonebot_plugin_video_analysis/web_hub.py @@ -0,0 +1,176 @@ +"""视频解析 Web API 子应用(挂载到 /api/video_analysis)。 + +群策略(data/list.json v3,见 policy.py)的唯一 Web 读写入口, +鉴权走 hexi.web_hub.web_auth(OAuth2 + SQLite),与统一管理台 /hub 共用登录态。 +前端页面:hexi/web/src/plugins/video_analysis/index.tsx。 + +写入全部落在 `policy.PolicyStore` 上(加锁 + 原子替换 + 归一化), +所以这里不需要再做字段校验,只要校验群号形态。 + +另有一组「链接解析 + 预览」接口(/parse、/jobs、/jobs/{id}/refresh): +粘链接 → 起任务 → 出下载链接 + 页面内预览,实现在 services/web_jobs.py。 +**每条路由都要自带 `dependencies=[auth]`** —— mount 层没有兜底, +漏一条就是匿名可访问(包括这条"让服务器去抓任意 URL"的接口)。 +""" + +from __future__ import annotations + +from fastapi import FastAPI +from fastapi.responses import JSONResponse + +from nonebot import get_adapter +from nonebot.adapters.onebot.v11 import Adapter + +from hexi.web_hub.web_auth import require_admin + +from .policy import PLANS, PLATFORMS, STORE, Policy +from .services.web_jobs import JOBS, extract_urls + +API = require_admin + + +def _ok(data=None, msg: str = "ok") -> JSONResponse: + return JSONResponse({"status": 0, "msg": msg, "data": data}) + + +def _fail(msg: str, status: int = 400) -> JSONResponse: + return JSONResponse({"status": status, "msg": msg}) + + +async def _group_names() -> dict[str, str]: + """群号 → 群名;拿不到 bot(未连接)时返回空表,前端只显示群号。""" + try: + bots = get_adapter(Adapter).bots + bot = next(iter(bots.values()), None) + if bot is None: + return {} + return { + str(g["group_id"]): g.get("group_name") or "" + for g in await bot.get_group_list() + } + except Exception: # noqa: BLE001 — 未连接/适配器未加载都按拿不到处理 + return {} + + +def _payload(policy: Policy) -> dict: + """策略的 JSON 视图;sends_link 是"实际会不会发链接"(供前端置灰)。""" + data = policy.to_dict() + data["sends_link"] = policy.sends_link + return data + + +def build_admin_app() -> FastAPI | None: + """构建群策略管理 API 子应用(挂载到 /api/video_analysis)。""" + app = FastAPI(title="Video Analysis API") + auth = require_admin + + @app.get("/overview", response_class=JSONResponse, dependencies=[auth]) + async def overview(): + """一次拿全:群策略列表 + 默认节 + 黑名单 + 平台/方案选项。""" + names = await _group_names() + groups = [ + { + "group_id": gid, + "group_name": names.get(gid, ""), + "online": gid in names, + "policy": _payload(policy), + } + for gid, policy in sorted(STORE.all_groups().items(), key=_sort_key) + ] + return _ok( + { + "groups": groups, + "default": _payload(STORE.default_policy()), + "blacklist": STORE.blacklist(), + "platforms": list(PLATFORMS), + "plans": list(PLANS), + "online": bool(names), + } + ) + + @app.post("/group", response_class=JSONResponse, dependencies=[auth]) + async def save_group(data: dict): + """新增/覆盖单个群的策略(群不存在则加入白名单)。""" + gid = str(data.get("group_id", "")).strip() + if not gid.isdigit(): + return _fail("群号必须是纯数字") + policy = await STORE.set_group(gid, Policy.from_dict(data.get("policy"))) + return _ok(_payload(policy), f"群 {gid} 策略已保存") + + @app.delete("/group/{group_id}", response_class=JSONResponse, dependencies=[auth]) + async def remove_group(group_id: str): + """移出白名单(该群不再解析)。""" + if not await STORE.remove_group(group_id): + return _fail(f"群 {group_id} 不在白名单中") + return _ok({"group_id": group_id}, f"群 {group_id} 已移出白名单") + + @app.post("/default", response_class=JSONResponse, dependencies=[auth]) + async def save_default(data: dict): + """保存默认策略(私聊与未配置群使用)。""" + policy = await STORE.set_default(Policy.from_dict(data.get("policy", data))) + return _ok(_payload(policy), "默认策略已保存") + + @app.post("/blacklist", response_class=JSONResponse, dependencies=[auth]) + async def save_blacklist(data: dict): + """整体替换全局黑名单(QQ 列表)。""" + raw = data.get("blacklist", []) + if not isinstance(raw, list): + return _fail("blacklist 必须是列表") + values = [str(x).strip() for x in raw if str(x).strip()] + invalid = [v for v in values if not v.isdigit()] + if invalid: + return _fail(f"黑名单只能是 QQ 号:{'、'.join(invalid)}") + + stored = await STORE.set_blacklist(values) + return _ok({"blacklist": stored}, "黑名单已保存") + + # ── 链接解析 + 预览(services/web_jobs.py) ────────────── + # 与群策略无关:解析结果按默认策略(default 节)的 plan / upload_public 存储。 + + @app.post("/parse", response_class=JSONResponse, dependencies=[auth]) + async def parse(data: dict): + """整段文本里挑链接 → 逐个起任务;同 URL 已有在跑的任务时直接复用。""" + urls = extract_urls(str(data.get("text") or "")) + if not urls: + return _fail("没找到链接") + jobs = await JOBS.submit(urls, force=bool(data.get("force"))) + return _ok( + {"jobs": [j.to_dict() for j in jobs], "urls": urls}, + f"已提交 {len(jobs)} 个任务", + ) + + @app.get("/jobs", response_class=JSONResponse, dependencies=[auth]) + async def list_jobs(): + """全部任务,新的在前(前端按 1.5s 轮询这一个接口)。""" + return _ok({"jobs": [j.to_dict() for j in JOBS.list_jobs()]}) + + @app.post( + "/jobs/{job_id}/refresh", response_class=JSONResponse, dependencies=[auth] + ) + async def refresh_job(job_id: str): + """重传已落盘文件换一批新链接(预签名链接 1 小时过期)。""" + job = await JOBS.refresh(job_id) + if job is None: + return _fail("任务不存在,或还没有可刷新的文件") + return _ok({"job": job.to_dict()}, "链接已刷新") + + @app.delete("/jobs/{job_id}", response_class=JSONResponse, dependencies=[auth]) + async def remove_job(job_id: str): + """删掉一条任务(还在跑的一并取消);只清任务表,temp 里的文件交给清理任务。""" + if not JOBS.remove(job_id): + return _fail("任务不存在") + return _ok({"job_id": job_id}, "任务已清理") + + @app.post("/jobs/clear", response_class=JSONResponse, dependencies=[auth]) + async def clear_jobs(): + """清掉所有已完成/失败的任务(排队、运行中的不动)。""" + removed = JOBS.clear_finished() + return _ok({"removed": removed}, f"已清理 {removed} 条任务") + + return app + + +def _sort_key(item: tuple[str, Policy]) -> tuple[int, int, str]: + """群号按数值排序(非数字的排最后,不参与数值比较)。""" + gid = item[0] + return (1, 0, gid) if not gid.isdigit() else (0, int(gid), "") diff --git a/hexi/web/package-lock.json b/hexi/web/package-lock.json index 74d466b..4286b16 100644 --- a/hexi/web/package-lock.json +++ b/hexi/web/package-lock.json @@ -13,7 +13,8 @@ "@heroui/styles": "^3.2.4", "react": "^19.2.8", "react-dom": "^19.2.8", - "react-router-dom": "^7.18.3" + "react-router-dom": "^7.18.3", + "yet-another-react-lightbox": "^3.32.2" }, "devDependencies": { "@tailwindcss/vite": "^4.3.3", @@ -30,6 +31,7 @@ "resolved": "https://registry.npmjs.org/@adobe/react-spectrum/-/react-spectrum-3.47.3.tgz", "integrity": "sha512-tWZG59+xTbXIqeZB4uyuo2qljk3nRK66IkdBmMgbl2p6peApkKbrtJ/YOM8rW4gemdmv7qoJlbIHvy8aJkYsCQ==", "license": "Apache-2.0", + "peer": true, "dependencies": { "@internationalized/date": "^3.12.3", "@react-types/shared": "^3.36.1", @@ -83,7 +85,6 @@ "resolved": "https://registry.npmjs.org/@formatjs/ecma402-abstract/-/ecma402-abstract-2.3.6.tgz", "integrity": "sha512-HJnTFeRM2kVFVr5gr5kH1XP6K0JcJtE7Lzvtr3FS/so5f1kpsqqqxy5JF+FRaO6H2qmcMfAUIox7AJteieRtVw==", "license": "MIT", - "peer": true, "dependencies": { "@formatjs/fast-memoize": "2.2.7", "@formatjs/intl-localematcher": "0.6.2", @@ -96,7 +97,6 @@ "resolved": "https://registry.npmjs.org/@formatjs/fast-memoize/-/fast-memoize-2.2.7.tgz", "integrity": "sha512-Yabmi9nSvyOMrlSeGGWDiH7rf3a7sIwplbvo/dlz9WCIjzIQAfy1RMf4S0X3yG724n5Ghu2GmEl5NJIV6O9sZQ==", "license": "MIT", - "peer": true, "dependencies": { "tslib": "^2.8.0" } @@ -106,7 +106,6 @@ "resolved": "https://registry.npmjs.org/@formatjs/icu-messageformat-parser/-/icu-messageformat-parser-2.11.4.tgz", "integrity": "sha512-7kR78cRrPNB4fjGFZg3Rmj5aah8rQj9KPzuLsmcSn4ipLXQvC04keycTI1F7kJYDwIXtT2+7IDEto842CfZBtw==", "license": "MIT", - "peer": true, "dependencies": { "@formatjs/ecma402-abstract": "2.3.6", "@formatjs/icu-skeleton-parser": "1.8.16", @@ -118,7 +117,6 @@ "resolved": "https://registry.npmjs.org/@formatjs/icu-skeleton-parser/-/icu-skeleton-parser-1.8.16.tgz", "integrity": "sha512-H13E9Xl+PxBd8D5/6TVUluSpxGNvFSlN/b3coUp0e0JpuWXXnQDiavIpY3NnvSp4xhEMoXyyBvVfdFX8jglOHQ==", "license": "MIT", - "peer": true, "dependencies": { "@formatjs/ecma402-abstract": "2.3.6", "tslib": "^2.8.0" @@ -129,7 +127,6 @@ "resolved": "https://registry.npmjs.org/@formatjs/intl-localematcher/-/intl-localematcher-0.6.2.tgz", "integrity": "sha512-XOMO2Hupl0wdd172Y06h6kLpBz6Dv+J4okPLl4LPtzbr8f66WbIoy4ev98EBuZ6ZK4h5ydTN6XneT4QVpD7cdA==", "license": "MIT", - "peer": true, "dependencies": { "tslib": "^2.8.0" } @@ -204,7 +201,6 @@ "resolved": "https://registry.npmjs.org/@internationalized/message/-/message-3.1.10.tgz", "integrity": "sha512-nc0Or6EdWHqZRcsXb6P9hBIpLsfSl/ILh0rk5h/OVBpzmhdExXtPy2cQtWsq8XKRBpRHwDNnAHt4OpolcB7dog==", "license": "Apache-2.0", - "peer": true, "dependencies": { "@swc/helpers": "^0.5.0", "intl-messageformat": "^10.1.0" @@ -1163,6 +1159,7 @@ "integrity": "sha512-AnzbBERsrLKtk2XSfTbYRLjQPdy116Sty4q+T+Bp3IC4l6jNBvreVPAHmpq9qhXQM7CXZPjLVmGMw9sy+hxQ3w==", "devOptional": true, "license": "MIT", + "peer": true, "dependencies": { "csstype": "^3.2.2" } @@ -1173,6 +1170,7 @@ "integrity": "sha512-fMPwH9v7r/pp43yUd2/Mbiex5KouJwwR3dzHkhLREUC6764VyDsqxhAxv6OFEYR1RhjOyD1naqba8ECDBe7ZQg==", "devOptional": true, "license": "MIT", + "peer": true, "peerDependencies": { "@types/react": "^19.2.0" } @@ -1257,8 +1255,7 @@ "version": "10.6.0", "resolved": "https://registry.npmjs.org/decimal.js/-/decimal.js-10.6.0.tgz", "integrity": "sha512-YpgQiITW3JXGntzdUmyUR1V812Hn8T1YVXhCu+wO3OpS4eU9l4YdD3qjyiKdV6mvV29zapkMeD390UVEf2lkUg==", - "license": "MIT", - "peer": true + "license": "MIT" }, "node_modules/detect-libc": { "version": "2.1.2", @@ -1349,7 +1346,6 @@ "resolved": "https://registry.npmjs.org/intl-messageformat/-/intl-messageformat-10.7.18.tgz", "integrity": "sha512-m3Ofv/X/tV8Y3tHXLohcuVuhWKo7BBq62cqY15etqmLxg2DZ34AGGgQDeR+SCta2+zICb1NX83af0GJmbQ1++g==", "license": "BSD-3-Clause", - "peer": true, "dependencies": { "@formatjs/ecma402-abstract": "2.3.6", "@formatjs/fast-memoize": "2.2.7", @@ -1697,6 +1693,7 @@ "integrity": "sha512-qcJu88Q2IWqJsDD529JKMdwGm/dvInW4HvQnRwiH9JtihJvzGOscDtHE3x1pBKeUOTysQ8kVmLnJ2kJu7yhcGA==", "dev": true, "license": "MIT", + "peer": true, "engines": { "node": ">=12" }, @@ -1749,6 +1746,7 @@ "resolved": "https://registry.npmjs.org/react/-/react-19.2.8.tgz", "integrity": "sha512-PWaYA1L/q9u2u7xYQi+Y3L3Yfnie7XyLeaJICV1MGD6LprsBxcAqGjYyr0eY3p+QdsA+x/Irkt4Qif8D63+Sbw==", "license": "MIT", + "peer": true, "engines": { "node": ">=0.10.0" } @@ -1758,6 +1756,7 @@ "resolved": "https://registry.npmjs.org/react-aria/-/react-aria-3.51.0.tgz", "integrity": "sha512-AyWLw0XR38cFPwBu/ErgGaVrc5dupLEKmRlMXTGvFKOtbaGRQ2+yQJkjVhpdHhoRhU4+G+tJDFeHDTS8tK3bfQ==", "license": "Apache-2.0", + "peer": true, "dependencies": { "@internationalized/date": "^3.12.3", "@internationalized/number": "^3.6.7", @@ -1779,6 +1778,7 @@ "resolved": "https://registry.npmjs.org/react-aria-components/-/react-aria-components-1.20.0.tgz", "integrity": "sha512-BMbpIgoV9aELeBrB0Y120NgoigHb5OdcJwc+4e7uSnbTbamea6lo+gqcc4LAxzMaK3Jf+7LI1oCDE6yANsmxIQ==", "license": "Apache-2.0", + "peer": true, "dependencies": { "@internationalized/date": "^3.12.3", "@internationalized/string": "^3.2.10", @@ -1798,6 +1798,7 @@ "resolved": "https://registry.npmjs.org/react-dom/-/react-dom-19.2.8.tgz", "integrity": "sha512-rVprimfGBG3DR+Tq0IQG2DT5PxKth1WIGDmj5yPmlzr4YBe7uyE+Du4oVqTDXZSHGGGXRtTJEGSSePyQCMBglQ==", "license": "MIT", + "peer": true, "dependencies": { "scheduler": "^0.27.0" }, @@ -1974,7 +1975,8 @@ "version": "4.3.3", "resolved": "https://registry.npmjs.org/tailwindcss/-/tailwindcss-4.3.3.tgz", "integrity": "sha512-gOhV3P7ufE62QDGg1zVaTgCR+EtPv92k2nIhVcVKcLmxT1sUBsQGhnZj175j+MqRt4zLF7ic+sCYjfhxMxj7YQ==", - "license": "MIT" + "license": "MIT", + "peer": true }, "node_modules/tapable": { "version": "2.3.3", @@ -2051,6 +2053,7 @@ "integrity": "sha512-cFKLV/PRgAUlIRm5WjMjJ86jrftzpqcgH+Us+DS8mI3CDNiH30Whrz8uHL3+MOLPAgqbMBAqWdAHAphOAM+z/Q==", "dev": true, "license": "MIT", + "peer": true, "dependencies": { "lightningcss": "^1.33.0", "picomatch": "^4.0.5", @@ -2383,6 +2386,32 @@ "type": "opencollective", "url": "https://opencollective.com/parcel" } + }, + "node_modules/yet-another-react-lightbox": { + "version": "3.32.2", + "resolved": "https://registry.npmjs.org/yet-another-react-lightbox/-/yet-another-react-lightbox-3.32.2.tgz", + "integrity": "sha512-F4HtHQfUNpvkj+AmECgWM4XRdCqMY5gXpKgOUx39+T+FyxLe8II4SK/pwMyYj2X54KH9lFSQeHYY1/GfYf3SdA==", + "license": "MIT", + "engines": { + "node": ">=14" + }, + "funding": { + "url": "https://github.com/sponsors/igordanchenko" + }, + "peerDependencies": { + "@types/react": "^16 || ^17 || ^18 || ^19", + "@types/react-dom": "^16 || ^17 || ^18 || ^19", + "react": "^16.8.0 || ^17 || ^18 || ^19", + "react-dom": "^16.8.0 || ^17 || ^18 || ^19" + }, + "peerDependenciesMeta": { + "@types/react": { + "optional": true + }, + "@types/react-dom": { + "optional": true + } + } } } } diff --git a/hexi/web/package.json b/hexi/web/package.json index 43c155b..c9e47b4 100644 --- a/hexi/web/package.json +++ b/hexi/web/package.json @@ -14,7 +14,8 @@ "@heroui/styles": "^3.2.4", "react": "^19.2.8", "react-dom": "^19.2.8", - "react-router-dom": "^7.18.3" + "react-router-dom": "^7.18.3", + "yet-another-react-lightbox": "^3.32.2" }, "devDependencies": { "@tailwindcss/vite": "^4.3.3", diff --git a/hexi/web/src/plugins/video_analysis/index.tsx b/hexi/web/src/plugins/video_analysis/index.tsx new file mode 100644 index 0000000..8edd8e4 --- /dev/null +++ b/hexi/web/src/plugins/video_analysis/index.tsx @@ -0,0 +1,881 @@ +import { useCallback, useEffect, useMemo, useRef, useState } from 'react' +import { + Button, Card, Chip, Drawer, Input, Label, ListBox, Select, Spinner, Switch, TextArea, TextField, + Tooltip, TooltipContent, toast, useOverlayState, +} from '@heroui/react' +import { ArrowDownToLine, ArrowUpRightFromSquare, CirclePlay, TrashBin, Xmark } from '@gravity-ui/icons' +import Lightbox from 'yet-another-react-lightbox' +import type { LightboxProps, Slide } from 'yet-another-react-lightbox' +import Inline from 'yet-another-react-lightbox/plugins/inline' +import Video from 'yet-another-react-lightbox/plugins/video' +import Captions from 'yet-another-react-lightbox/plugins/captions' +import Counter from 'yet-another-react-lightbox/plugins/counter' +import Thumbnails from 'yet-another-react-lightbox/plugins/thumbnails' +import Zoom from 'yet-another-react-lightbox/plugins/zoom' +import 'yet-another-react-lightbox/styles.css' +import 'yet-another-react-lightbox/plugins/captions.css' +import 'yet-another-react-lightbox/plugins/counter.css' +import 'yet-another-react-lightbox/plugins/thumbnails.css' +import { api, hubGroups } from '../../api/client' +import { PageHead, Tabs, selectCls, tdCls, thCls, type TabItem } from '../../components/ui' +import { fmtBytes, fmtTime } from '../../lib/format' + +const va = api('video_analysis') + +function data(res: any) { + if (res && typeof res === 'object' && typeof res.status === 'number') { + if (res.status !== 0) throw new Error(res.msg || '请求失败') + return res.data + } + return res +} + +interface Policy { + auto: boolean + auto_link: string[] + ban_link: string[] + plan: string + upload_public: boolean + send_link: boolean + upload_group_file: boolean + group_file_platforms: string[] + sends_link?: boolean +} + +interface GroupRow { group_id: string; group_name: string; online: boolean; policy: Policy } +interface JobFile { + name: string + kind: 'image' | 'video' | 'file' + size: number + local_url: string + public_url: string + error: string +} +interface WebJob { + id: string + url: string + platform: string | null + title: string + status: string + stage: string + files: JobFile[] + cover_url: string + error: string + created_at: number + updated_at: number +} +interface Overview { + groups: GroupRow[] + default: Policy + blacklist: string[] + platforms: string[] + plans: string[] + online: boolean +} +interface GroupOption { group_id: number; group_name: string } + +const EMPTY: Policy = { + auto: false, auto_link: [], ban_link: [], plan: 'C', + upload_public: false, send_link: false, upload_group_file: false, + group_file_platforms: [], +} + +const groupFileText = (p: Policy) => { + if (!p.upload_group_file) return '关' + return p.group_file_platforms.length ? `开·仅${p.group_file_platforms.join('/')}` : '开' +} + +const onOff = (v: boolean) => (v ? '开' : '关') + +// ─────────────────────── 链接解析(web_jobs.py) ─────────────────────── + +const POLL_MS = 1500 +const STATUS_TEXT: Record = { queued: '排队中', running: '解析中', done: '完成', failed: '失败' } +const isActive = (j: WebJob) => j.status === 'queued' || j.status === 'running' +// 运行中显示 stage(解析中/上传中),其余按状态表 +const jobStatusText = (j: WebJob) => (j.status === 'running' ? j.stage || '解析中' : STATUS_TEXT[j.status] || j.status) +// Chip 的语义色(HeroUI 就这五个:accent/danger/default/success/warning) +const STATUS_COLOR: Record = { + queued: 'default', running: 'warning', done: 'success', failed: 'danger', +} + +async function copyText(text: string) { + try { await navigator.clipboard.writeText(text); return true } + catch { + const ta = document.createElement('textarea'); ta.value = text; document.body.appendChild(ta); ta.select() + const ok = document.execCommand('copy'); ta.remove(); return ok + } +} + +// 浮层(yet-another-react-lightbox)默认文案是英文,跟页面其它部分对齐一下 +// (插件各自往 Labels 上加了键:Zoom in/out 来自 Zoom、Caption 来自 Captions) +const LIGHTBOX_LABELS: LightboxProps['labels'] = { + Previous: '上一张', + Next: '下一张', + Close: '关闭', + Slide: '媒体', + Carousel: '媒体列表', + Lightbox: '媒体预览', + Caption: '文件信息', + 'Zoom in': '放大', + 'Zoom out': '缩小', +} + +// 预览用的 MIME:产物统一 mp4(video_downloader 写死了 merge_output_format), +// 其余几种只是兜底,浏览器放不了的容器会在播放器里显示加载失败,不会白屏 +const VIDEO_MIME: Record = { + mp4: 'video/mp4', m4v: 'video/mp4', webm: 'video/webm', mov: 'video/quicktime', + mkv: 'video/x-matroska', ts: 'video/mp2t', flv: 'video/x-flv', +} + +function videoMime(name: string) { + return VIDEO_MIME[name.split('.').pop()?.toLowerCase() || ''] || 'video/mp4' +} + +function toSlide(file: JobFile, src: string): Slide { + const title = `${file.name} · ${fmtBytes(file.size)}` + if (file.kind === 'video') { + return { type: 'video', title, sources: [{ src, type: videoMime(file.name) }] } + } + return { src, alt: file.name, title } +} + +type MainTab = 'parse' | 'policy' +const MAIN_TABS: TabItem[] = [ + { key: 'parse', label: '链接解析' }, + { key: 'policy', label: '群策略' }, +] + +export default function VideoAnalysisPage() { + const [tab, setTab] = useState('parse') + // 页面撑满 main 的内容盒(main = 100vh - 顶栏 4rem - 自己的 padding p-4/p-6), + // 下面的工作区才能靠 flex-1 拿到"剩下的高度"——比写死 calc(100vh - 13rem) + // 这种算法稳:页头和顶栏 tab 改高度也不会把布局撑出滚动条。 + return ( +
+
+ 粘链接解析预览 · 群策略配置} + /> + setTab(v as MainTab)} /> +
+ {/* 滚动收在内容区里(群策略那个长表格不再撑动整页) */} +
+ {tab === 'parse' && } + {tab === 'policy' && } +
+
+ ) +} + + +/** 群策略:白名单群列表 + 默认策略 + 全局黑名单(原页面主体) */ +function PolicyView() { + const [ov, setOv] = useState(null) + const [botGroups, setBotGroups] = useState([]) + const [err, setErr] = useState('') + const [loading, setLoading] = useState(true) + const [editing, setEditing] = useState<{ group_id: string; policy: Policy; isNew: boolean } | null>(null) + const [saving, setSaving] = useState(false) + const drawer = useOverlayState() + + const load = useCallback(async () => { + setLoading(true); setErr('') + try { + setOv(await data(await va.get('overview'))) + } catch (e: any) { + setErr(e.message || '加载失败') + } finally { + setLoading(false) + } + }, []) + useEffect(() => { load() }, [load]) + // 群列表只在「添加群」的下拉里用;bot 未连接时用输入框兜底 + useEffect(() => { hubGroups().then(d => setBotGroups((d && d.items) || [])).catch(() => setBotGroups([])) }, []) + + const openNew = () => { + setEditing({ group_id: botGroups[0] ? String(botGroups[0].group_id) : '', policy: { ...EMPTY }, isNew: true }) + drawer.open() + } + const openEdit = (row: GroupRow) => { + const p = row.policy + setEditing({ + group_id: row.group_id, + isNew: false, + policy: { + auto: p.auto, auto_link: [...p.auto_link], ban_link: [...p.ban_link], plan: p.plan, + upload_public: p.upload_public, send_link: p.send_link, upload_group_file: p.upload_group_file, + group_file_platforms: [...p.group_file_platforms], + }, + }) + drawer.open() + } + + const saveGroup = async () => { + if (!editing) return + if (!/^\d+$/.test(editing.group_id)) { toast.danger('群号必须是纯数字'); return } + setSaving(true) + try { + await va.post('group', { group_id: editing.group_id, policy: editing.policy }) + toast.success(`群 ${editing.group_id} 策略已保存`) + drawer.close() + await load() + } catch (e: any) { toast.danger(e.message || '保存失败') } finally { setSaving(false) } + } + + const removeGroup = async (row: GroupRow) => { + if (!window.confirm(`确定把群 ${row.group_name || row.group_id} 移出白名单?该群将不再解析链接。`)) return + try { await va.del(`group/${row.group_id}`); toast.success('已移出白名单'); await load() } catch (e: any) { toast.danger(e.message) } + } + + if (loading && !ov) return
+ + return ( + <> +
+

+ {ov && !ov.online + ? bot 未连接,群名显示不出来 + : '白名单:列表里的群才会解析'} +

+ +
+ {err &&

{err}

} + + {ov && ( + <> + + 群策略(白名单:列表里的群才会解析) + + + + + + + + + + + {ov.groups.map(r => ( + + + + + + + + + + + + ))} + {!ov.groups.length && ( + + )} + +
群自动解析存储公网链接群文件自动策略禁用策略
+ {r.group_name ? `${r.group_name}(${r.group_id})` : r.group_id} + {!r.online && 不在群列表} + {onOff(r.policy.auto)}{r.policy.plan}{onOff(r.policy.upload_public)} + {r.policy.sends_link ? '开' : (r.policy.send_link ? '开·无公网' : '关')} + {groupFileText(r.policy)}{r.policy.auto_link.join('、') || '—'}{r.policy.ban_link.join('、') || '—'} + + +
还没有配置任何群,点右上角「添加群」
+
+
+ + + + + )} + + + + + + + + {editing?.isNew ? '添加群' : `群 ${editing?.group_id} 策略`} + + + + + {editing && ov && ( +
+ {editing.isNew && ( +
+ {botGroups.length > 0 ? ( + + ) : ( + + + setEditing({ ...editing, group_id: e.target.value.trim() })} /> + + )} +
+ )} + setEditing({ ...editing, policy: p })} + /> +
+ )} +
+ + + + +
+
+
+
+ + ) +} + + +function PolicyEditor({ + value, platforms, plans, onChange, +}: { + value: Policy + platforms: string[] + plans: string[] + onChange: (p: Policy) => void +}) { + const set = (patch: Partial) => onChange({ ...value, ...patch }) + return ( +
+ set({ auto: v })}> + 自动解析(群里出现链接就解析,无需 @bot) + + +
+
+ +
+ set({ upload_public: v })}> + 上传公网 + +
+ +
+

自动策略 (命中平台即使关掉自动解析也会解析)

+ set({ auto_link: v })} /> +
+ +
+

禁用策略 (命中平台自动/手动都不解析)

+ set({ ban_link: v })} /> +
+ +
+ set({ send_link: v })}> + 发送下载链接 + + {!value.upload_public &&

没上传公网时链接指向局域网 S3,群友打不开,所以这一项需要先开「上传公网」

} + set({ upload_group_file: v })}> + 同时上传群文件 + + {value.upload_group_file && ( +
+

+ 限定平台 (一个都不选 = 所有平台都传;选了则其余平台只发消息、不传群文件) +

+ set({ group_file_platforms: v })} + /> +
+ )} +
+
+ ) +} + + +function PlatformChips({ + value, platforms, tone, onChange, +}: { + value: string[] + platforms: string[] + tone: 'auto' | 'ban' | 'file' + onChange: (v: string[]) => void +}) { + const activeCls = tone === 'ban' + ? 'bg-red-50 text-red-600 border-red-300' + : tone === 'file' + ? 'bg-indigo-50 text-indigo-700 border-indigo-300' + : 'bg-emerald-50 text-emerald-700 border-emerald-300' + return ( +
+ {platforms.map(p => { + const active = value.includes(p) + return ( + + ) + })} +
+ ) +} + + +function DefaultCard({ ov, onSaved }: { ov: Overview; onSaved: () => Promise }) { + const [policy, setPolicy] = useState(ov.default) + const [busy, setBusy] = useState(false) + useEffect(() => { setPolicy(ov.default) }, [ov.default]) + + const save = async () => { + setBusy(true) + try { await va.post('default', { policy }); toast.success('默认策略已保存'); await onSaved() } + catch (e: any) { toast.danger(e.message || '保存失败') } finally { setBusy(false) } + } + + return ( + + + 默认策略 私聊与未配置群使用 + + + + + + + ) +} + + +function BlacklistCard({ ov, onSaved }: { ov: Overview; onSaved: () => Promise }) { + const [text, setText] = useState(ov.blacklist.join('\n')) + const [busy, setBusy] = useState(false) + useEffect(() => { setText(ov.blacklist.join('\n')) }, [ov.blacklist]) + + const save = async () => { + const list = text.split(/[\n,,\s]+/).map(x => x.trim()).filter(Boolean) + setBusy(true) + try { await va.post('blacklist', { blacklist: list }); toast.success('黑名单已保存'); await onSaved() } + catch (e: any) { toast.danger(e.message || '保存失败') } finally { setBusy(false) } + } + + return ( + + + 全局黑名单 这些 QQ 在所有群都不解析 + + + + +