feat(video-analysis): 群策略 v3 / 群文件投递通道 / Web 管理页

- policy.py:per-group 正交策略(自动解析 / 自动策略 / 禁用策略 / 存储 A·B·C /
  公网 / 链接 / 群文件 + 平台限定),list.json v1/v2 → v3 自动迁移,
  写入统一走 PolicyStore(加锁 + .tmp 原子替换 + 字段归一)
- 群文件并行通道 group_file.py:打包 zip(可选 pyzipper AES-256)后优先走 S3 预签名、
  本地直传兜底;设了密码但 pyzipper 不可用就放弃上传,不退化成明文
- list_proc.py 收敛到「视频策略」统一入口,权限判定改走 policy
- Web 管理页 /hub/video_analysis(群策略 + 链接解析面板)与 services/web_jobs.py
  (只复用纯函数层,Web 上下文不发消息;内存任务表 + 并发闸门 + 超时)
- 媒体命名统一到 utils.py({作者}_{作者id}/{作品名}[_短码]),cleanup 回收空目录
- 测试:policy / 命名 / 群文件 / web_jobs 四组

顺带 pyproject 的 pytest 加 testpaths=tests(避免收进 debug/ 下的调试脚本)。

Co-Authored-By: Claude Code <noreply@anthropic.com>
This commit is contained in:
2026-09-22 14:23:32 +08:00
co-authored by Claude Code
parent 51b08ccb68
commit 4badcfcf32
29 changed files with 4838 additions and 670 deletions
@@ -0,0 +1,271 @@
"""群文件上传 —— 与消息发送并行的第二条投递通道。
策略开关见 `policy.Policy.upload_group_file`:开着的时候,媒体照常发到群里,
同时另传一份到群文件(群友可随时下载、不占聊天记录)。
投递前会按全局配置打包(`video_analysis_group_file_zip`)成一个 zip:
配了解压密码(`video_analysis_group_file_password`)就用 AES-256 加密,
**密码设置了但 pyzipper 不可用时直接放弃上传,绝不退化成传明文**。
(无密码的普通 zip 走标准库,零依赖。)
失败只记日志(bot 没有群文件权限、超出群文件大小上限等都属于预期内的失败),
不影响消息发送链路;上传成功后不额外发消息,避免刷屏。
"""
from __future__ import annotations
import asyncio
import re
import zipfile
from collections.abc import Iterable
from datetime import datetime
from pathlib import Path
from typing import TYPE_CHECKING, Any
from nonebot import get_bot, logger
if TYPE_CHECKING: # 仅类型检查:运行期不导入,单测可裸加载本模块
from ..policy import Policy
try: # 缺失时只影响「加密打包」这一路,见 build_archive
import pyzipper
except ImportError: # pragma: no cover
pyzipper = None # type: ignore[assignment]
def encryption_available() -> bool:
"""加密打包是否可用(pyzipper 已安装)。"""
return pyzipper is not None
def build_archive(
files: Iterable[Path | str],
title: str = "",
*,
password: str = "",
out_dir: Path | None = None,
rel_dir: str = "",
) -> Path:
"""把文件打包成一个 zip,返回产物路径;password 非空则 AES-256 加密。
out_dir 缺省落 `hexi/data/temp/archive/{作者目录}`(rel_dir 由调用方传入,
与源媒体同一套目录结构,见 sender.media_rel_dir_of),由 cleanup 按天清理。
加密需要 pyzipper,缺失时抛 RuntimeError —— 调用方应当**放弃上传**。
"""
paths = [Path(f) for f in files]
if not paths:
raise ValueError("没有可打包的文件")
missing = [p for p in paths if not p.exists()]
if missing:
raise FileNotFoundError(f"待打包文件不存在: {missing[0]}")
if password and pyzipper is None:
raise RuntimeError("配置了解压密码,但 pyzipper 未安装,无法加密打包")
archive = _resolve_target(out_dir, title, rel_dir)
if password:
with pyzipper.AESZipFile(
archive,
"w",
compression=pyzipper.ZIP_DEFLATED,
encryption=pyzipper.WZ_AES,
) as zf:
zf.setpassword(password.encode("utf-8"))
_write_members(zf, paths)
else:
with zipfile.ZipFile(archive, "w", zipfile.ZIP_DEFLATED) as zf:
_write_members(zf, paths)
logger.info(
f"群文件打包完成: {archive.name}({len(paths)} 个文件"
f"{',已加密' if password else ''})"
)
return archive
def _default_archive_dir() -> Path:
"""默认落 temp/archive(延迟导入 utils:单测裸加载本模块时没有包上下文)。"""
from ...utils import get_temp_root
return get_temp_root("archive")
def _safe_stem(title: str, max_length: int = 15) -> str:
"""标题 → 安全的文件名片段。
原始标题(抖音文案、YouTube 标题)可能带 `/`、换行、控制字符与 `#话题`,
这些进不了文件名,统一换成 `_` 后截断。
"""
cleaned = re.sub(r"#\S+", "", title)
cleaned = re.sub(r'[\\/:*?"<>|\s]+', "_", cleaned.strip())
cleaned = re.sub(r"_+", "_", cleaned).strip("_")
return cleaned[:max_length].strip("_")
def _resolve_target(out_dir: Path | None, title: str, rel_dir: str = "") -> Path:
"""产物最终路径:默认 temp/archive/{rel_dir},重名自动加序号。
空标题 / 占位符 / 已带「群文件」前缀的一律回退成默认名。
"""
base = Path(out_dir) if out_dir is not None else _default_archive_dir()
if rel_dir:
base = base / rel_dir
base.mkdir(parents=True, exist_ok=True)
stamp = f"{datetime.now():%H%M%S}"
stem = _safe_stem(title) if title else ""
if not stem or stem in ("title", "video") or stem.startswith("群文件"):
return _unique_path(base / f"群文件_{stamp}.zip")
return _unique_path(base / f"{stem}_群文件_{stamp}.zip")
def _unique_path(path: Path) -> Path:
"""重名追加 _2/_3…(同 utils.ensure_unique_path 的语义,就地实现)。"""
if not path.exists():
return path
index = 2
while True:
candidate = path.with_name(f"{path.stem}_{index}{path.suffix}")
if not candidate.exists():
return candidate
index += 1
def _write_members(zf: Any, paths: list[Path]) -> None:
"""写入成员:同名文件自动加序号,避免在包里互相覆盖。"""
used: set[str] = set()
for path in paths:
name = path.name
if name in used:
index = 2
while f"{path.stem}_{index}{path.suffix}" in used:
index += 1
name = f"{path.stem}_{index}{path.suffix}"
used.add(name)
zf.write(path, arcname=name)
def _file_uri(path: Path) -> str:
"""本地文件 → `file:///D:/a/b.zip`(正斜杠,中文/空格不转义)。
与图片/视频消息发给 NapCat 的形式一致;直接传 Windows 反斜杠路径会被
它的 realpath 判成 ENOENT(见 upload_group_file 的降级说明)。
"""
return "file:///" + str(path.resolve()).replace("\\", "/")
async def _call_upload(
file_value: str, group_id: int, name: str, folder_id: str | None = None
) -> None:
params: dict = {"group_id": group_id, "file": file_value, "name": name}
if folder_id:
params["folder"] = folder_id
await get_bot().call_api("upload_group_file", **params)
def _s3_url(path: Path, policy: "Policy | None") -> str:
"""把文件传到局域网 S3 换预签名链接(延迟导入 s3,便于单测裸加载本模块)。"""
from .s3 import upload_with_plan
url, _ = upload_with_plan(path, policy=policy)
return url
async def upload_group_file(
file_path: Path | str,
group_id: int,
*,
folder_id: str | None = None,
policy: Policy | None = None,
name: str | None = None,
) -> bool:
"""上传单个文件到群文件,返回是否成功。
两级投递,**S3 链接优先、本地直传兜底**:
本环境实测 NapCat 读不到 bot 进程写的本地文件(`realpath ... ENOENT`,
媒体消息的本地直发 55 次全失败、换 S3 链接后次次成功),所以直传只留作
S3 不可用时的兜底。policy 决定 S3 落到哪个桶(plan)。
本地直传用 `file:///D:/…` 形式(正斜杠、不转义中文)——图片/视频消息就是
这么发的,同机部署时能work。
name 可覆盖群文件列表里显示的名字(多图作品的成员是 001.jpg,调用方会补上
作品名前缀,免得多张图在群文件里全叫 001.jpg)。
"""
path = Path(file_path)
if not path.exists():
logger.warning(f"群文件上传:文件不存在 {path}")
return False
display_name = name or path.name
# 1) 局域网 S3 预签名链接(boto3 同步,丢线程池)
url = await asyncio.to_thread(_s3_url, path, policy)
if url:
try:
await _call_upload(url, group_id, display_name, folder_id)
except Exception as e:
logger.warning(
f"群文件 S3 链接上传失败(群 {group_id} / {display_name}): "
f"{str(e)[:160]};改用本地路径重试"
)
else:
logger.info(f"群文件上传成功(群 {group_id},S3 链接): {display_name}")
return True
else:
logger.warning("群文件:S3 中转没拿到链接,改用本地路径直传")
# 2) 兜底:file:/// 直传本地文件
try:
await _call_upload(_file_uri(path), group_id, display_name, folder_id)
except Exception as e:
logger.warning(
f"群文件上传失败(群 {group_id} / {display_name}): {str(e)[:160]}"
)
return False
logger.info(f"群文件上传成功(群 {group_id},本地直传): {display_name}")
return True
async def upload_group_files(
files: Iterable[Path | str],
group_id: int,
*,
title: str = "",
zip_files: bool = True,
password: str = "",
policy: Policy | None = None,
rel_dir: str = "",
) -> bool:
"""群文件投递入口:按配置打包(可加密)后传一个包,或逐个传原文件。
rel_dir 是源媒体所在的 `{作者}_{作者id}[/{作品名}]` 子目录,打包产物落到
`temp/archive/{rel_dir}`(由调用方从文件路径推出来,见 sender)。
"""
paths = [Path(f) for f in files]
if not paths:
return False
if not zip_files:
# 多图作品逐个传时,成员名是 001.jpg…,补上作品名前缀便于在群文件里辨认
work = Path(rel_dir).name if rel_dir else ""
prefix = f"{work}_" if work and len(paths) > 1 else ""
results = [
await upload_group_file(
p, group_id, policy=policy, name=f"{prefix}{p.name}"
)
for p in paths
]
return any(results)
try:
archive = await asyncio.to_thread(
build_archive, paths, title, password=password, rel_dir=rel_dir
)
except Exception as e:
logger.error(f"群文件打包失败,跳过本次群文件上传: {e}")
return False
return await upload_group_file(archive, group_id, policy=policy)
@@ -1,7 +1,6 @@
"""统一 S3 存储模块 — 合并本地局域网 S3 和公网 MinIO"""
from pathlib import Path
from time import strftime, localtime
import boto3
from botocore.config import Config
@@ -9,6 +8,9 @@ from nonebot import logger
from hexi.web_hub.web_config import get_effective_value
from ...policy import Policy
from ...utils import media_key_of
# 该插件模块名,用于读取统一配置值库(Web 修改后生效)
_PLUGIN_ID = "hexi.plugins.nonebot_plugin_video_analysis"
@@ -171,11 +173,13 @@ def upload_to_public_s3(file_path: str | Path) -> tuple[str, str]:
"""
上传到公网 MinIO,返回 (公网URL, file_key)
私聊场景使用,生成可公网访问的链接
私聊场景使用,生成可公网访问的链接。
key 按本地落盘结构推导:`{作者}_{作者id}/{作品名}[_{短码}].ext`
(不再用 `{年月日}/` 前缀,见 utils.media_key_of)。
"""
file_path = Path(file_path)
current_day = strftime("%Y-%m-%d", localtime())
file_key = f"{current_day}/{file_path.name}"
file_key = media_key_of(file_path)
try:
url = _get_public_s3().upload_public(str(file_path), file_key)
return url, file_key
@@ -184,22 +188,19 @@ def upload_to_public_s3(file_path: str | Path) -> tuple[str, str]:
return "", ""
def upload_to_local_s3(
title: str, image_post: bool, file_path: str | Path, plan: str | None = None
) -> str:
def upload_to_local_s3(file_path: str | Path, plan: str | None = None) -> str:
"""
上传到局域网 S3,返回预签名 URL
plan="A" → PLANA 桶
plan="B" → PLANB 桶
None → PLANC 桶(默认)
key 与公网一致:`{作者}_{作者id}/{作品名}[_{短码}].ext`
(不再有 `{年月日}/` 前缀,也不再重复套一层标题目录)。
"""
file_path = Path(file_path)
current_day = strftime("%Y-%m-%d", localtime())
if image_post:
file_key = f"{current_day}/{title}/{file_path.name}"
else:
file_key = f"{current_day}/{file_path.name}"
file_key = media_key_of(file_path)
if plan == "A":
client = _get_local_s3_plana()
@@ -224,35 +225,24 @@ def delete_from_public_s3(file_key: str) -> bool:
def upload_with_plan(
file_path: str | Path,
*,
plan: str | None = None,
is_private: bool = False,
title: str = "",
image_post: bool = False,
policy: Policy | None = None,
) -> tuple[str, str | None]:
"""
统一上传入口:按 plan 路由到对应的本地桶,并按需上传公网
统一上传入口:按策略路由本地桶,并按需上传公网
plan="A" → PLANA(仅本地)
plan="B" → PLANB + 公网
私聊 → PLANB + 公网
默认 → PLANC(仅本地)
policy.plan → PLANA / PLANB / PLANC(默认 C)
policy.upload_public → 是否额外上传公网(决定能否发下载链接)
key 由文件路径推导(utils.media_key_of),调用方不再传标题。
Returns:
(local_url, public_url_or_none)
"""
# 本地上传
if plan == "A":
local_url = upload_to_local_s3(title, image_post, file_path, plan="A")
elif plan == "B":
local_url = upload_to_local_s3(title, image_post, file_path, plan="B")
elif is_private:
local_url = upload_to_local_s3(title, image_post, file_path, plan="B")
else:
local_url = upload_to_local_s3(title, image_post, file_path) # PLANC
policy = policy or Policy()
local_url = upload_to_local_s3(file_path, plan=policy.plan)
# 公网上传(仅 PLANB 或私聊)
public_url = None
if plan == "B" or is_private:
if policy.upload_public:
public_url, _ = upload_to_public_s3(file_path)
return local_url, public_url