Files
HeXi/tests/test_video_media_naming.py
sansenhoshiandClaude Code 4badcfcf32 feat(video-analysis): 群策略 v3 / 群文件投递通道 / Web 管理页
- policy.py:per-group 正交策略(自动解析 / 自动策略 / 禁用策略 / 存储 A·B·C /
  公网 / 链接 / 群文件 + 平台限定),list.json v1/v2 → v3 自动迁移,
  写入统一走 PolicyStore(加锁 + .tmp 原子替换 + 字段归一)
- 群文件并行通道 group_file.py:打包 zip(可选 pyzipper AES-256)后优先走 S3 预签名、
  本地直传兜底;设了密码但 pyzipper 不可用就放弃上传,不退化成明文
- list_proc.py 收敛到「视频策略」统一入口,权限判定改走 policy
- Web 管理页 /hub/video_analysis(群策略 + 链接解析面板)与 services/web_jobs.py
  (只复用纯函数层,Web 上下文不发消息;内存任务表 + 并发闸门 + 超时)
- 媒体命名统一到 utils.py({作者}_{作者id}/{作品名}[_短码]),cleanup 回收空目录
- 测试:policy / 命名 / 群文件 / web_jobs 四组

顺带 pyproject 的 pytest 加 testpaths=tests(避免收进 debug/ 下的调试脚本)。

Co-Authored-By: Claude Code <noreply@anthropic.com>
2026-09-22 14:23:32 +08:00

215 lines
8.2 KiB
Python
Raw Permalink Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""媒体命名 / S3 key (utils.py 纯函数) 单元测试
utils.py 只依赖标准库,故用 importlib 按文件路径裸加载(与
test_video_group_file.py 同法)。media_key_of / media_rel_dir_of 依赖
get_temp_root 的返回值,测试里 monkeypatch 成临时目录。
"""
import importlib.util
import sys
from pathlib import Path
import pytest
_MODULE_PATH = (
Path(__file__).resolve().parents[1]
/ "hexi"
/ "plugins"
/ "nonebot_plugin_video_analysis"
/ "utils.py"
)
@pytest.fixture(scope="module")
def u():
"""以独立模块名加载 utils.py,避免与插件包 __init__ 冲突"""
spec = importlib.util.spec_from_file_location("video_utils_under_test", _MODULE_PATH)
module = importlib.util.module_from_spec(spec)
sys.modules[spec.name] = module
spec.loader.exec_module(module)
yield module
sys.modules.pop(spec.name, None)
@pytest.fixture
def temp_root(u, tmp_path, monkeypatch):
"""把 get_temp_root 指到 tmp_path(media_key_of 只做路径解析,不需 mkdir)"""
root = tmp_path / "temp"
root.mkdir()
monkeypatch.setattr(u, "get_temp_root", lambda sub="": root / sub if sub else root)
return root
# ───────────────────────── 作者目录 ─────────────────────────
def test_author_dir_with_id(u):
"""有作者 id 就是稳定的 `{昵称}_{id}`,不带码"""
assert u.build_author_dir("示例作者", "12345") == "示例作者_12345"
assert u.build_author_dir("示例作者", "12345", source="x", at=1) == (
"示例作者_12345"
)
def test_author_dir_nickname_only_gets_time_code(u):
"""只有昵称 → 追加 4 位时间短码:分不清同名作者,宁可不聚合也不混目录"""
at = 1_700_000_000
code = u.short_time_code(at)
assert u.build_author_dir("示例作者", "", at=at) == f"示例作者_{code}"
assert u.build_author_dir("示例作者", at=at) == f"示例作者_{code}"
for junk in ("0", "NA", "none", "None", " null "):
assert u.build_author_dir("示例作者", junk, at=at) == f"示例作者_{code}"
# 不同时刻 = 不同目录(同一作者的不同作品不会互相覆盖)
other = u.build_author_dir("示例作者", at=at + 3600)
assert other != f"示例作者_{code}"
assert other.startswith("示例作者_") and len(other.split("_")[-1]) == 4
def test_author_dir_uses_source_code_when_anonymous(u):
"""连昵称都没有 → 未知作者 + 来源短码(同一来源稳定,不同来源不撞)"""
url = "https://www.douyin.com/video/7412345678901234567"
first = u.build_author_dir(None, None, source=url)
assert first.startswith("未知作者_")
assert first == u.build_author_dir("", "", source=url) # 同来源稳定
second = u.build_author_dir(None, None, source=url + "8")
assert second != first
# 没有来源时才退化成时间短码
at = 1_700_000_000
assert u.build_author_dir(None, None, at=at) == (
f"未知作者_{u.short_time_code(at)}"
)
def test_author_dir_treats_placeholder_nicknames_as_anonymous(u):
"""抓取层拿不到昵称时的兜底串不算作者名,走来源码"""
url = "https://www.bilibili.com/opus/123"
expected = u.build_author_dir(None, None, source=url)
for placeholder in ("未知作者", "小红书用户", "B站用户", "b站用户"):
assert u.build_author_dir(placeholder, "", source=url) == expected
# 纯 emoji 昵称 slugify 后为空 → 同样按匿名处理
assert u.build_author_dir("🎬🎬", "", source=url) == expected
def test_short_source_code_is_stable_base36(u):
url = "https://www.douyin.com/video/7412345678901234567"
code = u.short_source_code(url)
assert len(code) == 4 and code.isalnum()
assert code == u.short_source_code(url) # 可复现(不能用带盐的 hash())
assert code != u.short_source_code(url + "8")
def test_author_dir_sanitizes_illegal_chars(u):
"""Windows 非法字符不能进目录名(slugify 直接删掉,空白转 -)"""
assert u.build_author_dir("a/b:c*d?", "12 3") == "abcd_12-3"
assert "/" not in u.build_author_dir("a/b", "1/2")
def test_author_dir_dedupes_same_handle(u):
"""X 这类站点上传者名就是 handle,别产出 someone_someone"""
assert u.build_author_dir("@Someone", "@Someone") == "someone"
def test_author_dir_keeps_long_sec_uid_distinct(u):
"""sec_uid 公共前缀就有 17 字符,截断到 20 会让不同作者撞同一目录"""
first = "MS4wLjABAAAA" + "x" * 45
second = "MS4wLjABAAAA" + "y" * 45
assert u.build_author_dir("n", first) != u.build_author_dir("n", second)
# ───────────────────────── 作品名 ─────────────────────────
def test_work_stem_strips_hashtags_and_spaces(u):
assert u.build_work_stem("旅行 #随手拍 vlog") == "旅行-vlog"
def test_work_stem_empty_falls_back(u):
assert u.build_work_stem("") == "作品"
assert u.build_work_stem(None) == "作品"
assert u.build_work_stem("🎬") == "作品"
def test_work_stem_truncates(u):
assert len(u.build_work_stem("标题" * 20)) <= 15
# ───────────────────────── 短码与重名 ─────────────────────────
def test_short_time_code_is_four_char_base36(u):
assert u.short_time_code(0) == "0000"
code = u.short_time_code(1_700_000_000)
assert len(code) == 4 and code.isalnum()
# 相邻秒不同码
assert code != u.short_time_code(1_700_000_001)
def test_unique_media_path_without_collision(u, tmp_path):
target = tmp_path / "作品.mp4"
assert u.unique_media_path(target) == target
def test_unique_media_path_adds_code_then_index(u, tmp_path):
target = tmp_path / "作品.mp4"
target.write_bytes(b"x")
code = u.short_time_code(1_700_000_000)
second = u.unique_media_path(target, at=1_700_000_000)
assert second.name == f"作品_{code}.mp4"
second.write_bytes(b"x")
third = u.unique_media_path(target, at=1_700_000_000)
assert third.name == f"作品_{code}_2.mp4"
def test_unique_media_path_creates_author_dir(u, tmp_path):
"""落盘前才建作者目录:命名函数顺带确保父目录存在"""
target = tmp_path / "作者_123" / "作品.mp4"
assert u.unique_media_path(target) == target
assert target.parent.is_dir()
def test_unique_media_path_works_for_directories(u, tmp_path):
"""多图作品目录同名时同样加短码"""
note_dir = tmp_path / "作者_123" / "作品"
note_dir.mkdir(parents=True)
renamed = u.unique_media_path(note_dir, at=1_700_000_000)
assert renamed.name == f"作品_{u.short_time_code(1_700_000_000)}"
# ───────────────────────── S3 key ─────────────────────────
def test_media_key_keeps_author_dir_and_drops_platform(u, temp_root):
f = temp_root / "douyin" / "作者_123" / "作品.mp4"
assert u.media_key_of(f) == "作者_123/作品.mp4"
assert u.media_rel_dir_of(f) == "作者_123"
def test_media_key_of_multi_image_work(u, temp_root):
f = temp_root / "bilibili" / "作者_9" / "作品_ab12" / "001.jpg"
assert u.media_key_of(f) == "作者_9/作品_ab12/001.jpg"
assert u.media_rel_dir_of(f) == "作者_9/作品_ab12"
def test_media_key_of_archive_follows_same_rule(u, temp_root):
"""群文件 zip 落在 archive/{作者目录} 下,key 规则与媒体一致"""
f = temp_root / "archive" / "作者_9" / "作品_群文件_120606.zip"
assert u.media_key_of(f) == "作者_9/作品_群文件_120606.zip"
def test_media_key_of_outside_temp_falls_back_to_name(u, temp_root, tmp_path):
f = tmp_path / "elsewhere" / "x.mp4"
assert u.media_key_of(f) == "x.mp4"
assert u.media_rel_dir_of(f) == ""
def test_media_key_of_flat_file_under_platform(u, temp_root):
"""老数据/第三方产物:平台层下没有作者层 → 只用文件名"""
f = temp_root / "douyin" / "x.mp4"
assert u.media_key_of(f) == "x.mp4"
assert u.media_rel_dir_of(f) == ""