- policy.py:per-group 正交策略(自动解析 / 自动策略 / 禁用策略 / 存储 A·B·C /
公网 / 链接 / 群文件 + 平台限定),list.json v1/v2 → v3 自动迁移,
写入统一走 PolicyStore(加锁 + .tmp 原子替换 + 字段归一)
- 群文件并行通道 group_file.py:打包 zip(可选 pyzipper AES-256)后优先走 S3 预签名、
本地直传兜底;设了密码但 pyzipper 不可用就放弃上传,不退化成明文
- list_proc.py 收敛到「视频策略」统一入口,权限判定改走 policy
- Web 管理页 /hub/video_analysis(群策略 + 链接解析面板)与 services/web_jobs.py
(只复用纯函数层,Web 上下文不发消息;内存任务表 + 并发闸门 + 超时)
- 媒体命名统一到 utils.py({作者}_{作者id}/{作品名}[_短码]),cleanup 回收空目录
- 测试:policy / 命名 / 群文件 / web_jobs 四组
顺带 pyproject 的 pytest 加 testpaths=tests(避免收进 debug/ 下的调试脚本)。
Co-Authored-By: Claude Code <noreply@anthropic.com>
215 lines
8.2 KiB
Python
215 lines
8.2 KiB
Python
"""媒体命名 / S3 key (utils.py 纯函数) 单元测试
|
||
|
||
utils.py 只依赖标准库,故用 importlib 按文件路径裸加载(与
|
||
test_video_group_file.py 同法)。media_key_of / media_rel_dir_of 依赖
|
||
get_temp_root 的返回值,测试里 monkeypatch 成临时目录。
|
||
"""
|
||
|
||
import importlib.util
|
||
import sys
|
||
from pathlib import Path
|
||
|
||
import pytest
|
||
|
||
_MODULE_PATH = (
|
||
Path(__file__).resolve().parents[1]
|
||
/ "hexi"
|
||
/ "plugins"
|
||
/ "nonebot_plugin_video_analysis"
|
||
/ "utils.py"
|
||
)
|
||
|
||
|
||
@pytest.fixture(scope="module")
|
||
def u():
|
||
"""以独立模块名加载 utils.py,避免与插件包 __init__ 冲突"""
|
||
spec = importlib.util.spec_from_file_location("video_utils_under_test", _MODULE_PATH)
|
||
module = importlib.util.module_from_spec(spec)
|
||
sys.modules[spec.name] = module
|
||
spec.loader.exec_module(module)
|
||
yield module
|
||
sys.modules.pop(spec.name, None)
|
||
|
||
|
||
@pytest.fixture
|
||
def temp_root(u, tmp_path, monkeypatch):
|
||
"""把 get_temp_root 指到 tmp_path(media_key_of 只做路径解析,不需 mkdir)"""
|
||
root = tmp_path / "temp"
|
||
root.mkdir()
|
||
monkeypatch.setattr(u, "get_temp_root", lambda sub="": root / sub if sub else root)
|
||
return root
|
||
|
||
|
||
# ───────────────────────── 作者目录 ─────────────────────────
|
||
|
||
|
||
def test_author_dir_with_id(u):
|
||
"""有作者 id 就是稳定的 `{昵称}_{id}`,不带码"""
|
||
assert u.build_author_dir("示例作者", "12345") == "示例作者_12345"
|
||
assert u.build_author_dir("示例作者", "12345", source="x", at=1) == (
|
||
"示例作者_12345"
|
||
)
|
||
|
||
|
||
def test_author_dir_nickname_only_gets_time_code(u):
|
||
"""只有昵称 → 追加 4 位时间短码:分不清同名作者,宁可不聚合也不混目录"""
|
||
at = 1_700_000_000
|
||
code = u.short_time_code(at)
|
||
assert u.build_author_dir("示例作者", "", at=at) == f"示例作者_{code}"
|
||
assert u.build_author_dir("示例作者", at=at) == f"示例作者_{code}"
|
||
for junk in ("0", "NA", "none", "None", " null "):
|
||
assert u.build_author_dir("示例作者", junk, at=at) == f"示例作者_{code}"
|
||
|
||
# 不同时刻 = 不同目录(同一作者的不同作品不会互相覆盖)
|
||
other = u.build_author_dir("示例作者", at=at + 3600)
|
||
assert other != f"示例作者_{code}"
|
||
assert other.startswith("示例作者_") and len(other.split("_")[-1]) == 4
|
||
|
||
|
||
def test_author_dir_uses_source_code_when_anonymous(u):
|
||
"""连昵称都没有 → 未知作者 + 来源短码(同一来源稳定,不同来源不撞)"""
|
||
url = "https://www.douyin.com/video/7412345678901234567"
|
||
first = u.build_author_dir(None, None, source=url)
|
||
assert first.startswith("未知作者_")
|
||
assert first == u.build_author_dir("", "", source=url) # 同来源稳定
|
||
|
||
second = u.build_author_dir(None, None, source=url + "8")
|
||
assert second != first
|
||
|
||
# 没有来源时才退化成时间短码
|
||
at = 1_700_000_000
|
||
assert u.build_author_dir(None, None, at=at) == (
|
||
f"未知作者_{u.short_time_code(at)}"
|
||
)
|
||
|
||
|
||
def test_author_dir_treats_placeholder_nicknames_as_anonymous(u):
|
||
"""抓取层拿不到昵称时的兜底串不算作者名,走来源码"""
|
||
url = "https://www.bilibili.com/opus/123"
|
||
expected = u.build_author_dir(None, None, source=url)
|
||
for placeholder in ("未知作者", "小红书用户", "B站用户", "b站用户"):
|
||
assert u.build_author_dir(placeholder, "", source=url) == expected
|
||
# 纯 emoji 昵称 slugify 后为空 → 同样按匿名处理
|
||
assert u.build_author_dir("🎬🎬", "", source=url) == expected
|
||
|
||
|
||
def test_short_source_code_is_stable_base36(u):
|
||
url = "https://www.douyin.com/video/7412345678901234567"
|
||
code = u.short_source_code(url)
|
||
assert len(code) == 4 and code.isalnum()
|
||
assert code == u.short_source_code(url) # 可复现(不能用带盐的 hash())
|
||
assert code != u.short_source_code(url + "8")
|
||
|
||
|
||
def test_author_dir_sanitizes_illegal_chars(u):
|
||
"""Windows 非法字符不能进目录名(slugify 直接删掉,空白转 -)"""
|
||
assert u.build_author_dir("a/b:c*d?", "12 3") == "abcd_12-3"
|
||
assert "/" not in u.build_author_dir("a/b", "1/2")
|
||
|
||
|
||
def test_author_dir_dedupes_same_handle(u):
|
||
"""X 这类站点上传者名就是 handle,别产出 someone_someone"""
|
||
assert u.build_author_dir("@Someone", "@Someone") == "someone"
|
||
|
||
|
||
def test_author_dir_keeps_long_sec_uid_distinct(u):
|
||
"""sec_uid 公共前缀就有 17 字符,截断到 20 会让不同作者撞同一目录"""
|
||
first = "MS4wLjABAAAA" + "x" * 45
|
||
second = "MS4wLjABAAAA" + "y" * 45
|
||
assert u.build_author_dir("n", first) != u.build_author_dir("n", second)
|
||
|
||
|
||
# ───────────────────────── 作品名 ─────────────────────────
|
||
|
||
|
||
def test_work_stem_strips_hashtags_and_spaces(u):
|
||
assert u.build_work_stem("旅行 #随手拍 vlog") == "旅行-vlog"
|
||
|
||
|
||
def test_work_stem_empty_falls_back(u):
|
||
assert u.build_work_stem("") == "作品"
|
||
assert u.build_work_stem(None) == "作品"
|
||
assert u.build_work_stem("🎬") == "作品"
|
||
|
||
|
||
def test_work_stem_truncates(u):
|
||
assert len(u.build_work_stem("标题" * 20)) <= 15
|
||
|
||
|
||
# ───────────────────────── 短码与重名 ─────────────────────────
|
||
|
||
|
||
def test_short_time_code_is_four_char_base36(u):
|
||
assert u.short_time_code(0) == "0000"
|
||
code = u.short_time_code(1_700_000_000)
|
||
assert len(code) == 4 and code.isalnum()
|
||
# 相邻秒不同码
|
||
assert code != u.short_time_code(1_700_000_001)
|
||
|
||
|
||
def test_unique_media_path_without_collision(u, tmp_path):
|
||
target = tmp_path / "作品.mp4"
|
||
assert u.unique_media_path(target) == target
|
||
|
||
|
||
def test_unique_media_path_adds_code_then_index(u, tmp_path):
|
||
target = tmp_path / "作品.mp4"
|
||
target.write_bytes(b"x")
|
||
code = u.short_time_code(1_700_000_000)
|
||
|
||
second = u.unique_media_path(target, at=1_700_000_000)
|
||
assert second.name == f"作品_{code}.mp4"
|
||
|
||
second.write_bytes(b"x")
|
||
third = u.unique_media_path(target, at=1_700_000_000)
|
||
assert third.name == f"作品_{code}_2.mp4"
|
||
|
||
|
||
def test_unique_media_path_creates_author_dir(u, tmp_path):
|
||
"""落盘前才建作者目录:命名函数顺带确保父目录存在"""
|
||
target = tmp_path / "作者_123" / "作品.mp4"
|
||
assert u.unique_media_path(target) == target
|
||
assert target.parent.is_dir()
|
||
|
||
|
||
def test_unique_media_path_works_for_directories(u, tmp_path):
|
||
"""多图作品目录同名时同样加短码"""
|
||
note_dir = tmp_path / "作者_123" / "作品"
|
||
note_dir.mkdir(parents=True)
|
||
renamed = u.unique_media_path(note_dir, at=1_700_000_000)
|
||
assert renamed.name == f"作品_{u.short_time_code(1_700_000_000)}"
|
||
|
||
|
||
# ───────────────────────── S3 key ─────────────────────────
|
||
|
||
|
||
def test_media_key_keeps_author_dir_and_drops_platform(u, temp_root):
|
||
f = temp_root / "douyin" / "作者_123" / "作品.mp4"
|
||
assert u.media_key_of(f) == "作者_123/作品.mp4"
|
||
assert u.media_rel_dir_of(f) == "作者_123"
|
||
|
||
|
||
def test_media_key_of_multi_image_work(u, temp_root):
|
||
f = temp_root / "bilibili" / "作者_9" / "作品_ab12" / "001.jpg"
|
||
assert u.media_key_of(f) == "作者_9/作品_ab12/001.jpg"
|
||
assert u.media_rel_dir_of(f) == "作者_9/作品_ab12"
|
||
|
||
|
||
def test_media_key_of_archive_follows_same_rule(u, temp_root):
|
||
"""群文件 zip 落在 archive/{作者目录} 下,key 规则与媒体一致"""
|
||
f = temp_root / "archive" / "作者_9" / "作品_群文件_120606.zip"
|
||
assert u.media_key_of(f) == "作者_9/作品_群文件_120606.zip"
|
||
|
||
|
||
def test_media_key_of_outside_temp_falls_back_to_name(u, temp_root, tmp_path):
|
||
f = tmp_path / "elsewhere" / "x.mp4"
|
||
assert u.media_key_of(f) == "x.mp4"
|
||
assert u.media_rel_dir_of(f) == ""
|
||
|
||
|
||
def test_media_key_of_flat_file_under_platform(u, temp_root):
|
||
"""老数据/第三方产物:平台层下没有作者层 → 只用文件名"""
|
||
f = temp_root / "douyin" / "x.mp4"
|
||
assert u.media_key_of(f) == "x.mp4"
|
||
assert u.media_rel_dir_of(f) == ""
|