215 lines
8.2 KiB
Python
215 lines
8.2 KiB
Python
"""媒体命名 / S3 key (utils.py 纯函数) 单元测试
|
||||
|
|
|
|||
|
|
utils.py 只依赖标准库,故用 importlib 按文件路径裸加载(与
|
|||
|
|
test_video_group_file.py 同法)。media_key_of / media_rel_dir_of 依赖
|
|||
|
|
get_temp_root 的返回值,测试里 monkeypatch 成临时目录。
|
|||
|
|
"""
|
|||
|
|
|
|||
|
|
import importlib.util
|
|||
|
|
import sys
|
|||
|
|
from pathlib import Path
|
|||
|
|
|
|||
|
|
import pytest
|
|||
|
|
|
|||
|
|
_MODULE_PATH = (
|
|||
|
|
Path(__file__).resolve().parents[1]
|
|||
|
|
/ "hexi"
|
|||
|
|
/ "plugins"
|
|||
|
|
/ "nonebot_plugin_video_analysis"
|
|||
|
|
/ "utils.py"
|
|||
|
|
)
|
|||
|
|
|
|||
|
|
|
|||
|
|
@pytest.fixture(scope="module")
|
|||
|
|
def u():
|
|||
|
|
"""以独立模块名加载 utils.py,避免与插件包 __init__ 冲突"""
|
|||
|
|
spec = importlib.util.spec_from_file_location("video_utils_under_test", _MODULE_PATH)
|
|||
|
|
module = importlib.util.module_from_spec(spec)
|
|||
|
|
sys.modules[spec.name] = module
|
|||
|
|
spec.loader.exec_module(module)
|
|||
|
|
yield module
|
|||
|
|
sys.modules.pop(spec.name, None)
|
|||
|
|
|
|||
|
|
|
|||
|
|
@pytest.fixture
|
|||
|
|
def temp_root(u, tmp_path, monkeypatch):
|
|||
|
|
"""把 get_temp_root 指到 tmp_path(media_key_of 只做路径解析,不需 mkdir)"""
|
|||
|
|
root = tmp_path / "temp"
|
|||
|
|
root.mkdir()
|
|||
|
|
monkeypatch.setattr(u, "get_temp_root", lambda sub="": root / sub if sub else root)
|
|||
|
|
return root
|
|||
|
|
|
|||
|
|
|
|||
|
|
# ───────────────────────── 作者目录 ─────────────────────────
|
|||
|
|
|
|||
|
|
|
|||
|
|
def test_author_dir_with_id(u):
|
|||
|
|
"""有作者 id 就是稳定的 `{昵称}_{id}`,不带码"""
|
|||
|
|
assert u.build_author_dir("示例作者", "12345") == "示例作者_12345"
|
|||
|
|
assert u.build_author_dir("示例作者", "12345", source="x", at=1) == (
|
|||
|
|
"示例作者_12345"
|
|||
|
|
)
|
|||
|
|
|
|||
|
|
|
|||
|
|
def test_author_dir_nickname_only_gets_time_code(u):
|
|||
|
|
"""只有昵称 → 追加 4 位时间短码:分不清同名作者,宁可不聚合也不混目录"""
|
|||
|
|
at = 1_700_000_000
|
|||
|
|
code = u.short_time_code(at)
|
|||
|
|
assert u.build_author_dir("示例作者", "", at=at) == f"示例作者_{code}"
|
|||
|
|
assert u.build_author_dir("示例作者", at=at) == f"示例作者_{code}"
|
|||
|
|
for junk in ("0", "NA", "none", "None", " null "):
|
|||
|
|
assert u.build_author_dir("示例作者", junk, at=at) == f"示例作者_{code}"
|
|||
|
|
|
|||
|
|
# 不同时刻 = 不同目录(同一作者的不同作品不会互相覆盖)
|
|||
|
|
other = u.build_author_dir("示例作者", at=at + 3600)
|
|||
|
|
assert other != f"示例作者_{code}"
|
|||
|
|
assert other.startswith("示例作者_") and len(other.split("_")[-1]) == 4
|
|||
|
|
|
|||
|
|
|
|||
|
|
def test_author_dir_uses_source_code_when_anonymous(u):
|
|||
|
|
"""连昵称都没有 → 未知作者 + 来源短码(同一来源稳定,不同来源不撞)"""
|
|||
|
|
url = "https://www.douyin.com/video/7412345678901234567"
|
|||
|
|
first = u.build_author_dir(None, None, source=url)
|
|||
|
|
assert first.startswith("未知作者_")
|
|||
|
|
assert first == u.build_author_dir("", "", source=url) # 同来源稳定
|
|||
|
|
|
|||
|
|
second = u.build_author_dir(None, None, source=url + "8")
|
|||
|
|
assert second != first
|
|||
|
|
|
|||
|
|
# 没有来源时才退化成时间短码
|
|||
|
|
at = 1_700_000_000
|
|||
|
|
assert u.build_author_dir(None, None, at=at) == (
|
|||
|
|
f"未知作者_{u.short_time_code(at)}"
|
|||
|
|
)
|
|||
|
|
|
|||
|
|
|
|||
|
|
def test_author_dir_treats_placeholder_nicknames_as_anonymous(u):
|
|||
|
|
"""抓取层拿不到昵称时的兜底串不算作者名,走来源码"""
|
|||
|
|
url = "https://www.bilibili.com/opus/123"
|
|||
|
|
expected = u.build_author_dir(None, None, source=url)
|
|||
|
|
for placeholder in ("未知作者", "小红书用户", "B站用户", "b站用户"):
|
|||
|
|
assert u.build_author_dir(placeholder, "", source=url) == expected
|
|||
|
|
# 纯 emoji 昵称 slugify 后为空 → 同样按匿名处理
|
|||
|
|
assert u.build_author_dir("🎬🎬", "", source=url) == expected
|
|||
|
|
|
|||
|
|
|
|||
|
|
def test_short_source_code_is_stable_base36(u):
|
|||
|
|
url = "https://www.douyin.com/video/7412345678901234567"
|
|||
|
|
code = u.short_source_code(url)
|
|||
|
|
assert len(code) == 4 and code.isalnum()
|
|||
|
|
assert code == u.short_source_code(url) # 可复现(不能用带盐的 hash())
|
|||
|
|
assert code != u.short_source_code(url + "8")
|
|||
|
|
|
|||
|
|
|
|||
|
|
def test_author_dir_sanitizes_illegal_chars(u):
|
|||
|
|
"""Windows 非法字符不能进目录名(slugify 直接删掉,空白转 -)"""
|
|||
|
|
assert u.build_author_dir("a/b:c*d?", "12 3") == "abcd_12-3"
|
|||
|
|
assert "/" not in u.build_author_dir("a/b", "1/2")
|
|||
|
|
|
|||
|
|
|
|||
|
|
def test_author_dir_dedupes_same_handle(u):
|
|||
|
|
"""X 这类站点上传者名就是 handle,别产出 someone_someone"""
|
|||
|
|
assert u.build_author_dir("@Someone", "@Someone") == "someone"
|
|||
|
|
|
|||
|
|
|
|||
|
|
def test_author_dir_keeps_long_sec_uid_distinct(u):
|
|||
|
|
"""sec_uid 公共前缀就有 17 字符,截断到 20 会让不同作者撞同一目录"""
|
|||
|
|
first = "MS4wLjABAAAA" + "x" * 45
|
|||
|
|
second = "MS4wLjABAAAA" + "y" * 45
|
|||
|
|
assert u.build_author_dir("n", first) != u.build_author_dir("n", second)
|
|||
|
|
|
|||
|
|
|
|||
|
|
# ───────────────────────── 作品名 ─────────────────────────
|
|||
|
|
|
|||
|
|
|
|||
|
|
def test_work_stem_strips_hashtags_and_spaces(u):
|
|||
|
|
assert u.build_work_stem("旅行 #随手拍 vlog") == "旅行-vlog"
|
|||
|
|
|
|||
|
|
|
|||
|
|
def test_work_stem_empty_falls_back(u):
|
|||
|
|
assert u.build_work_stem("") == "作品"
|
|||
|
|
assert u.build_work_stem(None) == "作品"
|
|||
|
|
assert u.build_work_stem("🎬") == "作品"
|
|||
|
|
|
|||
|
|
|
|||
|
|
def test_work_stem_truncates(u):
|
|||
|
|
assert len(u.build_work_stem("标题" * 20)) <= 15
|
|||
|
|
|
|||
|
|
|
|||
|
|
# ───────────────────────── 短码与重名 ─────────────────────────
|
|||
|
|
|
|||
|
|
|
|||
|
|
def test_short_time_code_is_four_char_base36(u):
|
|||
|
|
assert u.short_time_code(0) == "0000"
|
|||
|
|
code = u.short_time_code(1_700_000_000)
|
|||
|
|
assert len(code) == 4 and code.isalnum()
|
|||
|
|
# 相邻秒不同码
|
|||
|
|
assert code != u.short_time_code(1_700_000_001)
|
|||
|
|
|
|||
|
|
|
|||
|
|
def test_unique_media_path_without_collision(u, tmp_path):
|
|||
|
|
target = tmp_path / "作品.mp4"
|
|||
|
|
assert u.unique_media_path(target) == target
|
|||
|
|
|
|||
|
|
|
|||
|
|
def test_unique_media_path_adds_code_then_index(u, tmp_path):
|
|||
|
|
target = tmp_path / "作品.mp4"
|
|||
|
|
target.write_bytes(b"x")
|
|||
|
|
code = u.short_time_code(1_700_000_000)
|
|||
|
|
|
|||
|
|
second = u.unique_media_path(target, at=1_700_000_000)
|
|||
|
|
assert second.name == f"作品_{code}.mp4"
|
|||
|
|
|
|||
|
|
second.write_bytes(b"x")
|
|||
|
|
third = u.unique_media_path(target, at=1_700_000_000)
|
|||
|
|
assert third.name == f"作品_{code}_2.mp4"
|
|||
|
|
|
|||
|
|
|
|||
|
|
def test_unique_media_path_creates_author_dir(u, tmp_path):
|
|||
|
|
"""落盘前才建作者目录:命名函数顺带确保父目录存在"""
|
|||
|
|
target = tmp_path / "作者_123" / "作品.mp4"
|
|||
|
|
assert u.unique_media_path(target) == target
|
|||
|
|
assert target.parent.is_dir()
|
|||
|
|
|
|||
|
|
|
|||
|
|
def test_unique_media_path_works_for_directories(u, tmp_path):
|
|||
|
|
"""多图作品目录同名时同样加短码"""
|
|||
|
|
note_dir = tmp_path / "作者_123" / "作品"
|
|||
|
|
note_dir.mkdir(parents=True)
|
|||
|
|
renamed = u.unique_media_path(note_dir, at=1_700_000_000)
|
|||
|
|
assert renamed.name == f"作品_{u.short_time_code(1_700_000_000)}"
|
|||
|
|
|
|||
|
|
|
|||
|
|
# ───────────────────────── S3 key ─────────────────────────
|
|||
|
|
|
|||
|
|
|
|||
|
|
def test_media_key_keeps_author_dir_and_drops_platform(u, temp_root):
|
|||
|
|
f = temp_root / "douyin" / "作者_123" / "作品.mp4"
|
|||
|
|
assert u.media_key_of(f) == "作者_123/作品.mp4"
|
|||
|
|
assert u.media_rel_dir_of(f) == "作者_123"
|
|||
|
|
|
|||
|
|
|
|||
|
|
def test_media_key_of_multi_image_work(u, temp_root):
|
|||
|
|
f = temp_root / "bilibili" / "作者_9" / "作品_ab12" / "001.jpg"
|
|||
|
|
assert u.media_key_of(f) == "作者_9/作品_ab12/001.jpg"
|
|||
|
|
assert u.media_rel_dir_of(f) == "作者_9/作品_ab12"
|
|||
|
|
|
|||
|
|
|
|||
|
|
def test_media_key_of_archive_follows_same_rule(u, temp_root):
|
|||
|
|
"""群文件 zip 落在 archive/{作者目录} 下,key 规则与媒体一致"""
|
|||
|
|
f = temp_root / "archive" / "作者_9" / "作品_群文件_120606.zip"
|
|||
|
|
assert u.media_key_of(f) == "作者_9/作品_群文件_120606.zip"
|
|||
|
|
|
|||
|
|
|
|||
|
|
def test_media_key_of_outside_temp_falls_back_to_name(u, temp_root, tmp_path):
|
|||
|
|
f = tmp_path / "elsewhere" / "x.mp4"
|
|||
|
|
assert u.media_key_of(f) == "x.mp4"
|
|||
|
|
assert u.media_rel_dir_of(f) == ""
|
|||
|
|
|
|||
|
|
|
|||
|
|
def test_media_key_of_flat_file_under_platform(u, temp_root):
|
|||
|
|
"""老数据/第三方产物:平台层下没有作者层 → 只用文件名"""
|
|||
|
|
f = temp_root / "douyin" / "x.mp4"
|
|||
|
|
assert u.media_key_of(f) == "x.mp4"
|
|||
|
|
assert u.media_rel_dir_of(f) == ""
|