feat(video-analysis): 群策略 v3 / 群文件投递通道 / Web 管理页

- policy.py:per-group 正交策略(自动解析 / 自动策略 / 禁用策略 / 存储 A·B·C /
  公网 / 链接 / 群文件 + 平台限定),list.json v1/v2 → v3 自动迁移,
  写入统一走 PolicyStore(加锁 + .tmp 原子替换 + 字段归一)
- 群文件并行通道 group_file.py:打包 zip(可选 pyzipper AES-256)后优先走 S3 预签名、
  本地直传兜底;设了密码但 pyzipper 不可用就放弃上传,不退化成明文
- list_proc.py 收敛到「视频策略」统一入口,权限判定改走 policy
- Web 管理页 /hub/video_analysis(群策略 + 链接解析面板)与 services/web_jobs.py
  (只复用纯函数层,Web 上下文不发消息;内存任务表 + 并发闸门 + 超时)
- 媒体命名统一到 utils.py({作者}_{作者id}/{作品名}[_短码]),cleanup 回收空目录
- 测试:policy / 命名 / 群文件 / web_jobs 四组

顺带 pyproject 的 pytest 加 testpaths=tests(避免收进 debug/ 下的调试脚本)。

Co-Authored-By: Claude Code <noreply@anthropic.com>
This commit is contained in:
2026-09-22 14:23:32 +08:00
co-authored by Claude Code
parent 51b08ccb68
commit 4badcfcf32
29 changed files with 4838 additions and 670 deletions
@@ -1,8 +1,9 @@
import hashlib
import re
import time
import unicodedata
from pathlib import Path
from time import strftime, localtime
from typing import List
from typing import List, Optional, Tuple
def get_data_dir() -> Path:
@@ -72,6 +73,9 @@ def ensure_unique_path(base_path: Path) -> Path:
"""
确保路径不冲突:如已存在则追加 _2, _3... 后缀
适用于文件和目录
注:媒体落盘已统一走 `unique_media_path`(同名加 4 位短码),本函数仅作
通用兜底保留。
"""
if not base_path.exists():
return base_path
@@ -88,52 +92,139 @@ def ensure_unique_path(base_path: Path) -> Path:
counter += 1
def clean_filename(filename: str, max_length: int = 120) -> str:
_B36_ALPHABET = "0123456789abcdefghijklmnopqrstuvwxyz"
#: 同名短码的模数 —— base36 四位(36**4 ≈ 168 万秒 ≈ 19.4 天一轮)
_B36_MOD = 36**4
def _to_base36(value: int, width: int = 4) -> str:
"""整数 → 定宽 base36(不足左侧补 0)"""
if value <= 0:
return "0" * width
digits = []
while value:
value, rem = divmod(value, 36)
digits.append(_B36_ALPHABET[rem])
return "".join(reversed(digits)).rjust(width, "0")
def short_time_code(at: Optional[float] = None) -> str:
"""4 位 base36 短码(同名兜底用,不可读时间,仅作区分码)"""
return _to_base36(int(time.time() if at is None else at) % _B36_MOD)
#: 各平台把"没有 id"写成过这些值,别让它们进目录名
_JUNK_AUTHOR_IDS = {"", "0", "none", "null", "na", "nan", "undefined"}
#: 抓取层拿不到昵称时的兜底串(slugify 后)—— 它们等于"没有作者信息"
_PLACEHOLDER_AUTHORS = {"未知作者", "小红书用户", "b站用户"}
def _clean_author_id(author_id: Optional[str]) -> str:
"""作者 id 归一:剔除占位值(yt-dlp 缺字段常给 `NA`),再 slugify。"""
raw = str(author_id or "").strip()
if raw.lower() in _JUNK_AUTHOR_IDS:
return ""
return slugify(raw, max_length=40)
def short_source_code(source: str) -> str:
"""来源串(作品 URL/id)→ 4 位 base36 短码。
与 `short_time_code` 的区别:同一个来源永远得到同一个码(md5 取摘要,
不能用内置 `hash()` —— 它有随机盐,重启后目录名会变)。
"""
清理文件名并添加时间前缀
支持多扩展名,如 .tar.gz
digest = hashlib.md5(source.encode("utf-8")).digest()
return _to_base36(int.from_bytes(digest[:4], "big") % _B36_MOD)
def build_author_dir(
author: Optional[str],
author_id: Optional[str] = None,
*,
source: Optional[str] = None,
at: Optional[float] = None,
) -> str:
"""`{作者}_{作者id}` 作者目录名(拿不到 id 时追加码值避免同名混目录)。
- 昵称 + 作者 id → `{昵称}_{id}`(昵称上限 30、id 上限 40;**不能**把 id 截到
20:`MS4wLjABAAAA…` 这类 sec_uid 公共前缀就有 17 字符,再截断必撞车)
- 只有昵称 → `{昵称}_{4 位时间短码}`:没有 id 就分不清同名作者,宁可不聚合
(同一作者的不同作品会各成一个目录)也不能把两个人混进同一个目录
- 连昵称都没有 → `未知作者_{来源短码}`(`source` 给作品 URL/id,同一来源
稳定、不同来源不撞);连 source 都没有 → 退化成时间短码
"""
slug_author = slugify(author or "", max_length=30)
if slug_author in _PLACEHOLDER_AUTHORS:
slug_author = ""
slug_id = _clean_author_id(author_id)
current_time = strftime("%H-%M-%S", localtime())
if slug_author and slug_id:
# 某些站点上传者名就是 handle(X 的 @someone),别产出 someone_someone
if slug_author == slug_id:
return slug_author
return f"{slug_author}_{slug_id}"
if slug_id:
return f"{slug_author or '未知作者'}_{slug_id}"
if slug_author:
return f"{slug_author}_{short_time_code(at)}"
if source:
return f"未知作者_{short_source_code(source)}"
return f"未知作者_{short_time_code(at)}"
p = Path(filename)
# 主文件名
name = p.stem
def build_work_stem(title: Optional[str]) -> str:
"""作品名做文件名/子目录名:slugify(沿用 15 字上限),空则 `作品`"""
return slugify(title or "", max_length=15) or "作品"
# 完整扩展名 (.tar.gz)
ext = "".join(p.suffixes)
# Unicode 标准化
name = unicodedata.normalize("NFKC", name)
def unique_media_path(path: Path, *, at: Optional[float] = None) -> Path:
"""同名才加 4 位短码:`{stem}_{码}{后缀}`,仍撞则再叠 `_2/_3…`
# 去掉 #tag
name = re.sub(r"#\S+", "", name)
文件与目录通用(目录无后缀)。命名发生在落盘前,因此"不存在"即直接采用;
顺带确保父目录存在(作者目录是按需创建的)。
"""
if path.exists():
code = short_time_code(at)
candidate = path.with_name(f"{path.stem}_{code}{path.suffix}")
index = 2
while candidate.exists():
candidate = path.with_name(f"{path.stem}_{code}_{index}{path.suffix}")
index += 1
path = candidate
# 非法字符替换
name = re.sub(r'[\\/:*?"<>|]', "_", name)
path.parent.mkdir(parents=True, exist_ok=True)
return path
# 中英文标点
name = re.sub(r"[&'\"。,:?!《》【】|]", "_", name)
# 空白 -> _
name = re.sub(r"\s+", "_", name)
def _temp_rel_parts(file_path: Path | str) -> Tuple[str, ...]:
"""相对 temp 根拆路径:`{平台}/{作者目录}/…`(归档目录同理)
# 只保留:中文、字母、数字、_
name = re.sub(r"[^\w一-鿿_]", "", name)
不在 temp 下、或没到"平台 + 作者目录"这一层(老数据 / 第三方产物)→ 空元组,
调用方退化为只用文件名。
"""
try:
rel = Path(file_path).resolve().relative_to(get_temp_root().resolve())
except (ValueError, OSError):
return ()
parts = rel.parts
return parts[1:] if len(parts) >= 3 else ()
# 合并 _
name = re.sub(r"_+", "_", name)
# 去首尾 _
name = name.strip("_")
def media_key_of(file_path: Path | str) -> str:
"""媒体文件的 S3 对象 key:`{作者}_{作者id}/{作品名}[_{码}].后缀`
# 长度控制
max_name_length = max_length - len(ext) - len(current_time) - 1
if len(name) > max_name_length:
name = name[:max_name_length].rstrip("_")
由本地路径反推(去掉平台层),保证桶里和 temp 里结构一致。
"""
parts = _temp_rel_parts(file_path)
return "/".join(parts) if parts else Path(file_path).name
return f"{current_time}_{name}{ext}"
def media_rel_dir_of(file_path: Path | str) -> str:
"""媒体文件所在的作者/作品子目录(相对 temp 根、去掉平台层),供归档复用"""
parts = _temp_rel_parts(file_path)
return "/".join(parts[:-1]) if len(parts) > 1 else ""
def parse_netscape_cookies(file_path: str) -> List[dict]: