feat(video-analysis): 群策略 v3 / 群文件投递通道 / Web 管理页
- policy.py:per-group 正交策略(自动解析 / 自动策略 / 禁用策略 / 存储 A·B·C /
公网 / 链接 / 群文件 + 平台限定),list.json v1/v2 → v3 自动迁移,
写入统一走 PolicyStore(加锁 + .tmp 原子替换 + 字段归一)
- 群文件并行通道 group_file.py:打包 zip(可选 pyzipper AES-256)后优先走 S3 预签名、
本地直传兜底;设了密码但 pyzipper 不可用就放弃上传,不退化成明文
- list_proc.py 收敛到「视频策略」统一入口,权限判定改走 policy
- Web 管理页 /hub/video_analysis(群策略 + 链接解析面板)与 services/web_jobs.py
(只复用纯函数层,Web 上下文不发消息;内存任务表 + 并发闸门 + 超时)
- 媒体命名统一到 utils.py({作者}_{作者id}/{作品名}[_短码]),cleanup 回收空目录
- 测试:policy / 命名 / 群文件 / web_jobs 四组
顺带 pyproject 的 pytest 加 testpaths=tests(避免收进 debug/ 下的调试脚本)。
Co-Authored-By: Claude Code <noreply@anthropic.com>
This commit is contained in:
@@ -1,8 +1,9 @@
|
||||
import hashlib
|
||||
import re
|
||||
import time
|
||||
import unicodedata
|
||||
from pathlib import Path
|
||||
from time import strftime, localtime
|
||||
from typing import List
|
||||
from typing import List, Optional, Tuple
|
||||
|
||||
|
||||
def get_data_dir() -> Path:
|
||||
@@ -72,6 +73,9 @@ def ensure_unique_path(base_path: Path) -> Path:
|
||||
"""
|
||||
确保路径不冲突:如已存在则追加 _2, _3... 后缀
|
||||
适用于文件和目录
|
||||
|
||||
注:媒体落盘已统一走 `unique_media_path`(同名加 4 位短码),本函数仅作
|
||||
通用兜底保留。
|
||||
"""
|
||||
if not base_path.exists():
|
||||
return base_path
|
||||
@@ -88,52 +92,139 @@ def ensure_unique_path(base_path: Path) -> Path:
|
||||
counter += 1
|
||||
|
||||
|
||||
def clean_filename(filename: str, max_length: int = 120) -> str:
|
||||
_B36_ALPHABET = "0123456789abcdefghijklmnopqrstuvwxyz"
|
||||
|
||||
#: 同名短码的模数 —— base36 四位(36**4 ≈ 168 万秒 ≈ 19.4 天一轮)
|
||||
_B36_MOD = 36**4
|
||||
|
||||
|
||||
def _to_base36(value: int, width: int = 4) -> str:
|
||||
"""整数 → 定宽 base36(不足左侧补 0)"""
|
||||
if value <= 0:
|
||||
return "0" * width
|
||||
digits = []
|
||||
while value:
|
||||
value, rem = divmod(value, 36)
|
||||
digits.append(_B36_ALPHABET[rem])
|
||||
return "".join(reversed(digits)).rjust(width, "0")
|
||||
|
||||
|
||||
def short_time_code(at: Optional[float] = None) -> str:
|
||||
"""4 位 base36 短码(同名兜底用,不可读时间,仅作区分码)"""
|
||||
return _to_base36(int(time.time() if at is None else at) % _B36_MOD)
|
||||
|
||||
|
||||
#: 各平台把"没有 id"写成过这些值,别让它们进目录名
|
||||
_JUNK_AUTHOR_IDS = {"", "0", "none", "null", "na", "nan", "undefined"}
|
||||
|
||||
#: 抓取层拿不到昵称时的兜底串(slugify 后)—— 它们等于"没有作者信息"
|
||||
_PLACEHOLDER_AUTHORS = {"未知作者", "小红书用户", "b站用户"}
|
||||
|
||||
|
||||
def _clean_author_id(author_id: Optional[str]) -> str:
|
||||
"""作者 id 归一:剔除占位值(yt-dlp 缺字段常给 `NA`),再 slugify。"""
|
||||
raw = str(author_id or "").strip()
|
||||
if raw.lower() in _JUNK_AUTHOR_IDS:
|
||||
return ""
|
||||
return slugify(raw, max_length=40)
|
||||
|
||||
|
||||
def short_source_code(source: str) -> str:
|
||||
"""来源串(作品 URL/id)→ 4 位 base36 短码。
|
||||
|
||||
与 `short_time_code` 的区别:同一个来源永远得到同一个码(md5 取摘要,
|
||||
不能用内置 `hash()` —— 它有随机盐,重启后目录名会变)。
|
||||
"""
|
||||
清理文件名并添加时间前缀
|
||||
支持多扩展名,如 .tar.gz
|
||||
digest = hashlib.md5(source.encode("utf-8")).digest()
|
||||
return _to_base36(int.from_bytes(digest[:4], "big") % _B36_MOD)
|
||||
|
||||
|
||||
def build_author_dir(
|
||||
author: Optional[str],
|
||||
author_id: Optional[str] = None,
|
||||
*,
|
||||
source: Optional[str] = None,
|
||||
at: Optional[float] = None,
|
||||
) -> str:
|
||||
"""`{作者}_{作者id}` 作者目录名(拿不到 id 时追加码值避免同名混目录)。
|
||||
|
||||
- 昵称 + 作者 id → `{昵称}_{id}`(昵称上限 30、id 上限 40;**不能**把 id 截到
|
||||
20:`MS4wLjABAAAA…` 这类 sec_uid 公共前缀就有 17 字符,再截断必撞车)
|
||||
- 只有昵称 → `{昵称}_{4 位时间短码}`:没有 id 就分不清同名作者,宁可不聚合
|
||||
(同一作者的不同作品会各成一个目录)也不能把两个人混进同一个目录
|
||||
- 连昵称都没有 → `未知作者_{来源短码}`(`source` 给作品 URL/id,同一来源
|
||||
稳定、不同来源不撞);连 source 都没有 → 退化成时间短码
|
||||
"""
|
||||
slug_author = slugify(author or "", max_length=30)
|
||||
if slug_author in _PLACEHOLDER_AUTHORS:
|
||||
slug_author = ""
|
||||
slug_id = _clean_author_id(author_id)
|
||||
|
||||
current_time = strftime("%H-%M-%S", localtime())
|
||||
if slug_author and slug_id:
|
||||
# 某些站点上传者名就是 handle(X 的 @someone),别产出 someone_someone
|
||||
if slug_author == slug_id:
|
||||
return slug_author
|
||||
return f"{slug_author}_{slug_id}"
|
||||
if slug_id:
|
||||
return f"{slug_author or '未知作者'}_{slug_id}"
|
||||
if slug_author:
|
||||
return f"{slug_author}_{short_time_code(at)}"
|
||||
if source:
|
||||
return f"未知作者_{short_source_code(source)}"
|
||||
return f"未知作者_{short_time_code(at)}"
|
||||
|
||||
p = Path(filename)
|
||||
|
||||
# 主文件名
|
||||
name = p.stem
|
||||
def build_work_stem(title: Optional[str]) -> str:
|
||||
"""作品名做文件名/子目录名:slugify(沿用 15 字上限),空则 `作品`"""
|
||||
return slugify(title or "", max_length=15) or "作品"
|
||||
|
||||
# 完整扩展名 (.tar.gz)
|
||||
ext = "".join(p.suffixes)
|
||||
|
||||
# Unicode 标准化
|
||||
name = unicodedata.normalize("NFKC", name)
|
||||
def unique_media_path(path: Path, *, at: Optional[float] = None) -> Path:
|
||||
"""同名才加 4 位短码:`{stem}_{码}{后缀}`,仍撞则再叠 `_2/_3…`
|
||||
|
||||
# 去掉 #tag
|
||||
name = re.sub(r"#\S+", "", name)
|
||||
文件与目录通用(目录无后缀)。命名发生在落盘前,因此"不存在"即直接采用;
|
||||
顺带确保父目录存在(作者目录是按需创建的)。
|
||||
"""
|
||||
if path.exists():
|
||||
code = short_time_code(at)
|
||||
candidate = path.with_name(f"{path.stem}_{code}{path.suffix}")
|
||||
index = 2
|
||||
while candidate.exists():
|
||||
candidate = path.with_name(f"{path.stem}_{code}_{index}{path.suffix}")
|
||||
index += 1
|
||||
path = candidate
|
||||
|
||||
# 非法字符替换
|
||||
name = re.sub(r'[\\/:*?"<>|]', "_", name)
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
return path
|
||||
|
||||
# 中英文标点
|
||||
name = re.sub(r"[&'\"。,:?!《》【】|]", "_", name)
|
||||
|
||||
# 空白 -> _
|
||||
name = re.sub(r"\s+", "_", name)
|
||||
def _temp_rel_parts(file_path: Path | str) -> Tuple[str, ...]:
|
||||
"""相对 temp 根拆路径:`{平台}/{作者目录}/…`(归档目录同理)
|
||||
|
||||
# 只保留:中文、字母、数字、_
|
||||
name = re.sub(r"[^\w一-鿿_]", "", name)
|
||||
不在 temp 下、或没到"平台 + 作者目录"这一层(老数据 / 第三方产物)→ 空元组,
|
||||
调用方退化为只用文件名。
|
||||
"""
|
||||
try:
|
||||
rel = Path(file_path).resolve().relative_to(get_temp_root().resolve())
|
||||
except (ValueError, OSError):
|
||||
return ()
|
||||
parts = rel.parts
|
||||
return parts[1:] if len(parts) >= 3 else ()
|
||||
|
||||
# 合并 _
|
||||
name = re.sub(r"_+", "_", name)
|
||||
|
||||
# 去首尾 _
|
||||
name = name.strip("_")
|
||||
def media_key_of(file_path: Path | str) -> str:
|
||||
"""媒体文件的 S3 对象 key:`{作者}_{作者id}/{作品名}[_{码}].后缀`
|
||||
|
||||
# 长度控制
|
||||
max_name_length = max_length - len(ext) - len(current_time) - 1
|
||||
if len(name) > max_name_length:
|
||||
name = name[:max_name_length].rstrip("_")
|
||||
由本地路径反推(去掉平台层),保证桶里和 temp 里结构一致。
|
||||
"""
|
||||
parts = _temp_rel_parts(file_path)
|
||||
return "/".join(parts) if parts else Path(file_path).name
|
||||
|
||||
return f"{current_time}_{name}{ext}"
|
||||
|
||||
def media_rel_dir_of(file_path: Path | str) -> str:
|
||||
"""媒体文件所在的作者/作品子目录(相对 temp 根、去掉平台层),供归档复用"""
|
||||
parts = _temp_rel_parts(file_path)
|
||||
return "/".join(parts[:-1]) if len(parts) > 1 else ""
|
||||
|
||||
|
||||
def parse_netscape_cookies(file_path: str) -> List[dict]:
|
||||
|
||||
Reference in New Issue
Block a user