Files
sansenhoshiandClaude Code 4badcfcf32 feat(video-analysis): 群策略 v3 / 群文件投递通道 / Web 管理页
- policy.py:per-group 正交策略(自动解析 / 自动策略 / 禁用策略 / 存储 A·B·C /
  公网 / 链接 / 群文件 + 平台限定),list.json v1/v2 → v3 自动迁移,
  写入统一走 PolicyStore(加锁 + .tmp 原子替换 + 字段归一)
- 群文件并行通道 group_file.py:打包 zip(可选 pyzipper AES-256)后优先走 S3 预签名、
  本地直传兜底;设了密码但 pyzipper 不可用就放弃上传,不退化成明文
- list_proc.py 收敛到「视频策略」统一入口,权限判定改走 policy
- Web 管理页 /hub/video_analysis(群策略 + 链接解析面板)与 services/web_jobs.py
  (只复用纯函数层,Web 上下文不发消息;内存任务表 + 并发闸门 + 超时)
- 媒体命名统一到 utils.py({作者}_{作者id}/{作品名}[_短码]),cleanup 回收空目录
- 测试:policy / 命名 / 群文件 / web_jobs 四组

顺带 pyproject 的 pytest 加 testpaths=tests(避免收进 debug/ 下的调试脚本)。

Co-Authored-By: Claude Code <noreply@anthropic.com>
2026-09-22 14:23:32 +08:00

254 lines
8.4 KiB
Python
Raw Permalink Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
import hashlib
import re
import time
import unicodedata
from pathlib import Path
from typing import List, Optional, Tuple
def get_data_dir() -> Path:
"""插件 data 目录:.../nonebot_plugin_video_analysis/data
注意 utils.py 位于插件根目录,.parent 即插件根,因此这里写死回到
插件自身 data/(cookies.txt / list.json / temp 都在此处)。
"""
return Path(__file__).resolve().parent / "data"
def get_temp_root(sub: str = "") -> Path:
"""插件媒体临时目录:data/temp[/sub] — 下载的媒体先进这里
发送时优先直接用这里的本地文件,失败才走 S3 链接(见 handlers/sender.py)。
"""
root = Path(__file__).resolve().parent.parent.parent / "data" / "temp"
if sub:
root = root / sub
root.mkdir(parents=True, exist_ok=True)
return root
def slugify(text: str, max_length: int = 80) -> str:
"""
将字符串转换为 URL slug
规则:
1. 去掉 #tag
2. Unicode 归一化
3. 转小写
4. 空白和分隔符替换为 -
5. 移除非法字符
6. 合并连续 -
7. 裁剪长度
"""
# 去掉 #标签
text = re.sub(r"#\S+", "", text)
# Unicode 标准化
text = unicodedata.normalize("NFKC", text)
# 转小写
text = text.lower()
# 空白字符 -> -
text = re.sub(r"\s+", "-", text)
# 允许:中文、字母、数字、-
text = re.sub(r"[^\w\-一-鿿]", "", text)
# 合并多个 -
text = re.sub(r"-{2,}", "-", text)
# 去掉首尾 -
text = text.strip("-")
# 控制长度
if len(text) > max_length:
text = text[:max_length].rstrip("-")
return text
def ensure_unique_path(base_path: Path) -> Path:
"""
确保路径不冲突:如已存在则追加 _2, _3... 后缀
适用于文件和目录
注:媒体落盘已统一走 `unique_media_path`(同名加 4 位短码),本函数仅作
通用兜底保留。
"""
if not base_path.exists():
return base_path
parent = base_path.parent
stem = base_path.stem
ext = base_path.suffix # 目录无后缀 -> ""
counter = 2
while True:
new_path = parent / f"{stem}_{counter}{ext}"
if not new_path.exists():
return new_path
counter += 1
_B36_ALPHABET = "0123456789abcdefghijklmnopqrstuvwxyz"
#: 同名短码的模数 —— base36 四位(36**4 ≈ 168 万秒 ≈ 19.4 天一轮)
_B36_MOD = 36**4
def _to_base36(value: int, width: int = 4) -> str:
"""整数 → 定宽 base36(不足左侧补 0)"""
if value <= 0:
return "0" * width
digits = []
while value:
value, rem = divmod(value, 36)
digits.append(_B36_ALPHABET[rem])
return "".join(reversed(digits)).rjust(width, "0")
def short_time_code(at: Optional[float] = None) -> str:
"""4 位 base36 短码(同名兜底用,不可读时间,仅作区分码)"""
return _to_base36(int(time.time() if at is None else at) % _B36_MOD)
#: 各平台把"没有 id"写成过这些值,别让它们进目录名
_JUNK_AUTHOR_IDS = {"", "0", "none", "null", "na", "nan", "undefined"}
#: 抓取层拿不到昵称时的兜底串(slugify 后)—— 它们等于"没有作者信息"
_PLACEHOLDER_AUTHORS = {"未知作者", "小红书用户", "b站用户"}
def _clean_author_id(author_id: Optional[str]) -> str:
"""作者 id 归一:剔除占位值(yt-dlp 缺字段常给 `NA`),再 slugify。"""
raw = str(author_id or "").strip()
if raw.lower() in _JUNK_AUTHOR_IDS:
return ""
return slugify(raw, max_length=40)
def short_source_code(source: str) -> str:
"""来源串(作品 URL/id)→ 4 位 base36 短码。
与 `short_time_code` 的区别:同一个来源永远得到同一个码(md5 取摘要,
不能用内置 `hash()` —— 它有随机盐,重启后目录名会变)。
"""
digest = hashlib.md5(source.encode("utf-8")).digest()
return _to_base36(int.from_bytes(digest[:4], "big") % _B36_MOD)
def build_author_dir(
author: Optional[str],
author_id: Optional[str] = None,
*,
source: Optional[str] = None,
at: Optional[float] = None,
) -> str:
"""`{作者}_{作者id}` 作者目录名(拿不到 id 时追加码值避免同名混目录)。
- 昵称 + 作者 id → `{昵称}_{id}`(昵称上限 30、id 上限 40;**不能**把 id 截到
20:`MS4wLjABAAAA…` 这类 sec_uid 公共前缀就有 17 字符,再截断必撞车)
- 只有昵称 → `{昵称}_{4 位时间短码}`:没有 id 就分不清同名作者,宁可不聚合
(同一作者的不同作品会各成一个目录)也不能把两个人混进同一个目录
- 连昵称都没有 → `未知作者_{来源短码}`(`source` 给作品 URL/id,同一来源
稳定、不同来源不撞);连 source 都没有 → 退化成时间短码
"""
slug_author = slugify(author or "", max_length=30)
if slug_author in _PLACEHOLDER_AUTHORS:
slug_author = ""
slug_id = _clean_author_id(author_id)
if slug_author and slug_id:
# 某些站点上传者名就是 handle(X 的 @someone),别产出 someone_someone
if slug_author == slug_id:
return slug_author
return f"{slug_author}_{slug_id}"
if slug_id:
return f"{slug_author or '未知作者'}_{slug_id}"
if slug_author:
return f"{slug_author}_{short_time_code(at)}"
if source:
return f"未知作者_{short_source_code(source)}"
return f"未知作者_{short_time_code(at)}"
def build_work_stem(title: Optional[str]) -> str:
"""作品名做文件名/子目录名:slugify(沿用 15 字上限),空则 `作品`"""
return slugify(title or "", max_length=15) or "作品"
def unique_media_path(path: Path, *, at: Optional[float] = None) -> Path:
"""同名才加 4 位短码:`{stem}_{码}{后缀}`,仍撞则再叠 `_2/_3…`
文件与目录通用(目录无后缀)。命名发生在落盘前,因此"不存在"即直接采用;
顺带确保父目录存在(作者目录是按需创建的)。
"""
if path.exists():
code = short_time_code(at)
candidate = path.with_name(f"{path.stem}_{code}{path.suffix}")
index = 2
while candidate.exists():
candidate = path.with_name(f"{path.stem}_{code}_{index}{path.suffix}")
index += 1
path = candidate
path.parent.mkdir(parents=True, exist_ok=True)
return path
def _temp_rel_parts(file_path: Path | str) -> Tuple[str, ...]:
"""相对 temp 根拆路径:`{平台}/{作者目录}/…`(归档目录同理)
不在 temp 下、或没到"平台 + 作者目录"这一层(老数据 / 第三方产物)→ 空元组,
调用方退化为只用文件名。
"""
try:
rel = Path(file_path).resolve().relative_to(get_temp_root().resolve())
except (ValueError, OSError):
return ()
parts = rel.parts
return parts[1:] if len(parts) >= 3 else ()
def media_key_of(file_path: Path | str) -> str:
"""媒体文件的 S3 对象 key:`{作者}_{作者id}/{作品名}[_{码}].后缀`
由本地路径反推(去掉平台层),保证桶里和 temp 里结构一致。
"""
parts = _temp_rel_parts(file_path)
return "/".join(parts) if parts else Path(file_path).name
def media_rel_dir_of(file_path: Path | str) -> str:
"""媒体文件所在的作者/作品子目录(相对 temp 根、去掉平台层),供归档复用"""
parts = _temp_rel_parts(file_path)
return "/".join(parts[:-1]) if len(parts) > 1 else ""
def parse_netscape_cookies(file_path: str) -> List[dict]:
"""解析 Netscape 格式 cookies 文件"""
cookies = []
with open(file_path, "r", encoding="utf-8") as f:
for line in f:
line = line.strip()
if not line or line.startswith("#"):
continue
parts = line.split("\t")
if len(parts) != 7:
continue
domain, flag, path, secure, expiry, name, value = parts
cookie = {
"name": name,
"value": value,
"domain": domain,
"path": path,
"secure": secure.upper() == "TRUE",
"sameSite": "Lax",
}
if expiry.isdigit() and int(expiry) > 0:
cookie["expires"] = int(expiry)
cookies.append(cookie)
return cookies