2026-09-22 14:23:32 +08:00
|
|
|
|
import hashlib
|
2026-09-01 13:13:40 +08:00
|
|
|
|
import re
|
2026-09-22 14:23:32 +08:00
|
|
|
|
import time
|
2026-09-01 13:13:40 +08:00
|
|
|
|
import unicodedata
|
|
|
|
|
|
from pathlib import Path
|
2026-09-22 14:23:32 +08:00
|
|
|
|
from typing import List, Optional, Tuple
|
2026-09-01 13:13:40 +08:00
|
|
|
|
|
|
|
|
|
|
|
2026-09-03 00:44:38 +08:00
|
|
|
|
def get_data_dir() -> Path:
|
|
|
|
|
|
"""插件 data 目录:.../nonebot_plugin_video_analysis/data
|
|
|
|
|
|
|
|
|
|
|
|
注意 utils.py 位于插件根目录,.parent 即插件根,因此这里写死回到
|
|
|
|
|
|
插件自身 data/(cookies.txt / list.json / temp 都在此处)。
|
|
|
|
|
|
"""
|
|
|
|
|
|
return Path(__file__).resolve().parent / "data"
|
|
|
|
|
|
|
|
|
|
|
|
|
2026-09-01 13:13:40 +08:00
|
|
|
|
def get_temp_root(sub: str = "") -> Path:
|
|
|
|
|
|
"""插件媒体临时目录:data/temp[/sub] — 下载的媒体先进这里
|
|
|
|
|
|
|
|
|
|
|
|
发送时优先直接用这里的本地文件,失败才走 S3 链接(见 handlers/sender.py)。
|
|
|
|
|
|
"""
|
|
|
|
|
|
root = Path(__file__).resolve().parent.parent.parent / "data" / "temp"
|
|
|
|
|
|
if sub:
|
|
|
|
|
|
root = root / sub
|
|
|
|
|
|
root.mkdir(parents=True, exist_ok=True)
|
|
|
|
|
|
return root
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def slugify(text: str, max_length: int = 80) -> str:
|
|
|
|
|
|
"""
|
|
|
|
|
|
将字符串转换为 URL slug
|
|
|
|
|
|
|
|
|
|
|
|
规则:
|
|
|
|
|
|
1. 去掉 #tag
|
|
|
|
|
|
2. Unicode 归一化
|
|
|
|
|
|
3. 转小写
|
|
|
|
|
|
4. 空白和分隔符替换为 -
|
|
|
|
|
|
5. 移除非法字符
|
|
|
|
|
|
6. 合并连续 -
|
|
|
|
|
|
7. 裁剪长度
|
|
|
|
|
|
"""
|
|
|
|
|
|
|
|
|
|
|
|
# 去掉 #标签
|
|
|
|
|
|
text = re.sub(r"#\S+", "", text)
|
|
|
|
|
|
|
|
|
|
|
|
# Unicode 标准化
|
|
|
|
|
|
text = unicodedata.normalize("NFKC", text)
|
|
|
|
|
|
|
|
|
|
|
|
# 转小写
|
|
|
|
|
|
text = text.lower()
|
|
|
|
|
|
|
|
|
|
|
|
# 空白字符 -> -
|
|
|
|
|
|
text = re.sub(r"\s+", "-", text)
|
|
|
|
|
|
|
|
|
|
|
|
# 允许:中文、字母、数字、-
|
|
|
|
|
|
text = re.sub(r"[^\w\-一-鿿]", "", text)
|
|
|
|
|
|
|
|
|
|
|
|
# 合并多个 -
|
|
|
|
|
|
text = re.sub(r"-{2,}", "-", text)
|
|
|
|
|
|
|
|
|
|
|
|
# 去掉首尾 -
|
|
|
|
|
|
text = text.strip("-")
|
|
|
|
|
|
|
|
|
|
|
|
# 控制长度
|
|
|
|
|
|
if len(text) > max_length:
|
|
|
|
|
|
text = text[:max_length].rstrip("-")
|
|
|
|
|
|
|
|
|
|
|
|
return text
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def ensure_unique_path(base_path: Path) -> Path:
|
|
|
|
|
|
"""
|
|
|
|
|
|
确保路径不冲突:如已存在则追加 _2, _3... 后缀
|
|
|
|
|
|
适用于文件和目录
|
2026-09-22 14:23:32 +08:00
|
|
|
|
|
|
|
|
|
|
注:媒体落盘已统一走 `unique_media_path`(同名加 4 位短码),本函数仅作
|
|
|
|
|
|
通用兜底保留。
|
2026-09-01 13:13:40 +08:00
|
|
|
|
"""
|
|
|
|
|
|
if not base_path.exists():
|
|
|
|
|
|
return base_path
|
|
|
|
|
|
|
|
|
|
|
|
parent = base_path.parent
|
|
|
|
|
|
stem = base_path.stem
|
|
|
|
|
|
ext = base_path.suffix # 目录无后缀 -> ""
|
|
|
|
|
|
|
|
|
|
|
|
counter = 2
|
|
|
|
|
|
while True:
|
|
|
|
|
|
new_path = parent / f"{stem}_{counter}{ext}"
|
|
|
|
|
|
if not new_path.exists():
|
|
|
|
|
|
return new_path
|
|
|
|
|
|
counter += 1
|
|
|
|
|
|
|
|
|
|
|
|
|
2026-09-22 14:23:32 +08:00
|
|
|
|
_B36_ALPHABET = "0123456789abcdefghijklmnopqrstuvwxyz"
|
2026-09-01 13:13:40 +08:00
|
|
|
|
|
2026-09-22 14:23:32 +08:00
|
|
|
|
#: 同名短码的模数 —— base36 四位(36**4 ≈ 168 万秒 ≈ 19.4 天一轮)
|
|
|
|
|
|
_B36_MOD = 36**4
|
2026-09-01 13:13:40 +08:00
|
|
|
|
|
|
|
|
|
|
|
2026-09-22 14:23:32 +08:00
|
|
|
|
def _to_base36(value: int, width: int = 4) -> str:
|
|
|
|
|
|
"""整数 → 定宽 base36(不足左侧补 0)"""
|
|
|
|
|
|
if value <= 0:
|
|
|
|
|
|
return "0" * width
|
|
|
|
|
|
digits = []
|
|
|
|
|
|
while value:
|
|
|
|
|
|
value, rem = divmod(value, 36)
|
|
|
|
|
|
digits.append(_B36_ALPHABET[rem])
|
|
|
|
|
|
return "".join(reversed(digits)).rjust(width, "0")
|
2026-09-01 13:13:40 +08:00
|
|
|
|
|
|
|
|
|
|
|
2026-09-22 14:23:32 +08:00
|
|
|
|
def short_time_code(at: Optional[float] = None) -> str:
|
|
|
|
|
|
"""4 位 base36 短码(同名兜底用,不可读时间,仅作区分码)"""
|
|
|
|
|
|
return _to_base36(int(time.time() if at is None else at) % _B36_MOD)
|
|
|
|
|
|
|
2026-09-01 13:13:40 +08:00
|
|
|
|
|
2026-09-22 14:23:32 +08:00
|
|
|
|
#: 各平台把"没有 id"写成过这些值,别让它们进目录名
|
|
|
|
|
|
_JUNK_AUTHOR_IDS = {"", "0", "none", "null", "na", "nan", "undefined"}
|
2026-09-01 13:13:40 +08:00
|
|
|
|
|
2026-09-22 14:23:32 +08:00
|
|
|
|
#: 抓取层拿不到昵称时的兜底串(slugify 后)—— 它们等于"没有作者信息"
|
|
|
|
|
|
_PLACEHOLDER_AUTHORS = {"未知作者", "小红书用户", "b站用户"}
|
2026-09-01 13:13:40 +08:00
|
|
|
|
|
|
|
|
|
|
|
2026-09-22 14:23:32 +08:00
|
|
|
|
def _clean_author_id(author_id: Optional[str]) -> str:
|
|
|
|
|
|
"""作者 id 归一:剔除占位值(yt-dlp 缺字段常给 `NA`),再 slugify。"""
|
|
|
|
|
|
raw = str(author_id or "").strip()
|
|
|
|
|
|
if raw.lower() in _JUNK_AUTHOR_IDS:
|
|
|
|
|
|
return ""
|
|
|
|
|
|
return slugify(raw, max_length=40)
|
2026-09-01 13:13:40 +08:00
|
|
|
|
|
|
|
|
|
|
|
2026-09-22 14:23:32 +08:00
|
|
|
|
def short_source_code(source: str) -> str:
|
|
|
|
|
|
"""来源串(作品 URL/id)→ 4 位 base36 短码。
|
|
|
|
|
|
|
|
|
|
|
|
与 `short_time_code` 的区别:同一个来源永远得到同一个码(md5 取摘要,
|
|
|
|
|
|
不能用内置 `hash()` —— 它有随机盐,重启后目录名会变)。
|
|
|
|
|
|
"""
|
|
|
|
|
|
digest = hashlib.md5(source.encode("utf-8")).digest()
|
|
|
|
|
|
return _to_base36(int.from_bytes(digest[:4], "big") % _B36_MOD)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def build_author_dir(
|
|
|
|
|
|
author: Optional[str],
|
|
|
|
|
|
author_id: Optional[str] = None,
|
|
|
|
|
|
*,
|
|
|
|
|
|
source: Optional[str] = None,
|
|
|
|
|
|
at: Optional[float] = None,
|
|
|
|
|
|
) -> str:
|
|
|
|
|
|
"""`{作者}_{作者id}` 作者目录名(拿不到 id 时追加码值避免同名混目录)。
|
|
|
|
|
|
|
|
|
|
|
|
- 昵称 + 作者 id → `{昵称}_{id}`(昵称上限 30、id 上限 40;**不能**把 id 截到
|
|
|
|
|
|
20:`MS4wLjABAAAA…` 这类 sec_uid 公共前缀就有 17 字符,再截断必撞车)
|
|
|
|
|
|
- 只有昵称 → `{昵称}_{4 位时间短码}`:没有 id 就分不清同名作者,宁可不聚合
|
|
|
|
|
|
(同一作者的不同作品会各成一个目录)也不能把两个人混进同一个目录
|
|
|
|
|
|
- 连昵称都没有 → `未知作者_{来源短码}`(`source` 给作品 URL/id,同一来源
|
|
|
|
|
|
稳定、不同来源不撞);连 source 都没有 → 退化成时间短码
|
|
|
|
|
|
"""
|
|
|
|
|
|
slug_author = slugify(author or "", max_length=30)
|
|
|
|
|
|
if slug_author in _PLACEHOLDER_AUTHORS:
|
|
|
|
|
|
slug_author = ""
|
|
|
|
|
|
slug_id = _clean_author_id(author_id)
|
|
|
|
|
|
|
|
|
|
|
|
if slug_author and slug_id:
|
|
|
|
|
|
# 某些站点上传者名就是 handle(X 的 @someone),别产出 someone_someone
|
|
|
|
|
|
if slug_author == slug_id:
|
|
|
|
|
|
return slug_author
|
|
|
|
|
|
return f"{slug_author}_{slug_id}"
|
|
|
|
|
|
if slug_id:
|
|
|
|
|
|
return f"{slug_author or '未知作者'}_{slug_id}"
|
|
|
|
|
|
if slug_author:
|
|
|
|
|
|
return f"{slug_author}_{short_time_code(at)}"
|
|
|
|
|
|
if source:
|
|
|
|
|
|
return f"未知作者_{short_source_code(source)}"
|
|
|
|
|
|
return f"未知作者_{short_time_code(at)}"
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def build_work_stem(title: Optional[str]) -> str:
|
|
|
|
|
|
"""作品名做文件名/子目录名:slugify(沿用 15 字上限),空则 `作品`"""
|
|
|
|
|
|
return slugify(title or "", max_length=15) or "作品"
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def unique_media_path(path: Path, *, at: Optional[float] = None) -> Path:
|
|
|
|
|
|
"""同名才加 4 位短码:`{stem}_{码}{后缀}`,仍撞则再叠 `_2/_3…`
|
|
|
|
|
|
|
|
|
|
|
|
文件与目录通用(目录无后缀)。命名发生在落盘前,因此"不存在"即直接采用;
|
|
|
|
|
|
顺带确保父目录存在(作者目录是按需创建的)。
|
|
|
|
|
|
"""
|
|
|
|
|
|
if path.exists():
|
|
|
|
|
|
code = short_time_code(at)
|
|
|
|
|
|
candidate = path.with_name(f"{path.stem}_{code}{path.suffix}")
|
|
|
|
|
|
index = 2
|
|
|
|
|
|
while candidate.exists():
|
|
|
|
|
|
candidate = path.with_name(f"{path.stem}_{code}_{index}{path.suffix}")
|
|
|
|
|
|
index += 1
|
|
|
|
|
|
path = candidate
|
2026-09-01 13:13:40 +08:00
|
|
|
|
|
2026-09-22 14:23:32 +08:00
|
|
|
|
path.parent.mkdir(parents=True, exist_ok=True)
|
|
|
|
|
|
return path
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def _temp_rel_parts(file_path: Path | str) -> Tuple[str, ...]:
|
|
|
|
|
|
"""相对 temp 根拆路径:`{平台}/{作者目录}/…`(归档目录同理)
|
|
|
|
|
|
|
|
|
|
|
|
不在 temp 下、或没到"平台 + 作者目录"这一层(老数据 / 第三方产物)→ 空元组,
|
|
|
|
|
|
调用方退化为只用文件名。
|
|
|
|
|
|
"""
|
|
|
|
|
|
try:
|
|
|
|
|
|
rel = Path(file_path).resolve().relative_to(get_temp_root().resolve())
|
|
|
|
|
|
except (ValueError, OSError):
|
|
|
|
|
|
return ()
|
|
|
|
|
|
parts = rel.parts
|
|
|
|
|
|
return parts[1:] if len(parts) >= 3 else ()
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def media_key_of(file_path: Path | str) -> str:
|
|
|
|
|
|
"""媒体文件的 S3 对象 key:`{作者}_{作者id}/{作品名}[_{码}].后缀`
|
|
|
|
|
|
|
|
|
|
|
|
由本地路径反推(去掉平台层),保证桶里和 temp 里结构一致。
|
|
|
|
|
|
"""
|
|
|
|
|
|
parts = _temp_rel_parts(file_path)
|
|
|
|
|
|
return "/".join(parts) if parts else Path(file_path).name
|
2026-09-01 13:13:40 +08:00
|
|
|
|
|
|
|
|
|
|
|
2026-09-22 14:23:32 +08:00
|
|
|
|
def media_rel_dir_of(file_path: Path | str) -> str:
|
|
|
|
|
|
"""媒体文件所在的作者/作品子目录(相对 temp 根、去掉平台层),供归档复用"""
|
|
|
|
|
|
parts = _temp_rel_parts(file_path)
|
|
|
|
|
|
return "/".join(parts[:-1]) if len(parts) > 1 else ""
|
2026-09-01 13:13:40 +08:00
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def parse_netscape_cookies(file_path: str) -> List[dict]:
|
|
|
|
|
|
"""解析 Netscape 格式 cookies 文件"""
|
|
|
|
|
|
cookies = []
|
|
|
|
|
|
with open(file_path, "r", encoding="utf-8") as f:
|
|
|
|
|
|
for line in f:
|
|
|
|
|
|
line = line.strip()
|
|
|
|
|
|
if not line or line.startswith("#"):
|
|
|
|
|
|
continue
|
|
|
|
|
|
parts = line.split("\t")
|
|
|
|
|
|
if len(parts) != 7:
|
|
|
|
|
|
continue
|
|
|
|
|
|
domain, flag, path, secure, expiry, name, value = parts
|
|
|
|
|
|
cookie = {
|
|
|
|
|
|
"name": name,
|
|
|
|
|
|
"value": value,
|
|
|
|
|
|
"domain": domain,
|
|
|
|
|
|
"path": path,
|
|
|
|
|
|
"secure": secure.upper() == "TRUE",
|
|
|
|
|
|
"sameSite": "Lax",
|
|
|
|
|
|
}
|
|
|
|
|
|
if expiry.isdigit() and int(expiry) > 0:
|
|
|
|
|
|
cookie["expires"] = int(expiry)
|
|
|
|
|
|
cookies.append(cookie)
|
|
|
|
|
|
return cookies
|