Files
HeXi/hexi/plugins/nonebot_plugin_video_analysis/utils.py
T

254 lines
8.4 KiB
Python
Raw Normal View History

import hashlib
import re
import time
import unicodedata
from pathlib import Path
from typing import List, Optional, Tuple
def get_data_dir() -> Path:
"""插件 data 目录:.../nonebot_plugin_video_analysis/data
注意 utils.py 位于插件根目录,.parent 即插件根,因此这里写死回到
插件自身 data/(cookies.txt / list.json / temp 都在此处)。
"""
return Path(__file__).resolve().parent / "data"
def get_temp_root(sub: str = "") -> Path:
"""插件媒体临时目录:data/temp[/sub] — 下载的媒体先进这里
发送时优先直接用这里的本地文件,失败才走 S3 链接(见 handlers/sender.py)。
"""
root = Path(__file__).resolve().parent.parent.parent / "data" / "temp"
if sub:
root = root / sub
root.mkdir(parents=True, exist_ok=True)
return root
def slugify(text: str, max_length: int = 80) -> str:
"""
将字符串转换为 URL slug
规则:
1. 去掉 #tag
2. Unicode 归一化
3. 转小写
4. 空白和分隔符替换为 -
5. 移除非法字符
6. 合并连续 -
7. 裁剪长度
"""
# 去掉 #标签
text = re.sub(r"#\S+", "", text)
# Unicode 标准化
text = unicodedata.normalize("NFKC", text)
# 转小写
text = text.lower()
# 空白字符 -> -
text = re.sub(r"\s+", "-", text)
# 允许:中文、字母、数字、-
text = re.sub(r"[^\w\-一-鿿]", "", text)
# 合并多个 -
text = re.sub(r"-{2,}", "-", text)
# 去掉首尾 -
text = text.strip("-")
# 控制长度
if len(text) > max_length:
text = text[:max_length].rstrip("-")
return text
def ensure_unique_path(base_path: Path) -> Path:
"""
确保路径不冲突:如已存在则追加 _2, _3... 后缀
适用于文件和目录
注:媒体落盘已统一走 `unique_media_path`(同名加 4 位短码),本函数仅作
通用兜底保留。
"""
if not base_path.exists():
return base_path
parent = base_path.parent
stem = base_path.stem
ext = base_path.suffix # 目录无后缀 -> ""
counter = 2
while True:
new_path = parent / f"{stem}_{counter}{ext}"
if not new_path.exists():
return new_path
counter += 1
_B36_ALPHABET = "0123456789abcdefghijklmnopqrstuvwxyz"
#: 同名短码的模数 —— base36 四位(36**4 ≈ 168 万秒 ≈ 19.4 天一轮)
_B36_MOD = 36**4
def _to_base36(value: int, width: int = 4) -> str:
"""整数 → 定宽 base36(不足左侧补 0)"""
if value <= 0:
return "0" * width
digits = []
while value:
value, rem = divmod(value, 36)
digits.append(_B36_ALPHABET[rem])
return "".join(reversed(digits)).rjust(width, "0")
def short_time_code(at: Optional[float] = None) -> str:
"""4 位 base36 短码(同名兜底用,不可读时间,仅作区分码)"""
return _to_base36(int(time.time() if at is None else at) % _B36_MOD)
#: 各平台把"没有 id"写成过这些值,别让它们进目录名
_JUNK_AUTHOR_IDS = {"", "0", "none", "null", "na", "nan", "undefined"}
#: 抓取层拿不到昵称时的兜底串(slugify 后)—— 它们等于"没有作者信息"
_PLACEHOLDER_AUTHORS = {"未知作者", "小红书用户", "b站用户"}
def _clean_author_id(author_id: Optional[str]) -> str:
"""作者 id 归一:剔除占位值(yt-dlp 缺字段常给 `NA`),再 slugify。"""
raw = str(author_id or "").strip()
if raw.lower() in _JUNK_AUTHOR_IDS:
return ""
return slugify(raw, max_length=40)
def short_source_code(source: str) -> str:
"""来源串(作品 URL/id)→ 4 位 base36 短码。
与 `short_time_code` 的区别:同一个来源永远得到同一个码(md5 取摘要,
不能用内置 `hash()` —— 它有随机盐,重启后目录名会变)。
"""
digest = hashlib.md5(source.encode("utf-8")).digest()
return _to_base36(int.from_bytes(digest[:4], "big") % _B36_MOD)
def build_author_dir(
author: Optional[str],
author_id: Optional[str] = None,
*,
source: Optional[str] = None,
at: Optional[float] = None,
) -> str:
"""`{作者}_{作者id}` 作者目录名(拿不到 id 时追加码值避免同名混目录)。
- 昵称 + 作者 id → `{昵称}_{id}`(昵称上限 30、id 上限 40;**不能**把 id 截到
20:`MS4wLjABAAAA…` 这类 sec_uid 公共前缀就有 17 字符,再截断必撞车)
- 只有昵称 → `{昵称}_{4 位时间短码}`:没有 id 就分不清同名作者,宁可不聚合
(同一作者的不同作品会各成一个目录)也不能把两个人混进同一个目录
- 连昵称都没有 → `未知作者_{来源短码}`(`source` 给作品 URL/id,同一来源
稳定、不同来源不撞);连 source 都没有 → 退化成时间短码
"""
slug_author = slugify(author or "", max_length=30)
if slug_author in _PLACEHOLDER_AUTHORS:
slug_author = ""
slug_id = _clean_author_id(author_id)
if slug_author and slug_id:
# 某些站点上传者名就是 handle(X 的 @someone),别产出 someone_someone
if slug_author == slug_id:
return slug_author
return f"{slug_author}_{slug_id}"
if slug_id:
return f"{slug_author or '未知作者'}_{slug_id}"
if slug_author:
return f"{slug_author}_{short_time_code(at)}"
if source:
return f"未知作者_{short_source_code(source)}"
return f"未知作者_{short_time_code(at)}"
def build_work_stem(title: Optional[str]) -> str:
"""作品名做文件名/子目录名:slugify(沿用 15 字上限),空则 `作品`"""
return slugify(title or "", max_length=15) or "作品"
def unique_media_path(path: Path, *, at: Optional[float] = None) -> Path:
"""同名才加 4 位短码:`{stem}_{码}{后缀}`,仍撞则再叠 `_2/_3…`
文件与目录通用(目录无后缀)。命名发生在落盘前,因此"不存在"即直接采用;
顺带确保父目录存在(作者目录是按需创建的)。
"""
if path.exists():
code = short_time_code(at)
candidate = path.with_name(f"{path.stem}_{code}{path.suffix}")
index = 2
while candidate.exists():
candidate = path.with_name(f"{path.stem}_{code}_{index}{path.suffix}")
index += 1
path = candidate
path.parent.mkdir(parents=True, exist_ok=True)
return path
def _temp_rel_parts(file_path: Path | str) -> Tuple[str, ...]:
"""相对 temp 根拆路径:`{平台}/{作者目录}/…`(归档目录同理)
不在 temp 下、或没到"平台 + 作者目录"这一层(老数据 / 第三方产物)→ 空元组,
调用方退化为只用文件名。
"""
try:
rel = Path(file_path).resolve().relative_to(get_temp_root().resolve())
except (ValueError, OSError):
return ()
parts = rel.parts
return parts[1:] if len(parts) >= 3 else ()
def media_key_of(file_path: Path | str) -> str:
"""媒体文件的 S3 对象 key:`{作者}_{作者id}/{作品名}[_{码}].后缀`
由本地路径反推(去掉平台层),保证桶里和 temp 里结构一致。
"""
parts = _temp_rel_parts(file_path)
return "/".join(parts) if parts else Path(file_path).name
def media_rel_dir_of(file_path: Path | str) -> str:
"""媒体文件所在的作者/作品子目录(相对 temp 根、去掉平台层),供归档复用"""
parts = _temp_rel_parts(file_path)
return "/".join(parts[:-1]) if len(parts) > 1 else ""
def parse_netscape_cookies(file_path: str) -> List[dict]:
"""解析 Netscape 格式 cookies 文件"""
cookies = []
with open(file_path, "r", encoding="utf-8") as f:
for line in f:
line = line.strip()
if not line or line.startswith("#"):
continue
parts = line.split("\t")
if len(parts) != 7:
continue
domain, flag, path, secure, expiry, name, value = parts
cookie = {
"name": name,
"value": value,
"domain": domain,
"path": path,
"secure": secure.upper() == "TRUE",
"sameSite": "Lax",
}
if expiry.isdigit() and int(expiry) > 0:
cookie["expires"] = int(expiry)
cookies.append(cookie)
return cookies