2026-09-01 13:13:40 +08:00
|
|
|
|
import re
|
|
|
|
|
|
import unicodedata
|
|
|
|
|
|
from pathlib import Path
|
|
|
|
|
|
from time import strftime, localtime
|
|
|
|
|
|
from typing import List
|
|
|
|
|
|
|
|
|
|
|
|
|
2026-09-03 00:44:38 +08:00
|
|
|
|
def get_data_dir() -> Path:
|
|
|
|
|
|
"""插件 data 目录:.../nonebot_plugin_video_analysis/data
|
|
|
|
|
|
|
|
|
|
|
|
注意 utils.py 位于插件根目录,.parent 即插件根,因此这里写死回到
|
|
|
|
|
|
插件自身 data/(cookies.txt / list.json / temp 都在此处)。
|
|
|
|
|
|
"""
|
|
|
|
|
|
return Path(__file__).resolve().parent / "data"
|
|
|
|
|
|
|
|
|
|
|
|
|
2026-09-01 13:13:40 +08:00
|
|
|
|
def get_temp_root(sub: str = "") -> Path:
|
|
|
|
|
|
"""插件媒体临时目录:data/temp[/sub] — 下载的媒体先进这里
|
|
|
|
|
|
|
|
|
|
|
|
发送时优先直接用这里的本地文件,失败才走 S3 链接(见 handlers/sender.py)。
|
|
|
|
|
|
"""
|
|
|
|
|
|
root = Path(__file__).resolve().parent.parent.parent / "data" / "temp"
|
|
|
|
|
|
if sub:
|
|
|
|
|
|
root = root / sub
|
|
|
|
|
|
root.mkdir(parents=True, exist_ok=True)
|
|
|
|
|
|
return root
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def slugify(text: str, max_length: int = 80) -> str:
|
|
|
|
|
|
"""
|
|
|
|
|
|
将字符串转换为 URL slug
|
|
|
|
|
|
|
|
|
|
|
|
规则:
|
|
|
|
|
|
1. 去掉 #tag
|
|
|
|
|
|
2. Unicode 归一化
|
|
|
|
|
|
3. 转小写
|
|
|
|
|
|
4. 空白和分隔符替换为 -
|
|
|
|
|
|
5. 移除非法字符
|
|
|
|
|
|
6. 合并连续 -
|
|
|
|
|
|
7. 裁剪长度
|
|
|
|
|
|
"""
|
|
|
|
|
|
|
|
|
|
|
|
# 去掉 #标签
|
|
|
|
|
|
text = re.sub(r"#\S+", "", text)
|
|
|
|
|
|
|
|
|
|
|
|
# Unicode 标准化
|
|
|
|
|
|
text = unicodedata.normalize("NFKC", text)
|
|
|
|
|
|
|
|
|
|
|
|
# 转小写
|
|
|
|
|
|
text = text.lower()
|
|
|
|
|
|
|
|
|
|
|
|
# 空白字符 -> -
|
|
|
|
|
|
text = re.sub(r"\s+", "-", text)
|
|
|
|
|
|
|
|
|
|
|
|
# 允许:中文、字母、数字、-
|
|
|
|
|
|
text = re.sub(r"[^\w\-一-鿿]", "", text)
|
|
|
|
|
|
|
|
|
|
|
|
# 合并多个 -
|
|
|
|
|
|
text = re.sub(r"-{2,}", "-", text)
|
|
|
|
|
|
|
|
|
|
|
|
# 去掉首尾 -
|
|
|
|
|
|
text = text.strip("-")
|
|
|
|
|
|
|
|
|
|
|
|
# 控制长度
|
|
|
|
|
|
if len(text) > max_length:
|
|
|
|
|
|
text = text[:max_length].rstrip("-")
|
|
|
|
|
|
|
|
|
|
|
|
return text
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def ensure_unique_path(base_path: Path) -> Path:
|
|
|
|
|
|
"""
|
|
|
|
|
|
确保路径不冲突:如已存在则追加 _2, _3... 后缀
|
|
|
|
|
|
适用于文件和目录
|
|
|
|
|
|
"""
|
|
|
|
|
|
if not base_path.exists():
|
|
|
|
|
|
return base_path
|
|
|
|
|
|
|
|
|
|
|
|
parent = base_path.parent
|
|
|
|
|
|
stem = base_path.stem
|
|
|
|
|
|
ext = base_path.suffix # 目录无后缀 -> ""
|
|
|
|
|
|
|
|
|
|
|
|
counter = 2
|
|
|
|
|
|
while True:
|
|
|
|
|
|
new_path = parent / f"{stem}_{counter}{ext}"
|
|
|
|
|
|
if not new_path.exists():
|
|
|
|
|
|
return new_path
|
|
|
|
|
|
counter += 1
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def clean_filename(filename: str, max_length: int = 120) -> str:
|
|
|
|
|
|
"""
|
|
|
|
|
|
清理文件名并添加时间前缀
|
|
|
|
|
|
支持多扩展名,如 .tar.gz
|
|
|
|
|
|
"""
|
|
|
|
|
|
|
|
|
|
|
|
current_time = strftime("%H-%M-%S", localtime())
|
|
|
|
|
|
|
|
|
|
|
|
p = Path(filename)
|
|
|
|
|
|
|
|
|
|
|
|
# 主文件名
|
|
|
|
|
|
name = p.stem
|
|
|
|
|
|
|
|
|
|
|
|
# 完整扩展名 (.tar.gz)
|
|
|
|
|
|
ext = "".join(p.suffixes)
|
|
|
|
|
|
|
|
|
|
|
|
# Unicode 标准化
|
|
|
|
|
|
name = unicodedata.normalize("NFKC", name)
|
|
|
|
|
|
|
|
|
|
|
|
# 去掉 #tag
|
|
|
|
|
|
name = re.sub(r"#\S+", "", name)
|
|
|
|
|
|
|
|
|
|
|
|
# 非法字符替换
|
|
|
|
|
|
name = re.sub(r'[\\/:*?"<>|]', "_", name)
|
|
|
|
|
|
|
|
|
|
|
|
# 中英文标点
|
|
|
|
|
|
name = re.sub(r"[&'\"。,:?!《》【】|]", "_", name)
|
|
|
|
|
|
|
|
|
|
|
|
# 空白 -> _
|
|
|
|
|
|
name = re.sub(r"\s+", "_", name)
|
|
|
|
|
|
|
|
|
|
|
|
# 只保留:中文、字母、数字、_
|
|
|
|
|
|
name = re.sub(r"[^\w一-鿿_]", "", name)
|
|
|
|
|
|
|
|
|
|
|
|
# 合并 _
|
|
|
|
|
|
name = re.sub(r"_+", "_", name)
|
|
|
|
|
|
|
|
|
|
|
|
# 去首尾 _
|
|
|
|
|
|
name = name.strip("_")
|
|
|
|
|
|
|
|
|
|
|
|
# 长度控制
|
|
|
|
|
|
max_name_length = max_length - len(ext) - len(current_time) - 1
|
|
|
|
|
|
if len(name) > max_name_length:
|
|
|
|
|
|
name = name[:max_name_length].rstrip("_")
|
|
|
|
|
|
|
|
|
|
|
|
return f"{current_time}_{name}{ext}"
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def parse_netscape_cookies(file_path: str) -> List[dict]:
|
|
|
|
|
|
"""解析 Netscape 格式 cookies 文件"""
|
|
|
|
|
|
cookies = []
|
|
|
|
|
|
with open(file_path, "r", encoding="utf-8") as f:
|
|
|
|
|
|
for line in f:
|
|
|
|
|
|
line = line.strip()
|
|
|
|
|
|
if not line or line.startswith("#"):
|
|
|
|
|
|
continue
|
|
|
|
|
|
parts = line.split("\t")
|
|
|
|
|
|
if len(parts) != 7:
|
|
|
|
|
|
continue
|
|
|
|
|
|
domain, flag, path, secure, expiry, name, value = parts
|
|
|
|
|
|
cookie = {
|
|
|
|
|
|
"name": name,
|
|
|
|
|
|
"value": value,
|
|
|
|
|
|
"domain": domain,
|
|
|
|
|
|
"path": path,
|
|
|
|
|
|
"secure": secure.upper() == "TRUE",
|
|
|
|
|
|
"sameSite": "Lax",
|
|
|
|
|
|
}
|
|
|
|
|
|
if expiry.isdigit() and int(expiry) > 0:
|
|
|
|
|
|
cookie["expires"] = int(expiry)
|
|
|
|
|
|
cookies.append(cookie)
|
|
|
|
|
|
return cookies
|