Files
HeXi/hexi/plugins/nonebot_plugin_video_analysis/utils.py
T

154 lines
3.7 KiB
Python
Raw Normal View History

import re
import unicodedata
from pathlib import Path
from time import strftime, localtime
from typing import List
def get_temp_root(sub: str = "") -> Path:
"""插件媒体临时目录:data/temp[/sub] — 下载的媒体先进这里
发送时优先直接用这里的本地文件,失败才走 S3 链接(见 handlers/sender.py)。
"""
root = Path(__file__).resolve().parent.parent.parent / "data" / "temp"
if sub:
root = root / sub
root.mkdir(parents=True, exist_ok=True)
return root
def slugify(text: str, max_length: int = 80) -> str:
"""
将字符串转换为 URL slug
规则:
1. 去掉 #tag
2. Unicode 归一化
3. 转小写
4. 空白和分隔符替换为 -
5. 移除非法字符
6. 合并连续 -
7. 裁剪长度
"""
# 去掉 #标签
text = re.sub(r"#\S+", "", text)
# Unicode 标准化
text = unicodedata.normalize("NFKC", text)
# 转小写
text = text.lower()
# 空白字符 -> -
text = re.sub(r"\s+", "-", text)
# 允许:中文、字母、数字、-
text = re.sub(r"[^\w\-一-鿿]", "", text)
# 合并多个 -
text = re.sub(r"-{2,}", "-", text)
# 去掉首尾 -
text = text.strip("-")
# 控制长度
if len(text) > max_length:
text = text[:max_length].rstrip("-")
return text
def ensure_unique_path(base_path: Path) -> Path:
"""
确保路径不冲突:如已存在则追加 _2, _3... 后缀
适用于文件和目录
"""
if not base_path.exists():
return base_path
parent = base_path.parent
stem = base_path.stem
ext = base_path.suffix # 目录无后缀 -> ""
counter = 2
while True:
new_path = parent / f"{stem}_{counter}{ext}"
if not new_path.exists():
return new_path
counter += 1
def clean_filename(filename: str, max_length: int = 120) -> str:
"""
清理文件名并添加时间前缀
支持多扩展名,如 .tar.gz
"""
current_time = strftime("%H-%M-%S", localtime())
p = Path(filename)
# 主文件名
name = p.stem
# 完整扩展名 (.tar.gz)
ext = "".join(p.suffixes)
# Unicode 标准化
name = unicodedata.normalize("NFKC", name)
# 去掉 #tag
name = re.sub(r"#\S+", "", name)
# 非法字符替换
name = re.sub(r'[\\/:*?"<>|]', "_", name)
# 中英文标点
name = re.sub(r"[&'\"。,:?!《》【】|]", "_", name)
# 空白 -> _
name = re.sub(r"\s+", "_", name)
# 只保留:中文、字母、数字、_
name = re.sub(r"[^\w一-鿿_]", "", name)
# 合并 _
name = re.sub(r"_+", "_", name)
# 去首尾 _
name = name.strip("_")
# 长度控制
max_name_length = max_length - len(ext) - len(current_time) - 1
if len(name) > max_name_length:
name = name[:max_name_length].rstrip("_")
return f"{current_time}_{name}{ext}"
def parse_netscape_cookies(file_path: str) -> List[dict]:
"""解析 Netscape 格式 cookies 文件"""
cookies = []
with open(file_path, "r", encoding="utf-8") as f:
for line in f:
line = line.strip()
if not line or line.startswith("#"):
continue
parts = line.split("\t")
if len(parts) != 7:
continue
domain, flag, path, secure, expiry, name, value = parts
cookie = {
"name": name,
"value": value,
"domain": domain,
"path": path,
"secure": secure.upper() == "TRUE",
"sameSite": "Lax",
}
if expiry.isdigit() and int(expiry) > 0:
cookie["expires"] = int(expiry)
cookies.append(cookie)
return cookies