import hashlib import re import time import unicodedata from pathlib import Path from typing import List, Optional, Tuple def get_data_dir() -> Path: """插件 data 目录:.../nonebot_plugin_video_analysis/data 注意 utils.py 位于插件根目录,.parent 即插件根,因此这里写死回到 插件自身 data/(cookies.txt / list.json / temp 都在此处)。 """ return Path(__file__).resolve().parent / "data" def get_temp_root(sub: str = "") -> Path: """插件媒体临时目录:data/temp[/sub] — 下载的媒体先进这里 发送时优先直接用这里的本地文件,失败才走 S3 链接(见 handlers/sender.py)。 """ root = Path(__file__).resolve().parent.parent.parent / "data" / "temp" if sub: root = root / sub root.mkdir(parents=True, exist_ok=True) return root def slugify(text: str, max_length: int = 80) -> str: """ 将字符串转换为 URL slug 规则: 1. 去掉 #tag 2. Unicode 归一化 3. 转小写 4. 空白和分隔符替换为 - 5. 移除非法字符 6. 合并连续 - 7. 裁剪长度 """ # 去掉 #标签 text = re.sub(r"#\S+", "", text) # Unicode 标准化 text = unicodedata.normalize("NFKC", text) # 转小写 text = text.lower() # 空白字符 -> - text = re.sub(r"\s+", "-", text) # 允许:中文、字母、数字、- text = re.sub(r"[^\w\-一-鿿]", "", text) # 合并多个 - text = re.sub(r"-{2,}", "-", text) # 去掉首尾 - text = text.strip("-") # 控制长度 if len(text) > max_length: text = text[:max_length].rstrip("-") return text def ensure_unique_path(base_path: Path) -> Path: """ 确保路径不冲突:如已存在则追加 _2, _3... 后缀 适用于文件和目录 注:媒体落盘已统一走 `unique_media_path`(同名加 4 位短码),本函数仅作 通用兜底保留。 """ if not base_path.exists(): return base_path parent = base_path.parent stem = base_path.stem ext = base_path.suffix # 目录无后缀 -> "" counter = 2 while True: new_path = parent / f"{stem}_{counter}{ext}" if not new_path.exists(): return new_path counter += 1 _B36_ALPHABET = "0123456789abcdefghijklmnopqrstuvwxyz" #: 同名短码的模数 —— base36 四位(36**4 ≈ 168 万秒 ≈ 19.4 天一轮) _B36_MOD = 36**4 def _to_base36(value: int, width: int = 4) -> str: """整数 → 定宽 base36(不足左侧补 0)""" if value <= 0: return "0" * width digits = [] while value: value, rem = divmod(value, 36) digits.append(_B36_ALPHABET[rem]) return "".join(reversed(digits)).rjust(width, "0") def short_time_code(at: Optional[float] = None) -> str: """4 位 base36 短码(同名兜底用,不可读时间,仅作区分码)""" return _to_base36(int(time.time() if at is None else at) % _B36_MOD) #: 各平台把"没有 id"写成过这些值,别让它们进目录名 _JUNK_AUTHOR_IDS = {"", "0", "none", "null", "na", "nan", "undefined"} #: 抓取层拿不到昵称时的兜底串(slugify 后)—— 它们等于"没有作者信息" _PLACEHOLDER_AUTHORS = {"未知作者", "小红书用户", "b站用户"} def _clean_author_id(author_id: Optional[str]) -> str: """作者 id 归一:剔除占位值(yt-dlp 缺字段常给 `NA`),再 slugify。""" raw = str(author_id or "").strip() if raw.lower() in _JUNK_AUTHOR_IDS: return "" return slugify(raw, max_length=40) def short_source_code(source: str) -> str: """来源串(作品 URL/id)→ 4 位 base36 短码。 与 `short_time_code` 的区别:同一个来源永远得到同一个码(md5 取摘要, 不能用内置 `hash()` —— 它有随机盐,重启后目录名会变)。 """ digest = hashlib.md5(source.encode("utf-8")).digest() return _to_base36(int.from_bytes(digest[:4], "big") % _B36_MOD) def build_author_dir( author: Optional[str], author_id: Optional[str] = None, *, source: Optional[str] = None, at: Optional[float] = None, ) -> str: """`{作者}_{作者id}` 作者目录名(拿不到 id 时追加码值避免同名混目录)。 - 昵称 + 作者 id → `{昵称}_{id}`(昵称上限 30、id 上限 40;**不能**把 id 截到 20:`MS4wLjABAAAA…` 这类 sec_uid 公共前缀就有 17 字符,再截断必撞车) - 只有昵称 → `{昵称}_{4 位时间短码}`:没有 id 就分不清同名作者,宁可不聚合 (同一作者的不同作品会各成一个目录)也不能把两个人混进同一个目录 - 连昵称都没有 → `未知作者_{来源短码}`(`source` 给作品 URL/id,同一来源 稳定、不同来源不撞);连 source 都没有 → 退化成时间短码 """ slug_author = slugify(author or "", max_length=30) if slug_author in _PLACEHOLDER_AUTHORS: slug_author = "" slug_id = _clean_author_id(author_id) if slug_author and slug_id: # 某些站点上传者名就是 handle(X 的 @someone),别产出 someone_someone if slug_author == slug_id: return slug_author return f"{slug_author}_{slug_id}" if slug_id: return f"{slug_author or '未知作者'}_{slug_id}" if slug_author: return f"{slug_author}_{short_time_code(at)}" if source: return f"未知作者_{short_source_code(source)}" return f"未知作者_{short_time_code(at)}" def build_work_stem(title: Optional[str]) -> str: """作品名做文件名/子目录名:slugify(沿用 15 字上限),空则 `作品`""" return slugify(title or "", max_length=15) or "作品" def unique_media_path(path: Path, *, at: Optional[float] = None) -> Path: """同名才加 4 位短码:`{stem}_{码}{后缀}`,仍撞则再叠 `_2/_3…` 文件与目录通用(目录无后缀)。命名发生在落盘前,因此"不存在"即直接采用; 顺带确保父目录存在(作者目录是按需创建的)。 """ if path.exists(): code = short_time_code(at) candidate = path.with_name(f"{path.stem}_{code}{path.suffix}") index = 2 while candidate.exists(): candidate = path.with_name(f"{path.stem}_{code}_{index}{path.suffix}") index += 1 path = candidate path.parent.mkdir(parents=True, exist_ok=True) return path def _temp_rel_parts(file_path: Path | str) -> Tuple[str, ...]: """相对 temp 根拆路径:`{平台}/{作者目录}/…`(归档目录同理) 不在 temp 下、或没到"平台 + 作者目录"这一层(老数据 / 第三方产物)→ 空元组, 调用方退化为只用文件名。 """ try: rel = Path(file_path).resolve().relative_to(get_temp_root().resolve()) except (ValueError, OSError): return () parts = rel.parts return parts[1:] if len(parts) >= 3 else () def media_key_of(file_path: Path | str) -> str: """媒体文件的 S3 对象 key:`{作者}_{作者id}/{作品名}[_{码}].后缀` 由本地路径反推(去掉平台层),保证桶里和 temp 里结构一致。 """ parts = _temp_rel_parts(file_path) return "/".join(parts) if parts else Path(file_path).name def media_rel_dir_of(file_path: Path | str) -> str: """媒体文件所在的作者/作品子目录(相对 temp 根、去掉平台层),供归档复用""" parts = _temp_rel_parts(file_path) return "/".join(parts[:-1]) if len(parts) > 1 else "" def parse_netscape_cookies(file_path: str) -> List[dict]: """解析 Netscape 格式 cookies 文件""" cookies = [] with open(file_path, "r", encoding="utf-8") as f: for line in f: line = line.strip() if not line or line.startswith("#"): continue parts = line.split("\t") if len(parts) != 7: continue domain, flag, path, secure, expiry, name, value = parts cookie = { "name": name, "value": value, "domain": domain, "path": path, "secure": secure.upper() == "TRUE", "sameSite": "Lax", } if expiry.isdigit() and int(expiry) > 0: cookie["expires"] = int(expiry) cookies.append(cookie) return cookies