Add HeXi bot codebase: custom plugins, web frontends, tests
- hexi core: message handling, rate limiting, cooldown, plugin manager - Custom plugins: BF stats, daily check-in, quotes, persona cards, etc. - Community plugins vendored under hexi/plugins with local fixes - Web admin frontends (learning-chat, persona-admin), unified hexi/web - Tests for rate_limit/cooldown/memes/persona; poetry.lock Co-Authored-By: Claude <noreply@anthropic.com>
This commit is contained in:
@@ -0,0 +1,22 @@
|
||||
"""视频解析插件 — 入口(元数据 + 依赖 + 配置注册)。
|
||||
|
||||
业务入口在 handlers/entry.py,配置注册在 config.py。
|
||||
"""
|
||||
|
||||
from nonebot import require
|
||||
from nonebot.plugin import PluginMetadata
|
||||
|
||||
require("nonebot_plugin_alconna")
|
||||
|
||||
__plugin_meta__ = PluginMetadata(
|
||||
name="视频解析",
|
||||
description="解析分享的抖音/哔哩哔哩/YouTube等视频链接",
|
||||
usage="发送视频分享链接自动解析;@bot 时主动触发解析",
|
||||
type="application",
|
||||
)
|
||||
|
||||
# 显式导入子模块:注册配置 schema + 消息 matcher(配合 load_plugins 只加载到包层)
|
||||
from . import config as _config # noqa: E402
|
||||
from . import handlers as _handlers # noqa: E402
|
||||
|
||||
_config.register_config()
|
||||
@@ -0,0 +1,210 @@
|
||||
"""temp 目录清理 — 手动 / 自动可选
|
||||
|
||||
hexi/data/temp 是下载媒体中转区(见 utils.get_temp_root),发送成功后保留
|
||||
供用户取用。清理方式由配置 `video_analysis_temp_cleanup_mode` 决定:
|
||||
|
||||
- auto : 每日定时清理,删除超过配置天数未修改的文件(默认)
|
||||
- manual : 关闭定时任务,仅通过命令手动清理
|
||||
|
||||
命令(manual 模式下可用):
|
||||
- /清理temp [天数] 清理 temp 下超过 N 天(默认取配置)未修改的文件
|
||||
- /temp统计 查看 temp 目录占用情况
|
||||
|
||||
目录结构始终保留,占用中的文件自动跳过。
|
||||
"""
|
||||
|
||||
import asyncio
|
||||
import time
|
||||
from pathlib import Path
|
||||
from typing import Literal
|
||||
|
||||
from nonebot import logger, require, get_plugin_config, on_command
|
||||
from nonebot.adapters.onebot.v11 import Bot, MessageEvent
|
||||
from pydantic import BaseModel
|
||||
|
||||
from hexi.plugins.nonebot_plugin_hexi_core.custom_utils import check_manage
|
||||
|
||||
require("nonebot_plugin_apscheduler")
|
||||
from nonebot_plugin_apscheduler import scheduler
|
||||
|
||||
from nonebot_plugin_alconna import UniMessage
|
||||
|
||||
from .utils import get_temp_root
|
||||
|
||||
|
||||
class CleanupConfig(BaseModel):
|
||||
# 清理模式:auto=每日定时自动清理, manual=仅命令手动清理
|
||||
video_analysis_temp_cleanup_mode: Literal["auto", "manual"] = "auto"
|
||||
# 超过该天数未修改的 temp 文件会被清理(默认 7 天)
|
||||
video_analysis_temp_max_age_days: int = 7
|
||||
# 自动模式下每日清理时间(小时,0-23,默认凌晨 4 点)
|
||||
video_analysis_temp_cleanup_hour: int = 4
|
||||
|
||||
|
||||
cleanup_config = get_plugin_config(CleanupConfig)
|
||||
|
||||
|
||||
def _max_age_seconds(days: int | None = None) -> float:
|
||||
"""由天数配置换算为秒;非法值回落默认 7 天"""
|
||||
if days is None:
|
||||
days = cleanup_config.video_analysis_temp_max_age_days
|
||||
try:
|
||||
days = max(0, int(days))
|
||||
except (TypeError, ValueError):
|
||||
days = 7
|
||||
return days * 24 * 60 * 60
|
||||
|
||||
|
||||
def _file_is_stale(path: Path, max_age: float) -> bool:
|
||||
"""文件最后修改时间距今是否超过 max_age 秒"""
|
||||
try:
|
||||
return time.time() - path.stat().st_mtime > max_age
|
||||
except OSError:
|
||||
# 文件已被删除或不可访问 → 视为可清理(下一轮 unlink 会跳过)
|
||||
return True
|
||||
|
||||
|
||||
def _walk_files(root: Path) -> list[Path]:
|
||||
"""收集 root 下所有文件,按修改时间倒序(最新的在前)"""
|
||||
files = [p for p in root.rglob("*") if p.is_file()]
|
||||
files.sort(key=lambda p: p.stat().st_mtime, reverse=True)
|
||||
return files
|
||||
|
||||
|
||||
def clean_temp_files(sub: str = "", days: int | None = None) -> tuple[int, int]:
|
||||
"""清理 temp[/sub] 下超过期限的文件。
|
||||
|
||||
Returns:
|
||||
(removed, total) — 删除数、统计到的文件总数
|
||||
"""
|
||||
root = get_temp_root(sub)
|
||||
if not root.is_dir():
|
||||
return 0, 0
|
||||
|
||||
max_age = _max_age_seconds(days)
|
||||
files = _walk_files(root)
|
||||
removed = 0
|
||||
for path in files:
|
||||
if not _file_is_stale(path, max_age):
|
||||
continue
|
||||
try:
|
||||
path.unlink(missing_ok=True)
|
||||
removed += 1
|
||||
logger.info(f"temp 清理: 删除 {path}")
|
||||
except OSError as e:
|
||||
# 文件被占用(如发送中)等场景,留待下轮
|
||||
logger.warning(f"temp 清理: 跳过 {path} ({e})")
|
||||
return removed, len(files)
|
||||
|
||||
|
||||
def temp_stats(sub: str = "") -> dict:
|
||||
"""统计 temp[/sub] 目录:文件数、总大小(字节)"""
|
||||
root = get_temp_root(sub)
|
||||
if not root.is_dir():
|
||||
return {"files": 0, "bytes": 0}
|
||||
files = _walk_files(root)
|
||||
total_bytes = sum(p.stat().st_size for p in files if p.exists())
|
||||
return {"files": len(files), "bytes": total_bytes}
|
||||
|
||||
|
||||
# ── 手动清理命令(manual 模式,auto 模式下也可用) ──────────────
|
||||
# 清理是删除操作,要求管理及以上(check_manage:群主/群管理/超管)
|
||||
clean_cmd = on_command("清理temp", aliases={"清理临时文件", "清temp"}, priority=10, block=True)
|
||||
# 统计只读,对所有人开放
|
||||
stats_cmd = on_command("temp统计", aliases={"temp状态"}, priority=10, block=True)
|
||||
|
||||
|
||||
@clean_cmd.handle()
|
||||
async def _handle_clean(bot: Bot, event: MessageEvent):
|
||||
# 鉴权:仅群主/群管理/超管可清理
|
||||
if not await check_manage(bot, event):
|
||||
await UniMessage.text("只有管理以上才能清理 temp 哦~").send(at_sender=True)
|
||||
return
|
||||
# 解析可选天数参数:/清理temp 3 → 清理 3 天前的文件
|
||||
text = str(event.message).strip()
|
||||
tokens = text.replace("/", " ").split()
|
||||
days = None
|
||||
if len(tokens) >= 2:
|
||||
try:
|
||||
days = int(tokens[1])
|
||||
except ValueError:
|
||||
await UniMessage.text("天数参数不合法,示例:/清理temp 3").send()
|
||||
return
|
||||
removed, total = await _run_clean(days)
|
||||
if total == 0:
|
||||
await UniMessage.text("temp 目录目前是空的,没什么可清理的~").send()
|
||||
else:
|
||||
await UniMessage.text(f"temp 清理完成:删除了 {removed} 个文件(共 {total} 个文件)。").send()
|
||||
|
||||
|
||||
@stats_cmd.handle()
|
||||
async def _handle_stats(event: MessageEvent):
|
||||
st = await _run_stats()
|
||||
if st["files"] == 0:
|
||||
await UniMessage.text("temp 目录目前是空的。").send()
|
||||
else:
|
||||
size_mb = st["bytes"] / 1024 / 1024
|
||||
await UniMessage.text(
|
||||
f"temp 目录:共 {st['files']} 个文件,占用 {size_mb:.1f} MB。"
|
||||
).send()
|
||||
|
||||
|
||||
async def _run_clean(days: int | None = None) -> tuple[int, int]:
|
||||
"""执行清理并记录日志(供命令与定时任务共用)"""
|
||||
removed, total = await asyncio.to_thread(clean_temp_files, "", days)
|
||||
return removed, total
|
||||
|
||||
|
||||
async def _run_stats() -> dict:
|
||||
return await asyncio.to_thread(temp_stats)
|
||||
|
||||
|
||||
# ── 自动模式:每日定时清理 ──────────────────────────────────────
|
||||
|
||||
|
||||
def _register_auto_job() -> None:
|
||||
"""按当前配置注册/撤销定时清理任务(幂等,可反复调用)。
|
||||
|
||||
- auto 模式:注册每日 cron 清理任务(replace_existing 保证不重复)。
|
||||
- manual 模式:移除已注册的自动清理任务,仅保留手动命令。
|
||||
"""
|
||||
from apscheduler.jobstores.base import JobLookupError # noqa: PLC0415
|
||||
|
||||
if cleanup_config.video_analysis_temp_cleanup_mode == "auto":
|
||||
scheduler.add_job(
|
||||
_scheduled_cleanup,
|
||||
"cron",
|
||||
hour=cleanup_config.video_analysis_temp_cleanup_hour,
|
||||
minute=0,
|
||||
id="video_analysis_temp_cleanup",
|
||||
misfire_grace_time=3600,
|
||||
replace_existing=True,
|
||||
)
|
||||
logger.info(
|
||||
f"temp 自动清理已启用:每天 {cleanup_config.video_analysis_temp_cleanup_hour}:00"
|
||||
)
|
||||
else:
|
||||
try:
|
||||
scheduler.remove_job("video_analysis_temp_cleanup")
|
||||
logger.info("temp 自动清理已关闭(manual 模式)")
|
||||
except JobLookupError:
|
||||
pass
|
||||
|
||||
|
||||
def reload_cleanup_config() -> None:
|
||||
"""Web 保存清理配置后调用:按新配置重设自动清理任务。"""
|
||||
_register_auto_job()
|
||||
|
||||
|
||||
async def _scheduled_cleanup() -> None:
|
||||
"""每日定时清理 temp 目录(auto 模式)"""
|
||||
logger.info("temp 自动清理: 开始")
|
||||
try:
|
||||
removed, total = await asyncio.to_thread(clean_temp_files)
|
||||
logger.info(f"temp 自动清理完成:删除 {removed} 个,当前共 {total} 个文件")
|
||||
except Exception as e:
|
||||
logger.exception(f"temp 自动清理失败: {e}")
|
||||
|
||||
|
||||
# 插件加载时按模式注册(auto)或提示(manual)
|
||||
_register_auto_job()
|
||||
@@ -0,0 +1,104 @@
|
||||
"""统一配置注册:把插件配置按文档标准接入 hexi.web_config(Web 可读可改)。
|
||||
|
||||
- temp 清理配置(env pydantic) → register_model_config
|
||||
- S3 存储配置(原硬编码在 storage/s3.py) → register_config_items(store=s3 模块)
|
||||
- 群分组配置(data/list.json) → register_config_items(type=json, nosave)
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from hexi.config_standard import register_config_items, register_model_config
|
||||
|
||||
from . import cleanup, list_proc # noqa: F401
|
||||
from .storage import s3 as _s3mod
|
||||
|
||||
# 插件模块名 = plugin_id(与 NoneBot 模块名一致)
|
||||
_PLUGIN_ID = __package__
|
||||
|
||||
|
||||
def _reset_s3_caches(_values=None, store=None):
|
||||
"""保存 S3 配置后清空懒加载客户端缓存,让下次上传用新配置重建。"""
|
||||
if store is None:
|
||||
return
|
||||
for name in ("_local_s3_plana", "_local_s3_planb", "_local_s3_planc", "_public_s3"):
|
||||
try:
|
||||
setattr(store, name, None)
|
||||
except Exception: # noqa: BLE001
|
||||
pass
|
||||
|
||||
|
||||
def register_config() -> None:
|
||||
"""注册本插件全部配置(供 __init__.py 在导入期调用)。"""
|
||||
# 1) temp 清理配置
|
||||
register_model_config(
|
||||
_PLUGIN_ID,
|
||||
cleanup.cleanup_config,
|
||||
fields=[
|
||||
"video_analysis_temp_cleanup_mode",
|
||||
"video_analysis_temp_max_age_days",
|
||||
"video_analysis_temp_cleanup_hour",
|
||||
],
|
||||
labels={
|
||||
"video_analysis_temp_cleanup_mode": "temp 清理模式",
|
||||
"video_analysis_temp_max_age_days": "清理过期天数",
|
||||
"video_analysis_temp_cleanup_hour": "每日清理时间(时)",
|
||||
},
|
||||
descriptions={
|
||||
"video_analysis_temp_cleanup_mode": "auto=每日定时自动清理; manual=仅命令手动清理",
|
||||
"video_analysis_temp_max_age_days": "超过该天数未修改的文件会被清理",
|
||||
"video_analysis_temp_cleanup_hour": "自动模式下的每日清理小时(0-23)",
|
||||
},
|
||||
options={
|
||||
"video_analysis_temp_cleanup_mode": [
|
||||
{"value": "auto", "label": "自动(auto)"},
|
||||
{"value": "manual", "label": "手动(manual)"},
|
||||
]
|
||||
},
|
||||
types={"video_analysis_temp_cleanup_mode": "enum"},
|
||||
# 保存后按新配置重设自动定时任务
|
||||
apply_extra=lambda _values, _conf: cleanup.reload_cleanup_config(),
|
||||
)
|
||||
|
||||
# 2) S3 存储配置(来源无关)
|
||||
register_config_items(
|
||||
_PLUGIN_ID,
|
||||
[
|
||||
{"key": "LOCAL_S3_PLANA_ENDPOINT", "label": "PLANA 端点", "type": "string", "secret": True},
|
||||
{"key": "LOCAL_S3_PLANA_ACCESS_KEY", "label": "PLANA AccessKey", "type": "password", "secret": True},
|
||||
{"key": "LOCAL_S3_PLANA_SECRET_KEY", "label": "PLANA SecretKey", "type": "password", "secret": True},
|
||||
{"key": "LOCAL_S3_PLANA_BUCKET", "label": "PLANA 桶", "type": "string"},
|
||||
{"key": "LOCAL_S3_PLANB_ENDPOINT", "label": "PLANB 端点", "type": "string", "secret": True},
|
||||
{"key": "LOCAL_S3_PLANB_ACCESS_KEY", "label": "PLANB AccessKey", "type": "password", "secret": True},
|
||||
{"key": "LOCAL_S3_PLANB_SECRET_KEY", "label": "PLANB SecretKey", "type": "password", "secret": True},
|
||||
{"key": "LOCAL_S3_PLANB_BUCKET", "label": "PLANB 桶", "type": "string"},
|
||||
{"key": "LOCAL_S3_PLANC_ENDPOINT", "label": "PLANC 端点", "type": "string", "secret": True},
|
||||
{"key": "LOCAL_S3_PLANC_ACCESS_KEY", "label": "PLANC AccessKey", "type": "password", "secret": True},
|
||||
{"key": "LOCAL_S3_PLANC_SECRET_KEY", "label": "PLANC SecretKey", "type": "password", "secret": True},
|
||||
{"key": "LOCAL_S3_PLANC_BUCKET", "label": "PLANC 桶", "type": "string"},
|
||||
{"key": "PUBLIC_S3_ENDPOINT", "label": "公网端点", "type": "string"},
|
||||
{"key": "PUBLIC_S3_ACCESS_KEY", "label": "公网 AccessKey", "type": "password", "secret": True},
|
||||
{"key": "PUBLIC_S3_SECRET_KEY", "label": "公网 SecretKey", "type": "password", "secret": True},
|
||||
{"key": "PUBLIC_S3_BUCKET", "label": "公网桶", "type": "string"},
|
||||
{"key": "PUBLIC_S3_REGION", "label": "公网 Region", "type": "string"},
|
||||
{"key": "PUBLIC_S3_SECURE", "label": "公网 HTTPS", "type": "bool", "default": True},
|
||||
{"key": "PUBLIC_S3_DOMAIN", "label": "公网访问域名", "type": "string"},
|
||||
],
|
||||
store=_s3mod,
|
||||
apply_extra=_reset_s3_caches,
|
||||
)
|
||||
|
||||
# 3) 群分组配置(data/list.json, 权威源在插件自身)
|
||||
register_config_items(
|
||||
_PLUGIN_ID,
|
||||
[
|
||||
{
|
||||
"key": "group_config",
|
||||
"label": "群分组配置",
|
||||
"type": "json",
|
||||
"description": "data/list.json 内容。groups 为 {群号: {auto, plan(A/B), auto_link[]}},blacklist 为禁用用户 QQ 列表。白名单即 groups 的键。",
|
||||
"getter": list_proc.get_group_config_sync,
|
||||
"setter": list_proc.set_group_config_sync,
|
||||
"nosave": True,
|
||||
}
|
||||
],
|
||||
)
|
||||
@@ -0,0 +1,269 @@
|
||||
"""B站动态/图文/文章内容解析 — 基于 bilibili-api-python
|
||||
|
||||
借鉴 nonebot-plugin-parser 的 BilibiliParser:
|
||||
- bilibili.com/opus/{id}、bilibili.com/dynamic/{id}、t.bilibili.com/{id}
|
||||
→ 动态(图文 / 视频 / 纯文字)
|
||||
- bilibili.com/read/cv{id} → 专栏(转为图文动态解析)
|
||||
- b23.tv / bili2233.cn 短链 → 先重定向
|
||||
|
||||
返回 (title, 文件):
|
||||
- 图文动态/专栏 → (title, [图片路径列表])
|
||||
- 视频动态 → (title, 视频文件 Path,复用 yt-dlp 链路)
|
||||
- 纯文字动态 → (title, [])
|
||||
- 解析失败 → (None, None)
|
||||
"""
|
||||
|
||||
import re
|
||||
import tempfile
|
||||
from datetime import datetime
|
||||
from pathlib import Path
|
||||
from typing import Optional, Union
|
||||
|
||||
from nonebot import logger
|
||||
|
||||
from ..models import ContentFetchError
|
||||
from ..utils import parse_netscape_cookies, slugify
|
||||
|
||||
DATA_DIR = Path(__file__).resolve().parent.parent / "data"
|
||||
|
||||
BILI_UA = (
|
||||
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 "
|
||||
"(KHTML, like Gecko) Chrome/122.0.0.0 Safari/537.36"
|
||||
)
|
||||
BILI_REFERER = "https://www.bilibili.com/"
|
||||
|
||||
OPUS_RE = re.compile(r"bilibili\.com/(?:opus|dynamic)/(\d+)")
|
||||
T_BILI_RE = re.compile(r"t\.bilibili\.com/(\d+)")
|
||||
READ_RE = re.compile(r"bilibili\.com/read/cv(\d+)")
|
||||
BILI_SHORT_RE = re.compile(r"(?:b23\.tv|bili2233\.cn)/[0-9a-zA-Z._?%&+=/#]+")
|
||||
|
||||
|
||||
async def fetch_bilibili_content(
|
||||
url: str,
|
||||
) -> tuple[Optional[str], Optional[Union[Path, list[Path]]]]:
|
||||
"""解析 B站动态/图文/文章链接"""
|
||||
try:
|
||||
return await _parse(url)
|
||||
except Exception:
|
||||
logger.exception(f"B站内容解析失败: {url}")
|
||||
return None, None
|
||||
|
||||
|
||||
async def _parse(url: str):
|
||||
# 1. 短链重定向
|
||||
if "b23.tv" in url or "bili2233.cn" in url:
|
||||
resolved = await resolve_short_link(url)
|
||||
if resolved:
|
||||
logger.info(f"b23 短链重定向: {url} -> {resolved}")
|
||||
url = resolved
|
||||
|
||||
# 2. 类型与 ID 提取
|
||||
if m := READ_RE.search(url):
|
||||
return await _parse_article(int(m.group(1)))
|
||||
elif m := OPUS_RE.search(url):
|
||||
dynamic_id = int(m.group(1))
|
||||
elif m := T_BILI_RE.search(url):
|
||||
dynamic_id = int(m.group(1))
|
||||
else:
|
||||
logger.warning(f"无法识别的 B站链接: {url}")
|
||||
return None, None
|
||||
|
||||
# 3. 延迟导入 bilibili_api(未安装时不阻塞插件启动)
|
||||
from bilibili_api import request_settings, select_client
|
||||
|
||||
# 模拟浏览器指纹,避免被 B站风控限流(-509)
|
||||
select_client("curl_cffi")
|
||||
request_settings.set("impersonate", "chrome131")
|
||||
|
||||
from bilibili_api.dynamic import Dynamic
|
||||
|
||||
dynamic = Dynamic(dynamic_id, _build_credential())
|
||||
|
||||
# 4. 文章动态 → 转 opus
|
||||
if await dynamic.is_article():
|
||||
return await _parse_opus(dynamic.turn_to_opus(), "文章")
|
||||
|
||||
info = await dynamic.get_info()
|
||||
return await _parse_dynamic_info(info)
|
||||
|
||||
|
||||
async def _parse_article(read_id: int) -> tuple[str, Union[Path, list[Path]]]:
|
||||
"""专栏 cv{id} → 转为图文动态"""
|
||||
from bilibili_api.article import Article
|
||||
|
||||
# 文章接口对匿名请求风控更严(-509),必须带凭证
|
||||
article = Article(read_id, _build_credential())
|
||||
opus = await article.turn_to_opus()
|
||||
return await _parse_opus(opus, "文章")
|
||||
|
||||
|
||||
async def _parse_opus(opus, kind: str) -> tuple[str, Union[Path, list[Path]]]:
|
||||
"""图文动态/专栏解析(opus 接口返回 dict,直接访问)"""
|
||||
info = await opus.get_info()
|
||||
item = info.get("item") or {}
|
||||
basic = item.get("basic") or {}
|
||||
title = basic.get("title") or ""
|
||||
|
||||
images: list[str] = []
|
||||
texts: list[str] = []
|
||||
author = ""
|
||||
for module in item.get("modules") or []:
|
||||
if module.get("module_type") == "MODULE_TYPE_AUTHOR":
|
||||
author_info = module.get("module_author") or {}
|
||||
author = author_info.get("name", "")
|
||||
elif module.get("module_type") == "MODULE_TYPE_CONTENT":
|
||||
content = module.get("module_content") or {}
|
||||
for para in content.get("paragraphs") or []:
|
||||
if pic := (para.get("pic") or {}).get("pics"):
|
||||
images.extend(p.get("url", "") for p in pic)
|
||||
elif nodes := ((para.get("text") or {}).get("nodes")):
|
||||
if text := _extract_text(nodes):
|
||||
texts.append(text)
|
||||
|
||||
images = [u for u in images if u]
|
||||
text = title or (texts[0] if texts else "")
|
||||
logger.info(f"B站{kind}解析: 作者={author}, 标题={text[:40]}, 图片={len(images)} 张")
|
||||
|
||||
if not images:
|
||||
return text or f"B站{kind}", []
|
||||
|
||||
file_name = _build_file_name(author, text or f"B站{kind}", kind)
|
||||
file_paths = await _download_images(images, file_name)
|
||||
return text, file_paths
|
||||
|
||||
|
||||
async def _parse_dynamic_info(info: dict) -> tuple[str, Union[Path, list[Path]]]:
|
||||
"""动态解析(图文 / 视频 / 纯文字)"""
|
||||
item = info.get("item") or {}
|
||||
modules = item.get("modules") or {}
|
||||
author = ((modules.get("module_author") or {}).get("name")) or "B站用户"
|
||||
module_dynamic = modules.get("module_dynamic") or {}
|
||||
major = module_dynamic.get("major") or {}
|
||||
major_type = major.get("type", "")
|
||||
desc = ((module_dynamic.get("desc") or {}).get("text")) or ""
|
||||
|
||||
# 1. 视频动态 → 提取 bvid 走 yt-dlp
|
||||
if major_type == "MAJOR_TYPE_ARCHIVE":
|
||||
archive = major.get("archive") or {}
|
||||
bvid = archive.get("bvid")
|
||||
if bvid:
|
||||
from .video_downloader import download_video
|
||||
|
||||
title = archive.get("title") or desc or "B站视频动态"
|
||||
logger.info(f"B站视频动态: bvid={bvid} 标题={title[:40]}")
|
||||
video_path = await download_video(f"https://www.bilibili.com/video/{bvid}")
|
||||
if video_path:
|
||||
return title, video_path
|
||||
raise ContentFetchError(f"视频动态下载失败: {bvid}")
|
||||
return desc or "B站视频动态", []
|
||||
|
||||
# 2. 图文动态(新版 opus / 老版 draw)
|
||||
images: list[str] = []
|
||||
title = author
|
||||
if major_type == "MAJOR_TYPE_OPUS":
|
||||
opus = major.get("opus") or {}
|
||||
images = [pic.get("url", "") for pic in opus.get("pics") or []]
|
||||
desc = desc or ((opus.get("summary") or {}).get("text")) or ""
|
||||
title = opus.get("title") or title
|
||||
elif major_type == "MAJOR_TYPE_DRAW":
|
||||
images = [
|
||||
item.get("src", "")
|
||||
for item in (major.get("draw") or {}).get("items") or []
|
||||
]
|
||||
|
||||
images = [u for u in images if u]
|
||||
if images:
|
||||
file_name = _build_file_name(author, title, "动态")
|
||||
file_paths = await _download_images(images, file_name)
|
||||
logger.info(f"B站图文动态: 作者={author}, 标题={title[:40]}, 图片={len(images)} 张")
|
||||
return title, file_paths
|
||||
|
||||
# 3. 纯文字动态
|
||||
text = desc or "B站动态"
|
||||
logger.info(f"B站文字动态: {text[:30]}")
|
||||
return text, []
|
||||
|
||||
|
||||
def _extract_text(nodes: list) -> str:
|
||||
"""从文章段落节点提取文字"""
|
||||
parts = []
|
||||
for node in nodes or []:
|
||||
word = node.get("word") or {}
|
||||
if node.get("type") in (
|
||||
"TEXT_NODE_TYPE_WORD",
|
||||
"TEXT_NODE_TYPE_RICH",
|
||||
) and word.get("words"):
|
||||
parts.append(word["words"])
|
||||
return "".join(parts)
|
||||
|
||||
|
||||
def _build_file_name(author: str, title: str, kind: str) -> str:
|
||||
"""构建文件名 stem: {作者}_{标题}_{类型}_{时间}"""
|
||||
slug_author = slugify(author)
|
||||
slug_title = slugify(title or "", max_length=15)
|
||||
if not slug_title:
|
||||
slug_title = datetime.now().strftime("%H%M%S")
|
||||
time_suffix = datetime.now().strftime("%H%M%S")
|
||||
return f"{slug_author}_{slug_title}_{kind}_{time_suffix}"
|
||||
|
||||
|
||||
def _build_credential():
|
||||
"""从 cookies.txt 构建 B站凭证(无 SESSDATA 时返回 None 匿名访问)"""
|
||||
from bilibili_api import Credential
|
||||
|
||||
cookies_path = DATA_DIR / "cookies.txt"
|
||||
if not cookies_path.exists():
|
||||
return None
|
||||
cookies = parse_netscape_cookies(str(cookies_path))
|
||||
ck = {c["name"]: c["value"] for c in cookies}
|
||||
sessdata = ck.get("SESSDATA")
|
||||
if not sessdata:
|
||||
logger.warning("cookies.txt 中无 SESSDATA,B站将以匿名身份访问")
|
||||
return None
|
||||
return Credential(
|
||||
sessdata=sessdata,
|
||||
bili_jct=ck.get("bili_jct", ""),
|
||||
buvid3=ck.get("buvid3", ""),
|
||||
)
|
||||
|
||||
|
||||
async def _download_images(image_urls: list[str], file_name: str) -> list[Path]:
|
||||
"""并发下载图片(复用抖音图文的下载流程)"""
|
||||
import httpx
|
||||
|
||||
from .douyin_api import _process_note_with_parsed
|
||||
|
||||
tmp_root = Path(tempfile.gettempdir()) / "bilibili"
|
||||
tmp_root.mkdir(parents=True, exist_ok=True)
|
||||
headers = {
|
||||
"Referer": BILI_REFERER,
|
||||
"User-Agent": BILI_UA,
|
||||
}
|
||||
return await _process_note_with_parsed(
|
||||
[[u] for u in image_urls], None, tmp_root, file_name, headers
|
||||
)
|
||||
|
||||
|
||||
async def resolve_short_link(url: str) -> Optional[str]:
|
||||
"""b23 短链重定向(httpx 取 Location,最多 3 跳)"""
|
||||
import httpx
|
||||
|
||||
try:
|
||||
async with httpx.AsyncClient(
|
||||
headers={"User-Agent": BILI_UA},
|
||||
follow_redirects=False,
|
||||
timeout=10,
|
||||
) as client:
|
||||
current = url
|
||||
for _ in range(3):
|
||||
resp = await client.get(current)
|
||||
if resp.status_code >= 400:
|
||||
return None
|
||||
location = resp.headers.get("Location")
|
||||
if not location:
|
||||
return str(resp.url)
|
||||
current = location
|
||||
return current
|
||||
except Exception:
|
||||
logger.warning(f"b23 短链重定向失败: {url}")
|
||||
return None
|
||||
@@ -0,0 +1,425 @@
|
||||
"""抖音内容抓取 — 浏览器自动化 + API 拦截 + 媒体下载"""
|
||||
|
||||
import asyncio
|
||||
import subprocess
|
||||
from pathlib import Path
|
||||
from typing import Dict, List, Optional
|
||||
|
||||
import httpx
|
||||
from nonebot import logger
|
||||
from playwright.async_api import async_playwright
|
||||
|
||||
from ..models import DouyinFetchError
|
||||
from ..utils import ensure_unique_path, get_temp_root
|
||||
from .douyin_parser import (
|
||||
ParsedDouyinContent,
|
||||
extract_trailing_digits,
|
||||
is_animated_note,
|
||||
parse_animated_note_videos,
|
||||
parse_douyin_response,
|
||||
parse_note_images,
|
||||
parse_ssr_page,
|
||||
parse_video_urls,
|
||||
)
|
||||
|
||||
|
||||
def merge_video_audio(video_file: Path, audio_file: Path, out_file: Path):
|
||||
"""分轨视频与音频合并"""
|
||||
subprocess.run(
|
||||
[
|
||||
"ffmpeg",
|
||||
"-i",
|
||||
str(video_file),
|
||||
"-i",
|
||||
str(audio_file),
|
||||
"-c:v",
|
||||
"copy",
|
||||
"-c:a",
|
||||
"copy",
|
||||
"-movflags",
|
||||
"faststart",
|
||||
"-y",
|
||||
str(out_file),
|
||||
],
|
||||
check=True,
|
||||
)
|
||||
|
||||
|
||||
# ============================= 主入口 =============================
|
||||
|
||||
|
||||
async def fetch_douyin_content(
|
||||
douyin_url: str,
|
||||
cookies: List[dict],
|
||||
wait_seconds: int = 10,
|
||||
headless: bool = True,
|
||||
) -> tuple[str | None, Path | List[Path] | None]:
|
||||
"""
|
||||
获取抖音内容(视频或图文)
|
||||
|
||||
返回:
|
||||
(title, content)
|
||||
- 视频: (title, Path)
|
||||
- 图文: (title, List[Path])
|
||||
- 失败: (None, None)
|
||||
"""
|
||||
if not cookies:
|
||||
raise DouyinFetchError("cookies 为空")
|
||||
|
||||
tmp_root = get_temp_root("douyin")
|
||||
api_response: Optional[dict] = None
|
||||
api_response_favorite: Optional[dict] = None
|
||||
aweme_id = 0
|
||||
# 目标作品 id(从入口 URL 提取):aweme/post 返回的是作者作品列表,
|
||||
# 按此 id 精确匹配要解析的作品,避免取到作者的其他作品
|
||||
target_aweme_id = extract_trailing_digits(douyin_url) or ""
|
||||
referer_url = None
|
||||
|
||||
async with async_playwright() as p:
|
||||
browser = await p.chromium.launch(
|
||||
executable_path="C:/Program Files/Google/Chrome/Application/chrome.exe",
|
||||
headless=headless,
|
||||
args=[
|
||||
"--autoplay-policy=no-user-gesture-required",
|
||||
"--disable-features=AutoplayDisableSuppression",
|
||||
],
|
||||
)
|
||||
|
||||
context = await browser.new_context(
|
||||
viewport={"width": 1280, "height": 720},
|
||||
device_scale_factor=2,
|
||||
user_agent=(
|
||||
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) "
|
||||
"AppleWebKit/537.36 (KHTML, like Gecko) "
|
||||
"Chrome/122.0.0.0 Safari/537.36"
|
||||
),
|
||||
locale="zh-CN",
|
||||
)
|
||||
|
||||
await context.add_cookies(cookies)
|
||||
page = await context.new_page()
|
||||
|
||||
async def handle_response(response):
|
||||
nonlocal api_response, api_response_favorite, aweme_id, referer_url
|
||||
try:
|
||||
if (
|
||||
"https://www.douyin.com/note" in response.url
|
||||
or "https://www.douyin.com/video" in response.url
|
||||
):
|
||||
referer_url = response.url
|
||||
logger.info(f"作品链接:{referer_url}")
|
||||
aweme_id = extract_trailing_digits(response.url)
|
||||
logger.info(f"作品id:{aweme_id}")
|
||||
|
||||
if "aweme/v1/web/aweme/detail" in response.url:
|
||||
if not api_response:
|
||||
logger.info(f"捕获到视频 API: {response.url}")
|
||||
ct = response.headers.get("content-type", "")
|
||||
if "json" in ct:
|
||||
api_response = await response.json()
|
||||
logger.info("API 响应已捕获")
|
||||
return
|
||||
else:
|
||||
logger.info("API 已有捕获")
|
||||
return
|
||||
|
||||
elif "aweme/v1/web/aweme/post" in response.url:
|
||||
logger.info(f"捕获到图文 API: {response.url}")
|
||||
ct = response.headers.get("content-type", "")
|
||||
if "json" in ct:
|
||||
data = await response.json()
|
||||
aweme_list = data.get("aweme_list", [])
|
||||
if aweme_list:
|
||||
# 列表是作者的全部作品,优先按目标作品 id 精确匹配
|
||||
# (未登录/风控时目标作品可能不在列表里)
|
||||
match_id = aweme_id or target_aweme_id
|
||||
target_aweme = None
|
||||
if match_id:
|
||||
target_aweme = next(
|
||||
(
|
||||
a
|
||||
for a in aweme_list
|
||||
if str(a.get("aweme_id")) == str(match_id)
|
||||
),
|
||||
None,
|
||||
)
|
||||
if not target_aweme:
|
||||
target_aweme = aweme_list[0]
|
||||
if target_aweme.get("images"):
|
||||
logger.info("找到正确的图文 API 响应(包含图片数据)")
|
||||
data["aweme_list"] = [target_aweme]
|
||||
api_response = data
|
||||
return
|
||||
else:
|
||||
logger.warning(
|
||||
"此响应不包含有效的图片数据,等待下一个请求"
|
||||
)
|
||||
except Exception as e:
|
||||
logger.warning(f"处理响应失败: {e}")
|
||||
|
||||
page.on("response", handle_response)
|
||||
|
||||
await page.goto(douyin_url, wait_until="domcontentloaded")
|
||||
|
||||
max_wait = wait_seconds
|
||||
for i in range(max_wait):
|
||||
if api_response:
|
||||
logger.info(f"成功在第 {i + 1} 秒捕获 API 响应")
|
||||
break
|
||||
await page.wait_for_timeout(1000)
|
||||
|
||||
# 截取页面 HTML(在关闭浏览器前),用于 SSR 回退解析
|
||||
page_html = await page.content() if not api_response else None
|
||||
|
||||
await browser.close()
|
||||
|
||||
if not api_response and page_html:
|
||||
api_response = parse_ssr_page(page_html)
|
||||
if api_response:
|
||||
logger.info("通过 SSR 页面回退解析获取到图文数据")
|
||||
|
||||
if not api_response:
|
||||
raise DouyinFetchError("无法捕获 API 响应,请检查网络或 URL")
|
||||
|
||||
parsed = parse_douyin_response(api_response, referer_url)
|
||||
|
||||
headers = {
|
||||
"Referer": "https://www.douyin.com/",
|
||||
"User-Agent": (
|
||||
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) "
|
||||
"AppleWebKit/537.36 (KHTML, like Gecko) "
|
||||
"Chrome/122.0.0.0 Safari/537.36"
|
||||
),
|
||||
}
|
||||
|
||||
if parsed.media_type == "视频":
|
||||
content = await _process_video(
|
||||
api_response, tmp_root, parsed.file_name, aweme_id, headers
|
||||
)
|
||||
return parsed.file_name, content
|
||||
|
||||
elif parsed.media_type == "图片":
|
||||
# 先解析 images 列表,区分纯动图和图文/图+视频
|
||||
images_urls, video_url = parse_note_images(
|
||||
api_response, api_response_favorite, aweme_id
|
||||
)
|
||||
if images_urls:
|
||||
# 有图片(纯图文 或 图+视频混合作品)
|
||||
content = await _process_note_with_parsed(
|
||||
images_urls,
|
||||
video_url,
|
||||
tmp_root,
|
||||
parsed.file_name,
|
||||
headers,
|
||||
)
|
||||
else:
|
||||
# 纯动图(所有项都是视频)
|
||||
content = await _process_animated_note(
|
||||
api_response, tmp_root, parsed.file_name, headers
|
||||
)
|
||||
return parsed.file_name, content
|
||||
|
||||
return None, None
|
||||
|
||||
|
||||
# ============================= 视频下载 =============================
|
||||
|
||||
|
||||
async def _process_video(
|
||||
api_response: dict,
|
||||
tmp_root: Path,
|
||||
file_name: str,
|
||||
aweme_id: str,
|
||||
headers: Dict[str, str],
|
||||
) -> Path:
|
||||
"""处理视频内容,返回本地文件路径"""
|
||||
groups = parse_video_urls(api_response)
|
||||
|
||||
best_group = None
|
||||
for g in groups.values():
|
||||
if g["FULL"] or (g["VIDEO"] and g["AUDIO"]):
|
||||
best_group = g
|
||||
break
|
||||
|
||||
if not best_group:
|
||||
raise DouyinFetchError("没有可用的视频组合")
|
||||
|
||||
# 完整视频 — 流式下载
|
||||
if best_group["FULL"]:
|
||||
full = best_group["FULL"]
|
||||
logger.info(f"找到完整视频,数量: {len(full)}")
|
||||
best = max(full, key=lambda x: x["br"])
|
||||
logger.info(f"选择码率: {best['br']} - {best['url'][:60]}...")
|
||||
|
||||
output_path = ensure_unique_path(tmp_root / f"{file_name}.mp4")
|
||||
async with httpx.AsyncClient(headers=headers) as client:
|
||||
async with client.stream("GET", best["url"]) as resp:
|
||||
resp.raise_for_status()
|
||||
with open(output_path, "wb") as f:
|
||||
async for chunk in resp.aiter_bytes(8192):
|
||||
f.write(chunk)
|
||||
logger.info(f"视频下载完成: {output_path}")
|
||||
return output_path
|
||||
|
||||
# 分轨视频 — 分别下载后合并
|
||||
logger.info(
|
||||
f"使用分轨模式,视频数: {len(best_group['VIDEO'])}, "
|
||||
f"音频数: {len(best_group['AUDIO'])}"
|
||||
)
|
||||
video = max(best_group["VIDEO"], key=lambda x: x["br"])
|
||||
audio = max(best_group["AUDIO"], key=lambda x: x["br"])
|
||||
logger.info(f"选择视频码率: {video['br']}")
|
||||
logger.info(f"选择音频码率: {audio['br']}")
|
||||
|
||||
video_path = tmp_root / f"{file_name}_v.mp4"
|
||||
audio_path = tmp_root / f"{file_name}_a.mp4"
|
||||
output_path = ensure_unique_path(tmp_root / f"{file_name}.mp4")
|
||||
|
||||
async with httpx.AsyncClient(headers=headers) as client:
|
||||
logger.info("开始下载视频...")
|
||||
async with client.stream("GET", video["url"]) as v:
|
||||
v.raise_for_status()
|
||||
with open(video_path, "wb") as f:
|
||||
async for chunk in v.aiter_bytes(8192):
|
||||
f.write(chunk)
|
||||
|
||||
logger.info("开始下载音频...")
|
||||
async with client.stream("GET", audio["url"]) as a:
|
||||
a.raise_for_status()
|
||||
with open(audio_path, "wb") as f:
|
||||
async for chunk in a.aiter_bytes(8192):
|
||||
f.write(chunk)
|
||||
|
||||
logger.info("合并视频和音频...")
|
||||
merge_video_audio(video_path, audio_path, output_path)
|
||||
video_path.unlink()
|
||||
audio_path.unlink()
|
||||
|
||||
logger.info(f"视频下载完成: {output_path}")
|
||||
return output_path
|
||||
|
||||
|
||||
# ============================= 图文下载 =============================
|
||||
|
||||
|
||||
async def _process_note_with_parsed(
|
||||
images_urls: List[List[str]],
|
||||
video_url: Optional[str],
|
||||
tmp_root: Path,
|
||||
file_name: str,
|
||||
headers: Dict[str, str],
|
||||
) -> List[Path]:
|
||||
"""根据已解析的图片/视频 URL 列表,并行下载"""
|
||||
note_dir = ensure_unique_path(tmp_root / file_name)
|
||||
note_dir.mkdir(parents=True, exist_ok=True)
|
||||
logger.info(f"图文保存目录: {note_dir}")
|
||||
|
||||
async def _download_one(idx: int, url: str, ext: str = "") -> Optional[Path]:
|
||||
if not ext:
|
||||
ext = _infer_extension(url)
|
||||
filename = f"{idx:03d}{ext}"
|
||||
filepath = note_dir / filename
|
||||
try:
|
||||
async with httpx.AsyncClient(headers=headers) as client:
|
||||
async with client.stream("GET", url) as resp:
|
||||
resp.raise_for_status()
|
||||
with open(filepath, "wb") as f:
|
||||
async for chunk in resp.aiter_bytes(8192):
|
||||
f.write(chunk)
|
||||
logger.info(f"已保存: {filename}")
|
||||
return filepath
|
||||
except Exception as e:
|
||||
logger.error(f"下载 {idx} 失败: {e}")
|
||||
return None
|
||||
|
||||
tasks = [
|
||||
_download_one(i, url_list[0])
|
||||
for i, url_list in enumerate(images_urls, 1)
|
||||
]
|
||||
|
||||
# 图+视频混合作品:视频追加到下载任务
|
||||
if video_url:
|
||||
next_idx = len(images_urls) + 1
|
||||
tasks.append(_download_one(next_idx, video_url, ext=".mp4"))
|
||||
|
||||
results = await asyncio.gather(*tasks)
|
||||
saved_paths: List[Path] = [p for p in results if p is not None]
|
||||
|
||||
logger.info(f"图文下载完成,共 {len(saved_paths)} 个文件")
|
||||
return saved_paths
|
||||
|
||||
|
||||
async def _process_note(
|
||||
api_response: dict,
|
||||
api_response_favorite: dict,
|
||||
tmp_root: Path,
|
||||
file_name: str,
|
||||
aweme_id: str,
|
||||
headers: Dict[str, str],
|
||||
) -> List[Path]:
|
||||
"""处理图文内容,并行下载所有图片;图+视频混合作品同时下载视频"""
|
||||
images_urls, video_url = parse_note_images(
|
||||
api_response, api_response_favorite, aweme_id
|
||||
)
|
||||
logger.info(f"解析到的图片链接:{images_urls}")
|
||||
|
||||
if not images_urls and not video_url:
|
||||
raise DouyinFetchError("未找到图文链接")
|
||||
|
||||
return await _process_note_with_parsed(
|
||||
images_urls, video_url, tmp_root, file_name, headers
|
||||
)
|
||||
|
||||
|
||||
# ============================= 动图下载 =============================
|
||||
|
||||
|
||||
async def _process_animated_note(
|
||||
api_response: dict,
|
||||
tmp_root: Path,
|
||||
file_name: str,
|
||||
headers: Dict[str, str],
|
||||
) -> List[Path]:
|
||||
"""处理动图内容(media_type=42),并行下载所有无声 mp4 视频"""
|
||||
video_urls = parse_animated_note_videos(api_response)
|
||||
logger.info(f"解析到的动图视频链接: {video_urls}")
|
||||
|
||||
note_dir = ensure_unique_path(tmp_root / file_name)
|
||||
note_dir.mkdir(parents=True, exist_ok=True)
|
||||
logger.info(f"动图保存目录: {note_dir}")
|
||||
|
||||
async def _download_one(idx: int, url: str) -> Optional[Path]:
|
||||
filename = f"{idx:03d}.mp4"
|
||||
filepath = note_dir / filename
|
||||
try:
|
||||
async with httpx.AsyncClient(headers=headers) as client:
|
||||
async with client.stream("GET", url) as resp:
|
||||
resp.raise_for_status()
|
||||
with open(filepath, "wb") as f:
|
||||
async for chunk in resp.aiter_bytes(8192):
|
||||
f.write(chunk)
|
||||
logger.info(f"动图视频已保存: {filename}")
|
||||
return filepath
|
||||
except Exception as e:
|
||||
logger.error(f"下载动图视频 {idx} 失败: {e}")
|
||||
return None
|
||||
|
||||
tasks = [
|
||||
_download_one(i, url)
|
||||
for i, url in enumerate(video_urls, 1)
|
||||
]
|
||||
results = await asyncio.gather(*tasks)
|
||||
saved_paths: List[Path] = [p for p in results if p is not None]
|
||||
|
||||
logger.info(f"动图下载完成,共 {len(saved_paths)} 个视频")
|
||||
return saved_paths
|
||||
|
||||
|
||||
def _infer_extension(url: str) -> str:
|
||||
"""从 URL 推断文件扩展名"""
|
||||
extensions = [".webp", ".jpg", ".jpeg", ".png", ".gif", ".avif"]
|
||||
url_lower = url.lower()
|
||||
for ext in extensions:
|
||||
if ext in url_lower:
|
||||
return ext
|
||||
return ".jpg"
|
||||
@@ -0,0 +1,425 @@
|
||||
"""抖音 API 响应解析 — 纯数据提取,不涉及网络/下载"""
|
||||
|
||||
import json
|
||||
import re
|
||||
import urllib.parse
|
||||
from dataclasses import dataclass
|
||||
from datetime import datetime
|
||||
from typing import Dict, List, Optional
|
||||
|
||||
from nonebot import logger
|
||||
|
||||
from ..models import DouyinFetchError
|
||||
from ..utils import slugify
|
||||
|
||||
|
||||
# ============================= 数据结构 =============================
|
||||
|
||||
|
||||
@dataclass
|
||||
class ParsedDouyinContent:
|
||||
"""从 API 响应解析出的抖音内容元数据"""
|
||||
raw_title: str
|
||||
raw_nickname: str
|
||||
media_type: str # "视频" | "图片"
|
||||
file_name: str # 构建好的文件名 stem
|
||||
|
||||
|
||||
# ============================= URL 工具 =============================
|
||||
|
||||
|
||||
def classify_url(url: str) -> str:
|
||||
"""返回 FULL / VIDEO / AUDIO / UNKNOWN"""
|
||||
if "/media-video-" in url:
|
||||
return "VIDEO"
|
||||
if "/media-audio-" in url:
|
||||
return "AUDIO"
|
||||
if "mime_type=video_mp4" in url:
|
||||
return "FULL"
|
||||
return "UNKNOWN"
|
||||
|
||||
|
||||
def extract_group_key(url: str) -> str:
|
||||
"""提取分组 key(视频ID + l参数)"""
|
||||
try:
|
||||
m = re.search(r"/([0-9a-f]{8})/", url)
|
||||
vid = m.group(1) if m else "unknown"
|
||||
parsed = urllib.parse.urlparse(url)
|
||||
qs = urllib.parse.parse_qs(parsed.query)
|
||||
l = qs.get("l", [""])[0]
|
||||
return f"{vid}_{l}"
|
||||
except Exception:
|
||||
return url
|
||||
|
||||
|
||||
def extract_br(url: str) -> int:
|
||||
"""提取码率"""
|
||||
try:
|
||||
parsed = urllib.parse.urlparse(url)
|
||||
qs = urllib.parse.parse_qs(parsed.query)
|
||||
return int(qs.get("br", [0])[0])
|
||||
except Exception:
|
||||
return 0
|
||||
|
||||
|
||||
def extract_trailing_digits(url):
|
||||
"""从 URL 末尾提取数字 ID"""
|
||||
match = re.search(r"/(\d+)(?:\?|#|$)", url)
|
||||
return match.group(1) if match else None
|
||||
|
||||
|
||||
# ============================= 内容提取 =============================
|
||||
|
||||
|
||||
def extract_title_from_api(api_response: dict) -> str:
|
||||
"""从 API 响应提取标题,优先级: preview_title > desc > caption > aweme_list[0].desc"""
|
||||
aweme_detail = api_response.get("aweme_detail") or {}
|
||||
title = aweme_detail.get("preview_title")
|
||||
if not title:
|
||||
title = aweme_detail.get("desc")
|
||||
if not title:
|
||||
title = aweme_detail.get("caption")
|
||||
if not title:
|
||||
aweme_list = api_response.get("aweme_list") or []
|
||||
if aweme_list and isinstance(aweme_list, list):
|
||||
first = aweme_list[0] or {}
|
||||
title = first.get("desc")
|
||||
title_str = (title or "").strip()
|
||||
logger.info(f"RAW标题:{title_str}")
|
||||
if not title_str:
|
||||
logger.info("未获取到标题")
|
||||
return title_str
|
||||
|
||||
|
||||
def extract_author_nickname(api_response: dict) -> str:
|
||||
"""从 API 响应提取作者昵称,优先级: aweme_detail.author.nickname > aweme_list[0].author.nickname"""
|
||||
aweme_detail = api_response.get("aweme_detail") or {}
|
||||
author = aweme_detail.get("author") or {}
|
||||
nickname = author.get("nickname")
|
||||
if not nickname:
|
||||
aweme_list = api_response.get("aweme_list") or []
|
||||
if aweme_list and isinstance(aweme_list, list):
|
||||
first = aweme_list[0] or {}
|
||||
author = first.get("author") or {}
|
||||
nickname = author.get("nickname")
|
||||
logger.info(f"RAW作者昵称:{nickname}")
|
||||
if not nickname:
|
||||
nickname = "未知作者"
|
||||
logger.info("未获取到作者昵称,使用默认值")
|
||||
return nickname
|
||||
|
||||
|
||||
def detect_media_type(referer_url: str | None) -> str | None:
|
||||
"""根据页面 URL 检测媒体类型(视频/图片)"""
|
||||
if referer_url is None:
|
||||
return None
|
||||
if "video" in referer_url:
|
||||
return "视频"
|
||||
elif "note" in referer_url:
|
||||
return "图片"
|
||||
return None
|
||||
|
||||
|
||||
# ============================= 媒体 URL 解析 =============================
|
||||
|
||||
|
||||
def parse_video_urls(api_response: dict) -> Dict[str, Dict[str, List[dict]]]:
|
||||
"""从 API 响应解析视频 URL,返回分组结构: {group_id: {VIDEO: [...], AUDIO: [...], FULL: [...]}}"""
|
||||
captured = []
|
||||
try:
|
||||
video_data = api_response.get("aweme_detail", {}).get("video", {})
|
||||
bit_rates = video_data.get("bit_rate", [])
|
||||
for item in bit_rates:
|
||||
play_addr = item.get("play_addr", {})
|
||||
url_list = play_addr.get("url_list", [])
|
||||
br = item.get("bit_rate", 0)
|
||||
if url_list:
|
||||
url = url_list[0]
|
||||
captured.append(
|
||||
{
|
||||
"url": url,
|
||||
"br": br,
|
||||
"type": classify_url(url),
|
||||
"group": extract_group_key(url),
|
||||
}
|
||||
)
|
||||
logger.info(f"从 API 解析出 {len(captured)} 个视频链接")
|
||||
except Exception as e:
|
||||
logger.error(f"解析视频 URL 失败: {e}")
|
||||
raise DouyinFetchError(f"解析视频 URL 失败: {e}")
|
||||
|
||||
groups: Dict[str, Dict[str, List[dict]]] = {}
|
||||
for item in captured:
|
||||
g = item["group"]
|
||||
if g not in groups:
|
||||
groups[g] = {"FULL": [], "VIDEO": [], "AUDIO": []}
|
||||
if item["type"] in groups[g]:
|
||||
groups[g][item["type"]].append(item)
|
||||
return groups
|
||||
|
||||
|
||||
def parse_note_images(
|
||||
api_response: dict, api_response_favorite: dict, aweme_id: str
|
||||
) -> tuple[List[List[str]], Optional[str]]:
|
||||
"""
|
||||
从 API 响应解析图文/图+视频链接
|
||||
|
||||
images 列表中每项有两种结构:
|
||||
- 纯图片: 取 url_list
|
||||
- 视频: 取 video.play_addr.url_list
|
||||
|
||||
返回:
|
||||
(images_list, video_url)
|
||||
- images_list: 图片 URL 列表 [[url1, ...], ...]
|
||||
- video_url: 混合作品中的视频 URL(无视频时为 None)
|
||||
"""
|
||||
images_list: List[List[str]] = []
|
||||
video_url: Optional[str] = None
|
||||
try:
|
||||
aweme_list = api_response.get("aweme_list", None)
|
||||
if not aweme_list:
|
||||
raise DouyinFetchError(
|
||||
f"获取到的作品图片列表为空,解析图文链接失败:{aweme_list}"
|
||||
)
|
||||
images_part = aweme_list[0]
|
||||
images = images_part.get("images", [])
|
||||
if images:
|
||||
for image in images:
|
||||
if not isinstance(image, dict):
|
||||
continue
|
||||
# 优先检查 video.play_addr(有视频的项)
|
||||
video = image.get("video")
|
||||
if video and isinstance(video, dict):
|
||||
play_addr = video.get("play_addr")
|
||||
if play_addr and isinstance(play_addr, dict):
|
||||
play_urls = play_addr.get("url_list") or []
|
||||
if play_urls:
|
||||
video_url = play_urls[0]
|
||||
logger.info(
|
||||
f"从 images 中发现视频: {video_url[:60]}..."
|
||||
)
|
||||
continue
|
||||
# 无视频的项取 url_list
|
||||
url_list = image.get("url_list", [])
|
||||
if url_list:
|
||||
images_list.append(url_list)
|
||||
logger.info(f"从 API 解析出 {len(images_list)} 张图片")
|
||||
if video_url:
|
||||
logger.info("混合作品包含视频")
|
||||
else:
|
||||
raise DouyinFetchError(
|
||||
f"获取到的作品图片列表为空,解析图文链接失败:{aweme_list}"
|
||||
)
|
||||
except DouyinFetchError:
|
||||
raise
|
||||
except Exception as e:
|
||||
logger.error(f"解析图文链接失败: {e}")
|
||||
raise DouyinFetchError(f"解析图文链接失败: {e}")
|
||||
return images_list, video_url
|
||||
|
||||
|
||||
# ============================= SSR 页面回退解析 =============================
|
||||
|
||||
|
||||
def _extract_rsc_payload(html: str) -> Optional[str]:
|
||||
"""从 HTML 中提取 __pace_f.push 的 RSC 流数据(7:[...] 格式)"""
|
||||
pos = 0
|
||||
while True:
|
||||
m = re.search(r'self\.__pace_f\.push\(\[(\d+),"', html[pos:])
|
||||
if not m:
|
||||
break
|
||||
start = pos + m.end()
|
||||
# 手动找闭合引号,处理转义
|
||||
end = start
|
||||
while end < len(html):
|
||||
ch = html[end]
|
||||
if ch == "\\":
|
||||
end += 2
|
||||
elif ch == '"':
|
||||
break
|
||||
else:
|
||||
end += 1
|
||||
# RSC payload 是 JS 字符串字面量(\" \\n \\\\ 等转义),
|
||||
# 用 json.loads 按字符串语义一次解码,避免手动替换顺序错误
|
||||
# (2026-08-13 实测:双重转义页手动替换会产出非法 JSON)
|
||||
try:
|
||||
raw = json.loads('"' + html[start:end] + '"')
|
||||
except json.JSONDecodeError:
|
||||
pos = end + 1
|
||||
continue
|
||||
# 找 RSC 流块: "7:[...]"
|
||||
if raw.startswith("7:["):
|
||||
return raw[2:] # 去掉 "7:" 前缀
|
||||
pos = end + 1
|
||||
return None
|
||||
|
||||
|
||||
def _rsc_to_json(rsc: str) -> dict:
|
||||
"""将 React Flight 格式转为普通 JSON dict,处理 $undefined / $Lx 等特殊 token"""
|
||||
# $undefined -> null
|
||||
text = re.sub(r'\$undefined', 'null', rsc)
|
||||
# $L9 等 lazy ref -> null(避免解析报错)
|
||||
text = re.sub(r'\$L\d+', 'null', text)
|
||||
try:
|
||||
return json.loads(text)
|
||||
except json.JSONDecodeError:
|
||||
logger.error(f"SSR 页面 JSON 解析失败,数据前 500 字: {text[:500]}")
|
||||
raise DouyinFetchError("SSR 页面数据解析失败")
|
||||
|
||||
|
||||
def parse_ssr_page(html: str) -> Optional[dict]:
|
||||
"""
|
||||
从 SSR 页面 HTML 中提取图文数据,转换为 aweme_list 格式
|
||||
当 API 拦截未触发时作为回退方案
|
||||
|
||||
返回的 dict 可直接用于 parse_note_images / parse_animated_note_videos /
|
||||
extract_title_from_api / extract_author_nickname
|
||||
"""
|
||||
rsc = _extract_rsc_payload(html)
|
||||
if not rsc:
|
||||
return None
|
||||
|
||||
data = _rsc_to_json(rsc)
|
||||
if not isinstance(data, list) or len(data) < 4:
|
||||
return None
|
||||
|
||||
# React Flight 格式: ["$","$L9",null,{awemeId,aweme:{detail:{...}}}]
|
||||
page_data = data[3]
|
||||
if not isinstance(page_data, dict):
|
||||
return None
|
||||
|
||||
aweme = page_data.get("aweme") or {}
|
||||
if isinstance(aweme, str):
|
||||
aweme = {}
|
||||
detail = aweme.get("detail") or {}
|
||||
if isinstance(detail, str):
|
||||
detail = {}
|
||||
|
||||
author_info = detail.get("authorInfo") or {}
|
||||
|
||||
# 标准化: 构建 aweme_list 格式(图文解析函数依赖此结构)
|
||||
item = {
|
||||
"desc": detail.get("desc") or "",
|
||||
"images": detail.get("images") or [],
|
||||
"media_type": detail.get("awemeType") or 0,
|
||||
"author": {
|
||||
"nickname": author_info.get("nickname") or "",
|
||||
"uid": author_info.get("uid") or "",
|
||||
},
|
||||
}
|
||||
|
||||
# 标准化 urlList -> url_list, downloadUrlList -> download_url_list
|
||||
for img in item["images"]:
|
||||
if "urlList" in img and "url_list" not in img:
|
||||
img["url_list"] = img.pop("urlList")
|
||||
if "downloadUrlList" in img and "download_url_list" not in img:
|
||||
img["download_url_list"] = img.pop("downloadUrlList")
|
||||
# 动图的 video 字段
|
||||
video = img.get("video")
|
||||
if isinstance(video, dict):
|
||||
if "playAddr" in video and "play_addr" not in video:
|
||||
video["play_addr"] = video.pop("playAddr")
|
||||
|
||||
logger.info(f"SSR 页面解析成功: {len(item['images'])} 张图片, "
|
||||
f"media_type={item['media_type']}")
|
||||
|
||||
return {"aweme_list": [item]}
|
||||
|
||||
|
||||
# ============================= 文件名构建 =============================
|
||||
|
||||
|
||||
def build_file_name(
|
||||
raw_title: str,
|
||||
raw_nickname: str,
|
||||
media_type: str,
|
||||
) -> str:
|
||||
"""
|
||||
构建文件名 stem,格式: {作者}_{标题}_{类型}_{时间戳}
|
||||
|
||||
昵称不限长,标题最多 15 字符(slugify 后),末尾 HHMMSS 防覆盖。
|
||||
"""
|
||||
slug_nickname = slugify(raw_nickname)
|
||||
|
||||
if raw_title:
|
||||
slug_title = slugify(raw_title, max_length=15)
|
||||
else:
|
||||
slug_title = ""
|
||||
|
||||
if not slug_title:
|
||||
slug_title = datetime.now().strftime("%H%M%S")
|
||||
logger.info(f"标题为空,使用短时间戳: {slug_title}")
|
||||
|
||||
slug_type = slugify(media_type)
|
||||
time_suffix = datetime.now().strftime("%H%M%S")
|
||||
|
||||
return f"{slug_nickname}_{slug_title}_{slug_type}_{time_suffix}"
|
||||
|
||||
|
||||
# ============================= 动图检测 =============================
|
||||
|
||||
|
||||
def is_animated_note(api_response: dict) -> bool:
|
||||
"""检测图文是否为动图(media_type == 42),动图每张图片内含无声 mp4 视频"""
|
||||
aweme_list = api_response.get("aweme_list") or []
|
||||
if aweme_list and isinstance(aweme_list, list):
|
||||
return aweme_list[0].get("media_type") == 42
|
||||
return False
|
||||
|
||||
|
||||
def parse_animated_note_videos(api_response: dict) -> List[str]:
|
||||
"""从动图 API 响应解析视频链接(无音轨),返回最佳质量的视频 URL 列表"""
|
||||
video_urls = []
|
||||
try:
|
||||
aweme_list = api_response.get("aweme_list") or []
|
||||
if not aweme_list:
|
||||
raise DouyinFetchError("动图作品列表为空")
|
||||
|
||||
images = aweme_list[0].get("images") or []
|
||||
for image in images:
|
||||
video = image.get("video") or {}
|
||||
play_addr = video.get("play_addr") or {}
|
||||
url_list = play_addr.get("url_list") or []
|
||||
if url_list:
|
||||
video_urls.append(url_list[0])
|
||||
|
||||
logger.info(f"从动图 API 解析出 {len(video_urls)} 个视频")
|
||||
except DouyinFetchError:
|
||||
raise
|
||||
except Exception as e:
|
||||
logger.error(f"解析动图视频链接失败: {e}")
|
||||
raise DouyinFetchError(f"解析动图视频链接失败: {e}")
|
||||
|
||||
if not video_urls:
|
||||
raise DouyinFetchError("未找到动图视频链接")
|
||||
return video_urls
|
||||
|
||||
|
||||
# ============================= 一站式解析 =============================
|
||||
|
||||
|
||||
def parse_douyin_response(
|
||||
api_response: dict, referer_url: str | None
|
||||
) -> ParsedDouyinContent:
|
||||
"""
|
||||
一站式解析: 从 API 响应 + referer URL 中提取所有元数据并构建文件名
|
||||
"""
|
||||
media_type = detect_media_type(referer_url)
|
||||
if not media_type:
|
||||
raise DouyinFetchError("媒体类型无法获取,请检查url是否正确")
|
||||
|
||||
raw_title = extract_title_from_api(api_response)
|
||||
raw_nickname = extract_author_nickname(api_response)
|
||||
file_name = build_file_name(raw_title, raw_nickname, media_type)
|
||||
|
||||
logger.info(
|
||||
f"内容标题: {raw_title}, 作者: {raw_nickname}, "
|
||||
f"类型: {media_type}, 文件名: {file_name}"
|
||||
)
|
||||
|
||||
return ParsedDouyinContent(
|
||||
raw_title=raw_title,
|
||||
raw_nickname=raw_nickname,
|
||||
media_type=media_type,
|
||||
file_name=file_name,
|
||||
)
|
||||
@@ -0,0 +1,153 @@
|
||||
"""抖音图文 SSR 静态解析 — 不走浏览器,直接解析服务端渲染页面
|
||||
|
||||
与 douyin_api.py 的 playwright 拦截方案互补:
|
||||
- 视频作品 → fetch_douyin_content(playwright 拦截,可取全部清晰度)
|
||||
- 图文作品 → fetch_douyin_note_ssr(本模块,按 aweme_id 精准渲染,秒级返回)
|
||||
|
||||
SSR 数据(window._ROUTER_DATA / __pace_f RSC)是服务端按作品 ID 渲染的页面数据,
|
||||
不存在 playwright 拦截 aweme/v1/web/aweme/post 时取到作者其他作品的问题。
|
||||
|
||||
注意:必须用移动端 UA 请求 iesdouyin.com/share 端点,桌面 UA 会被 302 回主站
|
||||
并返回 JS 反爬壳(无任何 SSR 数据)。
|
||||
"""
|
||||
|
||||
import json
|
||||
import re
|
||||
from pathlib import Path
|
||||
from typing import List, Optional
|
||||
|
||||
import httpx
|
||||
from nonebot import logger
|
||||
|
||||
from ..models import DouyinFetchError
|
||||
from ..utils import get_temp_root
|
||||
from .douyin_api import _process_animated_note, _process_note_with_parsed
|
||||
from .douyin_parser import parse_douyin_response, parse_note_images, parse_ssr_page
|
||||
|
||||
MOBILE_UA = (
|
||||
"Mozilla/5.0 (iPhone; CPU iPhone OS 17_2 like Mac OS X) "
|
||||
"AppleWebKit/605.1.15 (KHTML, like Gecko) Version/17.2 "
|
||||
"Mobile/15E148 Safari/604.1"
|
||||
)
|
||||
ROUTER_DATA_RE = re.compile(r"window\._ROUTER_DATA\s*=\s*(.*?)</script>", re.S)
|
||||
# 短链重定向后统一为 iesdouyin.com/share/video/{id} 形态(图文作品也如此),
|
||||
# 因此 note / video 两种形态都要能提取 ID
|
||||
AWEME_ID_RE = re.compile(r"/(?:share/)?(?:note|video)/(\d+)")
|
||||
|
||||
|
||||
async def fetch_douyin_note_ssr(
|
||||
note_url: str,
|
||||
cookies: List[dict],
|
||||
) -> tuple[Optional[str], Optional[List[Path]]]:
|
||||
"""
|
||||
解析图文作品(SSR 静态解析,秒级返回)
|
||||
|
||||
接受 note / video 任意形态的抖音链接(重定向后图文也常是 share/video),
|
||||
按 SSR 内容判定:有 images → 图文/动图下载;无 images(真视频)→
|
||||
返回 (None, None) 交由 playwright 链路。
|
||||
|
||||
Returns:
|
||||
(title, file_paths) — 图片/视频本地路径列表;非图文返回 (None, None);
|
||||
失败抛 DouyinFetchError
|
||||
"""
|
||||
m = AWEME_ID_RE.search(note_url)
|
||||
if not m:
|
||||
raise DouyinFetchError(f"无法从链接提取作品 ID: {note_url}")
|
||||
vid = m.group(1)
|
||||
|
||||
api_response = await _fetch_ssr_api_response(note_url, vid, cookies)
|
||||
|
||||
# 内容判定:SSR 数据无 images → 真视频作品,回落 playwright
|
||||
aweme = (api_response.get("aweme_list") or [None])[0] or {}
|
||||
if not aweme.get("images"):
|
||||
logger.info("SSR 内容为视频作品(无 images),回退 playwright 链路")
|
||||
return None, None
|
||||
|
||||
# 复用原有解析链:标题 / 作者 / 文件名(referer 统一用 note 形态,
|
||||
# 避免输入是 share/video 时被误判为"视频")
|
||||
parsed = parse_douyin_response(
|
||||
api_response, f"https://www.douyin.com/note/{vid}"
|
||||
)
|
||||
logger.info(f"SSR 标题: {parsed.raw_title}")
|
||||
logger.info(f"SSR 作者: {parsed.raw_nickname}")
|
||||
|
||||
tmp_root = get_temp_root("douyin")
|
||||
headers = {
|
||||
"Referer": "https://www.douyin.com/",
|
||||
"User-Agent": (
|
||||
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) "
|
||||
"AppleWebKit/537.36 (KHTML, like Gecko) "
|
||||
"Chrome/122.0.0.0 Safari/537.36"
|
||||
),
|
||||
}
|
||||
|
||||
# 复用原有分派:纯图文 / 图+视频混合 → 并发下载;纯动图 → 逐张下载 mp4
|
||||
images_urls, video_url = parse_note_images(api_response, None, vid)
|
||||
if images_urls:
|
||||
file_paths = await _process_note_with_parsed(
|
||||
images_urls, video_url, tmp_root, parsed.file_name, headers
|
||||
)
|
||||
else:
|
||||
file_paths = await _process_animated_note(
|
||||
api_response, tmp_root, parsed.file_name, headers
|
||||
)
|
||||
return parsed.file_name, file_paths
|
||||
|
||||
|
||||
async def _fetch_ssr_api_response(
|
||||
note_url: str, vid: str, cookies: List[dict]
|
||||
) -> dict:
|
||||
"""请求 douyin.com note 页面并提取 SSR 数据,规范化为 aweme_list 格式
|
||||
|
||||
2026-08-13 起 iesdouyin.com/share/note 端点对所有请求(含登录态)只返回
|
||||
33KB 壳页(无任何作品数据),改从 www.douyin.com/note/{id} 提取
|
||||
__pace_f RSC 数据——服务端按作品 id 精确渲染,与 URL 一一对应,
|
||||
不存在 aweme/post 作者列表"取错作品"的问题。
|
||||
必须用桌面 UA + 登录 cookie(whale 风控对未登录请求截断作品数据)。
|
||||
"""
|
||||
url = f"https://www.douyin.com/note/{vid}"
|
||||
cookie_map = {c["name"]: c["value"] for c in cookies} if cookies else None
|
||||
headers = {
|
||||
"User-Agent": (
|
||||
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) "
|
||||
"AppleWebKit/537.36 (KHTML, like Gecko) "
|
||||
"Chrome/151.0.0.0 Safari/537.36"
|
||||
),
|
||||
"Referer": "https://www.douyin.com/",
|
||||
}
|
||||
async with httpx.AsyncClient(
|
||||
headers=headers, timeout=20, follow_redirects=True
|
||||
) as client:
|
||||
resp = await client.get(url, cookies=cookie_map)
|
||||
if resp.status_code >= 400:
|
||||
raise DouyinFetchError(f"SSR 页面请求失败: status={resp.status_code}")
|
||||
html = resp.text
|
||||
logger.info(f"SSR 页面: {url} status={resp.status_code} size={len(html)}")
|
||||
|
||||
# 1. 旧版 SSR: window._ROUTER_DATA(snake_case 结构,可直接复用原有解析函数)
|
||||
if m := ROUTER_DATA_RE.search(html):
|
||||
data = _clean_js_json(m.group(1))
|
||||
if data:
|
||||
loader = data.get("loaderData", {})
|
||||
page = loader.get("note_(id)/page") or loader.get("video_(id)/page")
|
||||
item_list = (page.get("videoInfoRes") or {}).get("item_list") or []
|
||||
if item_list:
|
||||
logger.info("SSR 数据来源: _ROUTER_DATA")
|
||||
return {"aweme_list": [item_list[0]]}
|
||||
# 2. 新版 SSR: __pace_f.push RSC 流(parse_ssr_page 已规范化为 aweme_list 格式)
|
||||
if parsed := parse_ssr_page(html):
|
||||
logger.info("SSR 数据来源: __pace_f RSC")
|
||||
return parsed
|
||||
raise DouyinFetchError("SSR 页面无图文数据(可能已被删除或风控)")
|
||||
|
||||
|
||||
def _clean_js_json(raw: str) -> Optional[dict]:
|
||||
"""清洗 JS 对象文本为合法 JSON($undefined / $Lx 占位符)"""
|
||||
text = raw.strip().rstrip(";").strip()
|
||||
text = re.sub(r"\$undefined", "null", text)
|
||||
text = re.sub(r"\$L\d+", "null", text)
|
||||
try:
|
||||
return json.loads(text)
|
||||
except json.JSONDecodeError:
|
||||
logger.warning("SSR JSON 解析失败")
|
||||
return None
|
||||
@@ -0,0 +1,366 @@
|
||||
"""小红书笔记解析 — window.__INITIAL_STATE__ 静态提取
|
||||
|
||||
借鉴 nonebot-plugin-parser 的 XiaoHongShuParser:
|
||||
- xhslink.com/cn 短链 → 重定向(保留 xsec_token 参数)
|
||||
- xiaohongshu.com/explore/{id}?xsec_token=... → 桌面端点
|
||||
- xiaohongshu.com/discovery/item/{id}?xsec_token=... → 移动端点(fallback)
|
||||
- 页面 HTML 提取 window.__INITIAL_STATE__ → noteDetailMap[id].note
|
||||
|
||||
请求端点依次尝试 国际站 rednote.com(含无水印原片 originVideoKey)
|
||||
→ 国内站 xiaohongshu.com(国内笔记国际站无数据,需 cookies.txt 登录态)。
|
||||
|
||||
返回 (title, 文件):
|
||||
- 图文笔记 → (title, [图片路径列表])
|
||||
- 视频笔记 → (title, 视频文件 Path,h265 无水印优先)
|
||||
- 纯文字笔记 → (title, [])
|
||||
- 解析失败 → (None, None)
|
||||
"""
|
||||
|
||||
import asyncio
|
||||
import json
|
||||
import re
|
||||
import tempfile
|
||||
import urllib.parse
|
||||
from datetime import datetime
|
||||
from pathlib import Path
|
||||
from typing import Optional, Union
|
||||
|
||||
from nonebot import logger
|
||||
|
||||
from ..models import ContentFetchError
|
||||
from ..utils import get_temp_root, parse_netscape_cookies, slugify
|
||||
|
||||
DATA_DIR = Path(__file__).resolve().parent.parent / "data"
|
||||
|
||||
REDNOTE_UA = (
|
||||
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 "
|
||||
"(KHTML, like Gecko) Chrome/122.0.0.0 Safari/537.36"
|
||||
)
|
||||
REDNOTE_REFERER = "https://www.xiaohongshu.com/"
|
||||
|
||||
REDNOTE_SHORT_RE = re.compile(r"xhslink\.(?:com|cn)/[A-Za-z0-9._?%&+=/#@-]+")
|
||||
EXPLORE_RE = re.compile(
|
||||
r"xiaohongshu\.com/explore/([0-9a-zA-Z]+)(\?[^\s\"'<>]+)?"
|
||||
)
|
||||
DISCOVERY_RE = re.compile(
|
||||
r"xiaohongshu\.com/discovery/item/([0-9a-zA-Z]+)(\?[^\s\"'<>]+)?"
|
||||
)
|
||||
INITIAL_STATE_RE = re.compile(r"window\.__INITIAL_STATE__=(.*?)</script>", re.S)
|
||||
|
||||
# 请求 host 顺序:国际站优先(含无水印原片),失败回退国内站(国内笔记)
|
||||
REDNOTE_HOSTS = ("www.rednote.com", "www.xiaohongshu.com")
|
||||
|
||||
|
||||
async def fetch_rednote_content(
|
||||
url: str,
|
||||
) -> tuple[Optional[str], Optional[Union[Path, list[Path]]]]:
|
||||
"""解析小红书笔记链接"""
|
||||
try:
|
||||
return await _parse(url)
|
||||
except Exception:
|
||||
logger.exception(f"小红书解析失败: {url}")
|
||||
return None, None
|
||||
|
||||
|
||||
async def _parse(url: str):
|
||||
# 1. 短链 → 重定向(重定向 URL 带 xsec_token,必须保留)
|
||||
if "xhslink" in url:
|
||||
resolved = await _resolve_short_link(url)
|
||||
if resolved:
|
||||
logger.info(f"xhslink 重定向: {url} -> {resolved}")
|
||||
url = resolved
|
||||
|
||||
# 2. explore 优先(国际站才有 originVideoKey 无水印原片),
|
||||
# 依次尝试 国际站 → 国内站;失败再用同一 id+query 回退 discovery 端点
|
||||
if m := EXPLORE_RE.search(url):
|
||||
note_id, query = m.group(1), (m.group(2) or "").lstrip("?")
|
||||
return await _fetch_with_retry(note_id, query, ("explore", "discovery"))
|
||||
if m := DISCOVERY_RE.search(url):
|
||||
note_id, query = m.group(1), (m.group(2) or "").lstrip("?")
|
||||
return await _fetch_with_retry(note_id, query, ("discovery",))
|
||||
|
||||
# 3. 短链重定向终态: xiaohongshu.com/explore?target_note_id={id}&xsec_token=...
|
||||
# (discovery/item 会 302 到该形态,note id 在 query 里),
|
||||
# 只保留访问必需的 xsec_token / xsec_source
|
||||
if m := re.search(r"xiaohongshu\.com/explore\?([^\"'<>]+)", url):
|
||||
qs = urllib.parse.parse_qs(m.group(1))
|
||||
note_id = (qs.get("target_note_id") or [""])[0]
|
||||
if note_id:
|
||||
keep = {
|
||||
k: v
|
||||
for k, v in qs.items()
|
||||
if k in ("xsec_token", "xsec_source")
|
||||
}
|
||||
query = urllib.parse.urlencode(keep, doseq=True)
|
||||
return await _fetch_with_retry(
|
||||
note_id, query, ("explore", "discovery")
|
||||
)
|
||||
|
||||
logger.warning(f"无法识别的小红书链接: {url}")
|
||||
return None, None
|
||||
|
||||
|
||||
async def _fetch_with_retry(
|
||||
note_id: str, query: str, paths: tuple[str, ...]
|
||||
) -> tuple[str, Union[Path, list[Path]]] | None:
|
||||
"""依次尝试 国际站 → 国内站 × 端点;全部失败后稍候重试一轮
|
||||
|
||||
小红书对突发请求会节流(页面 200 但 noteDetailMap 为空/无数据),
|
||||
重试一轮可绕过大部分瞬时风控;重试仍失败返回 (None, None)。
|
||||
"""
|
||||
for attempt in range(2):
|
||||
for host in REDNOTE_HOSTS:
|
||||
for path in paths:
|
||||
fetch = _parse_explore if path == "explore" else _parse_discovery
|
||||
try:
|
||||
return await fetch(host, note_id, query)
|
||||
except Exception as e:
|
||||
logger.warning(
|
||||
f"小红书 {host}/{path} 解析失败(第 {attempt + 1} 轮): {e}"
|
||||
)
|
||||
if attempt == 0:
|
||||
await asyncio.sleep(2)
|
||||
logger.warning(f"小红书解析失败(所有端点): {note_id}")
|
||||
return None, None
|
||||
|
||||
|
||||
async def _parse_explore(host: str, note_id: str, query: str):
|
||||
# 国际站 rednote.com 的页面数据含 video.consumer.originVideoKey(无水印原片)
|
||||
url = f"https://{host}/explore/{note_id}?{query}"
|
||||
logger.info(f"小红书 explore: {url}")
|
||||
html = await _fetch_page(
|
||||
url,
|
||||
headers={
|
||||
"User-Agent": REDNOTE_UA,
|
||||
"Referer": f"https://{host}/explore/{note_id}",
|
||||
"origin": f"https://{host}",
|
||||
"accept": (
|
||||
"text/html,application/xhtml+xml,application/xml;q=0.9,"
|
||||
"image/avif,image/webp,image/apng,*/*;q=0.8,"
|
||||
"application/signed-exchange;v=b3;q=0.7"
|
||||
),
|
||||
"cookie": _build_cookie_header(),
|
||||
},
|
||||
)
|
||||
note = _extract_note(html, note_id)
|
||||
return await _build_result(note)
|
||||
|
||||
|
||||
async def _parse_discovery(host: str, note_id: str, query: str):
|
||||
url = f"https://{host}/discovery/item/{note_id}?{query}"
|
||||
logger.info(f"小红书 discovery: {url}")
|
||||
html = await _fetch_page(
|
||||
url,
|
||||
headers={
|
||||
"User-Agent": REDNOTE_UA,
|
||||
"Referer": f"https://{host}/discovery/item/{note_id}",
|
||||
"origin": f"https://{host}",
|
||||
"x-requested-with": "XMLHttpRequest",
|
||||
"sec-fetch-site": "same-origin",
|
||||
"sec-fetch-mode": "cors",
|
||||
"sec-fetch-dest": "empty",
|
||||
"cookie": _build_cookie_header(),
|
||||
},
|
||||
)
|
||||
note = _extract_note(html, note_id)
|
||||
return await _build_result(note)
|
||||
|
||||
|
||||
def _build_cookie_header() -> str:
|
||||
"""从 cookies.txt 构建小红书登录 cookie 头(无登录态时返回空串)"""
|
||||
cookies_path = DATA_DIR / "cookies.txt"
|
||||
if not cookies_path.exists():
|
||||
return ""
|
||||
cookies = parse_netscape_cookies(str(cookies_path))
|
||||
rednote = [c for c in cookies if "xiaohongshu" in c.get("domain", "")]
|
||||
if not rednote:
|
||||
logger.warning("cookies.txt 中无小红书登录态,匿名访问(依赖链接 xsec_token)")
|
||||
return ""
|
||||
logger.info(f"小红书登录态: {len(rednote)} 条 cookie")
|
||||
return "; ".join(f"{c['name']}={c['value']}" for c in rednote)
|
||||
|
||||
|
||||
def _extract_note(html: str, note_id: str) -> dict:
|
||||
"""从 __INITIAL_STATE__ 提取笔记详情"""
|
||||
m = INITIAL_STATE_RE.search(html)
|
||||
if not m:
|
||||
raise ContentFetchError("小红书页面无 __INITIAL_STATE__(可能已删除或风控)")
|
||||
raw = m.group(1)
|
||||
# JS 语法清理:__INITIAL_STATE__ 不是纯 JSON
|
||||
# - undefined → null(老问题)
|
||||
# - new Map([]) / new Set([])(2026-08-24 实测:
|
||||
# "noteDetailMap":new Map([]) 不做处理 json.loads 必挂)
|
||||
raw = raw.replace("undefined", "null")
|
||||
raw = re.sub(r"new Map\([^)]*\)", "{}", raw)
|
||||
raw = re.sub(r"new Set\([^)]*\)", "[]", raw)
|
||||
try:
|
||||
data = json.loads(raw)
|
||||
except json.JSONDecodeError:
|
||||
raise ContentFetchError("小红书 __INITIAL_STATE__ JSON 解析失败")
|
||||
note = ((data.get("note") or {}).get("noteDetailMap") or {}).get(note_id)
|
||||
if not note:
|
||||
raise ContentFetchError(f"页面数据中未找到笔记 {note_id}")
|
||||
return note.get("note") or {}
|
||||
|
||||
|
||||
async def _build_result(note: dict) -> tuple[str, Union[Path, list[Path]]]:
|
||||
# 空壳 note(短链重定向带 undertake_note_error=该内容暂时无法查看)
|
||||
# → 笔记已删除/私密,直接报错而不是误判为纯文字笔记
|
||||
if not any(
|
||||
note.get(k) for k in ("title", "desc", "type", "imageList", "video")
|
||||
):
|
||||
raise ContentFetchError("笔记内容不可见(可能已删除/私密),无法解析")
|
||||
title = note.get("title") or ""
|
||||
desc = note.get("desc") or ""
|
||||
nickname = ((note.get("user") or {}).get("nickname")) or "小红书用户"
|
||||
text = title or desc or "小红书笔记"
|
||||
|
||||
# 1. 视频笔记 → 无水印原片优先
|
||||
if note.get("type") == "video" and note.get("video"):
|
||||
# 1a. 无水印原片(国际站数据 video.consumer.originVideoKey)
|
||||
consumer = (note["video"].get("consumer") or {})
|
||||
okey = consumer.get("originVideoKey")
|
||||
if okey:
|
||||
video_url = f"https://sns-video-bd.xhscdn.com/{okey}"
|
||||
logger.info(f"小红书视频: 无水印原片 originVideoKey={okey[:30]}...")
|
||||
file_name = _build_file_name(nickname, text, "视频")
|
||||
video_path = await _download_video(video_url, file_name)
|
||||
return text, video_path
|
||||
|
||||
# 1b. 无 originVideoKey(国内站数据)→ 从 stream 分组选无水印原片
|
||||
# 国内站 masterUrl 同样是 sns-video-v6 原片 CDN(无水印),
|
||||
# 但不同抓取批次返回的清晰度集合不同(同组多条/分组顺序不定),
|
||||
# 因此跨全部编码分组收集候选,取 size 最大(质量最高)的流。
|
||||
stream = ((note["video"].get("media") or {}).get("stream")) or {}
|
||||
candidates = [
|
||||
it
|
||||
for items in stream.values()
|
||||
if isinstance(items, list)
|
||||
for it in items
|
||||
if isinstance(it, dict) and it.get("masterUrl")
|
||||
]
|
||||
if candidates:
|
||||
best = max(
|
||||
candidates,
|
||||
key=lambda it: (it.get("size") or 0, it.get("avgBitrate") or 0),
|
||||
)
|
||||
video_url = best["masterUrl"]
|
||||
duration = best.get("duration", 0)
|
||||
logger.info(
|
||||
f"小红书视频: 无水印流 {best.get('qualityType')} "
|
||||
f"{best.get('width')}x{best.get('height')} {best.get('fps')}fps "
|
||||
f"size={best.get('size')} duration={duration}ms"
|
||||
)
|
||||
file_name = _build_file_name(nickname, text, "视频")
|
||||
video_path = await _download_video(video_url, file_name)
|
||||
return text, video_path
|
||||
raise ContentFetchError("小红书视频流解析失败")
|
||||
|
||||
# 2. 图文笔记
|
||||
images = [
|
||||
img.get("urlDefault") or img.get("url")
|
||||
for img in note.get("imageList") or []
|
||||
]
|
||||
images = [u for u in images if u]
|
||||
if not images:
|
||||
logger.info(f"小红书文字笔记: {text[:30]}")
|
||||
return text, []
|
||||
|
||||
file_name = _build_file_name(nickname, text, "笔记")
|
||||
file_paths = await _download_images(images, file_name)
|
||||
logger.info(f"小红书图文笔记: 作者={nickname}, 图片={len(images)} 张")
|
||||
return text, file_paths
|
||||
|
||||
|
||||
def _build_file_name(nickname: str, title: str, kind: str) -> str:
|
||||
"""构建文件名 stem: {作者}_{标题}_{类型}_{时间}"""
|
||||
slug_nickname = slugify(nickname)
|
||||
slug_title = slugify(title or "", max_length=15)
|
||||
if not slug_title:
|
||||
slug_title = datetime.now().strftime("%H%M%S")
|
||||
time_suffix = datetime.now().strftime("%H%M%S")
|
||||
return f"{slug_nickname}_{slug_title}_{kind}_{time_suffix}"
|
||||
|
||||
|
||||
async def _download_images(image_urls: list[str], file_name: str) -> list[Path]:
|
||||
"""并发下载图片(复用抖音图文的下载流程)"""
|
||||
import httpx
|
||||
|
||||
from .douyin_api import _process_note_with_parsed
|
||||
|
||||
tmp_root = get_temp_root("xiaohongshu")
|
||||
headers = {
|
||||
"Referer": REDNOTE_REFERER,
|
||||
"User-Agent": REDNOTE_UA,
|
||||
}
|
||||
return await _process_note_with_parsed(
|
||||
[[u] for u in image_urls], None, tmp_root, file_name, headers
|
||||
)
|
||||
|
||||
|
||||
async def _download_video(video_url: str, file_name: str) -> Path:
|
||||
"""流式下载视频
|
||||
|
||||
注意:sns-video-bd(无水印原片)不带 Referer 或带 xiaohongshu.com
|
||||
均可,但带 rednote.com Referer 会 403,因此不设 Referer。
|
||||
"""
|
||||
import httpx
|
||||
|
||||
tmp_root = get_temp_root("xiaohongshu")
|
||||
output_path = tmp_root / f"{file_name}.mp4"
|
||||
headers = {"User-Agent": REDNOTE_UA}
|
||||
async with httpx.AsyncClient(headers=headers, timeout=300) as client:
|
||||
async with client.stream("GET", video_url) as resp:
|
||||
resp.raise_for_status()
|
||||
with open(output_path, "wb") as f:
|
||||
async for chunk in resp.aiter_bytes(8192):
|
||||
f.write(chunk)
|
||||
logger.info(f"小红书视频下载完成: {output_path}")
|
||||
return output_path
|
||||
|
||||
|
||||
async def _fetch_page(url: str, headers: dict) -> str:
|
||||
import httpx
|
||||
|
||||
async with httpx.AsyncClient(
|
||||
headers=headers, timeout=20, follow_redirects=True
|
||||
) as client:
|
||||
resp = await client.get(url)
|
||||
if resp.status_code >= 400:
|
||||
raise ContentFetchError(f"小红书页面请求失败: status={resp.status_code}")
|
||||
return resp.text
|
||||
|
||||
|
||||
async def _resolve_short_link(url: str) -> Optional[str]:
|
||||
"""xhslink 短链重定向(最多 3 跳取最终 URL,重定向 URL 带 xsec_token)
|
||||
|
||||
借鉴 nonebot-plugin-parser:xhslink 用移动端 headers 请求
|
||||
(origin / x-requested-with 等),避免被当作非 App 来源拒绝。
|
||||
xhslink.cn 可能先跳到 xhslink.com 再跳小红书,需循环取跳。
|
||||
"""
|
||||
import httpx
|
||||
|
||||
headers = {
|
||||
"User-Agent": REDNOTE_UA,
|
||||
"Referer": REDNOTE_REFERER,
|
||||
"origin": "https://www.xiaohongshu.com",
|
||||
"x-requested-with": "XMLHttpRequest",
|
||||
}
|
||||
try:
|
||||
async with httpx.AsyncClient(
|
||||
headers=headers, follow_redirects=False, timeout=10
|
||||
) as client:
|
||||
current = url
|
||||
for _ in range(3):
|
||||
resp = await client.get(current)
|
||||
if resp.status_code >= 400:
|
||||
return None
|
||||
location = resp.headers.get("Location")
|
||||
if not location:
|
||||
return str(resp.url)
|
||||
# Location 可能是相对路径(如 /explore?...),需拼上当前 URL
|
||||
current = urllib.parse.urljoin(current, location)
|
||||
return current
|
||||
except Exception:
|
||||
logger.warning(f"xhslink 重定向失败: {url}")
|
||||
return None
|
||||
@@ -0,0 +1,264 @@
|
||||
"""通用视频下载 — yt-dlp + 直链探测"""
|
||||
|
||||
import asyncio
|
||||
import os
|
||||
import re
|
||||
import sys
|
||||
import tempfile
|
||||
from datetime import datetime
|
||||
from pathlib import Path
|
||||
from typing import Optional
|
||||
|
||||
from httpx import AsyncClient
|
||||
from nonebot import logger
|
||||
from yt_dlp import YoutubeDL
|
||||
from yt_dlp.utils import DownloadError
|
||||
|
||||
from ..utils import get_temp_root, slugify, ensure_unique_path
|
||||
|
||||
|
||||
def detect_platform(url: str) -> str:
|
||||
if "bilibili.com" in url or "b23.tv" in url:
|
||||
return "bilibili"
|
||||
if "twitter.com" in url or "x.com" in url:
|
||||
return "twitter"
|
||||
if "youtube.com" in url or "youtu.be" in url:
|
||||
return "youtube"
|
||||
if "douyin.com" in url or "v.douyin.com" in url or "iesdouyin.com" in url:
|
||||
return "douyin"
|
||||
return "other"
|
||||
|
||||
|
||||
def extract_uploader(info: dict) -> Optional[str]:
|
||||
"""从 yt-dlp info dict 提取上传者,优先级: uploader > channel > creator > uploader_id"""
|
||||
if not info:
|
||||
return None
|
||||
return (
|
||||
info.get("uploader")
|
||||
or info.get("channel")
|
||||
or info.get("creator")
|
||||
or info.get("uploader_id")
|
||||
)
|
||||
|
||||
|
||||
def get_ffmpeg_path() -> str:
|
||||
scripts_dir = os.path.dirname(sys.executable)
|
||||
ffmpeg_path = os.path.join(scripts_dir, "ffmpeg.exe")
|
||||
if os.path.exists(ffmpeg_path):
|
||||
return ffmpeg_path
|
||||
return "ffmpeg"
|
||||
|
||||
|
||||
def _get_data_dir() -> Path:
|
||||
"""获取 data/ 目录路径"""
|
||||
return Path(__file__).resolve().parent.parent / "data"
|
||||
|
||||
|
||||
async def _retry_download(
|
||||
loop: asyncio.AbstractEventLoop,
|
||||
url: str,
|
||||
base_opts: dict,
|
||||
max_retries: int = 3,
|
||||
) -> dict:
|
||||
"""带重试的 yt-dlp 下载,处理 RemoteDisconnected 等瞬态错误"""
|
||||
|
||||
def _run_yt():
|
||||
with YoutubeDL(base_opts) as ydl:
|
||||
return ydl.extract_info(url, download=True)
|
||||
|
||||
last_error = None
|
||||
for attempt in range(1, max_retries + 1):
|
||||
try:
|
||||
return await loop.run_in_executor(None, _run_yt)
|
||||
except DownloadError as e:
|
||||
last_error = e
|
||||
if attempt < max_retries:
|
||||
delay = 2 ** attempt # 2s, 4s, 8s
|
||||
logger.warning(
|
||||
f"yt-dlp 下载失败 (第 {attempt}/{max_retries} 次),"
|
||||
f"{delay}s 后重试: {str(e)[:120]}"
|
||||
)
|
||||
await asyncio.sleep(delay)
|
||||
else:
|
||||
logger.error(
|
||||
f"yt-dlp 重试 {max_retries} 次后仍失败: {str(e)[:120]}"
|
||||
)
|
||||
except Exception as e:
|
||||
# 非 DownloadError(如 OSError)不重试,直接抛出
|
||||
raise
|
||||
|
||||
raise last_error # type: ignore[misc]
|
||||
|
||||
|
||||
async def download_video(url: str) -> Optional[Path]:
|
||||
"""下载视频,支持直链和 yt-dlp"""
|
||||
|
||||
# ---------- 1. 直链探测 ----------
|
||||
direct_media_ext = re.search(
|
||||
r"\.(mp4|m3u8|ts|webm|mov|flv)(?:$|\?)", url, re.IGNORECASE
|
||||
)
|
||||
is_direct = bool(direct_media_ext)
|
||||
|
||||
if not is_direct:
|
||||
try:
|
||||
async with AsyncClient(follow_redirects=True, timeout=30) as client:
|
||||
head = await client.head(url, follow_redirects=True)
|
||||
ctype = head.headers.get("content-type", "")
|
||||
if ctype.startswith("video/") or "application/octet-stream" in ctype:
|
||||
is_direct = True
|
||||
except Exception:
|
||||
is_direct = False
|
||||
|
||||
if is_direct:
|
||||
temp_dir = tempfile.mkdtemp(prefix="direct_ytcache_", dir=get_temp_root("ytcache"))
|
||||
ext = "mp4"
|
||||
m = re.search(r"\.([a-zA-Z0-9]{2,5})(?:$|\?)", url)
|
||||
if m and len(m.group(1)) <= 5:
|
||||
ext = m.group(1)
|
||||
|
||||
url_stem = Path(url.split("?")[0]).stem or "video"
|
||||
slug_stem = slugify(url_stem, max_length=15)
|
||||
if not slug_stem:
|
||||
slug_stem = datetime.now().strftime("%H%M%S")
|
||||
time_suffix = datetime.now().strftime("%H%M%S")
|
||||
new_name = f"{slug_stem}_视频_{time_suffix}.{ext}"
|
||||
filename = os.path.join(temp_dir, new_name)
|
||||
|
||||
try:
|
||||
async with AsyncClient(follow_redirects=True, timeout=300) as client:
|
||||
async with client.stream("GET", url) as resp:
|
||||
resp.raise_for_status()
|
||||
with open(filename, "wb") as fh:
|
||||
async for chunk in resp.aiter_bytes(chunk_size=8192):
|
||||
fh.write(chunk)
|
||||
|
||||
final_path = ensure_unique_path(Path(filename))
|
||||
logger.info(f"直接下载完成: {final_path}")
|
||||
return final_path
|
||||
except Exception:
|
||||
logger.exception("直接下载失败,回退 yt-dlp")
|
||||
if os.path.exists(filename):
|
||||
os.remove(filename)
|
||||
|
||||
# ---------- 2. yt-dlp 下载 ----------
|
||||
platform = detect_platform(url)
|
||||
temp_dir = tempfile.mkdtemp(prefix="ytcache_", dir=get_temp_root("ytcache"))
|
||||
output_path = os.path.join(temp_dir, "%(title).80s.%(ext)s")
|
||||
|
||||
base_opts = {
|
||||
"outtmpl": output_path,
|
||||
"format": "bestvideo+bestaudio/best",
|
||||
"merge_output_format": "mp4",
|
||||
"noplaylist": True,
|
||||
"quiet": True,
|
||||
"ffmpeg_location": get_ffmpeg_path(),
|
||||
}
|
||||
|
||||
cookie_path = _get_data_dir() / "cookies.txt"
|
||||
|
||||
if platform in ("bilibili", "twitter", "youtube"):
|
||||
if not cookie_path.exists():
|
||||
raise RuntimeError(f"{platform} 需要 cookies.txt,但未找到")
|
||||
base_opts["cookiefile"] = str(cookie_path)
|
||||
logger.info(f"{platform} 使用 cookies.txt 下载")
|
||||
|
||||
ua = (
|
||||
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) "
|
||||
"AppleWebKit/537.36 (KHTML, like Gecko) "
|
||||
"Chrome/122.0.0.0 Safari/537.36"
|
||||
)
|
||||
|
||||
if platform == "bilibili":
|
||||
base_opts["http_headers"] = {
|
||||
"User-Agent": ua,
|
||||
"Referer": "https://www.bilibili.com/",
|
||||
"Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,image/webp,*/*;q=0.8",
|
||||
"Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
|
||||
"Accept-Encoding": "gzip, deflate, br",
|
||||
}
|
||||
base_opts["extractor_args"] = {
|
||||
"bilibili": {
|
||||
"header": [
|
||||
"Referer:https://www.bilibili.com/",
|
||||
f"User-Agent:{ua}",
|
||||
"Accept:text/html,application/xhtml+xml,application/xml;q=0.9,image/webp,*/*;q=0.8",
|
||||
"Accept-Language:zh-CN,zh;q=0.9,en;q=0.8",
|
||||
]
|
||||
}
|
||||
}
|
||||
elif platform == "twitter":
|
||||
base_opts["http_headers"] = {
|
||||
"User-Agent": ua,
|
||||
"Referer": "https://x.com/",
|
||||
"Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,image/webp,*/*;q=0.8",
|
||||
"Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
|
||||
}
|
||||
base_opts["extractor_args"] = {"twitter": {"api": ["syndication"]}}
|
||||
elif platform == "youtube":
|
||||
base_opts["http_headers"] = {
|
||||
"User-Agent": ua,
|
||||
"Referer": "https://www.youtube.com/",
|
||||
"Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,image/webp,*/*;q=0.8",
|
||||
"Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
|
||||
}
|
||||
|
||||
else:
|
||||
logger.info("其他平台,默认无 cookies 下载")
|
||||
|
||||
loop = asyncio.get_event_loop()
|
||||
|
||||
try:
|
||||
info = await _retry_download(loop, url, base_opts)
|
||||
except Exception:
|
||||
logger.exception("yt-dlp 下载失败")
|
||||
# YouTube: cookies 可能触发 bot 检测导致只返回图片无视频格式
|
||||
# 回退无 cookie 模式重试
|
||||
if platform == "youtube" and "cookiefile" in base_opts:
|
||||
logger.info("YouTube 回退无 cookies 模式重试...")
|
||||
base_opts.pop("cookiefile", None)
|
||||
base_opts.pop("http_headers", None)
|
||||
# 清理失败残留
|
||||
for f in Path(temp_dir).glob("*.*"):
|
||||
try:
|
||||
f.unlink()
|
||||
except Exception:
|
||||
pass
|
||||
try:
|
||||
info = await _retry_download(loop, url, base_opts, max_retries=2)
|
||||
except Exception:
|
||||
logger.exception("yt-dlp 无 cookies 重试也失败")
|
||||
return None
|
||||
else:
|
||||
return None
|
||||
|
||||
files = list(Path(temp_dir).glob("*.*"))
|
||||
if not files:
|
||||
return None
|
||||
|
||||
original_file = files[0]
|
||||
|
||||
# 构建新文件名
|
||||
uploader = extract_uploader(info or {})
|
||||
title = ((info or {}).get("title") or "").strip()
|
||||
|
||||
slug_title = slugify(title, max_length=15) if title else ""
|
||||
if not slug_title:
|
||||
slug_title = datetime.now().strftime("%H%M%S")
|
||||
|
||||
time_suffix = datetime.now().strftime("%H%M%S")
|
||||
if uploader:
|
||||
slug_uploader = slugify(str(uploader))
|
||||
new_stem = f"{slug_uploader}_{slug_title}_视频_{time_suffix}"
|
||||
else:
|
||||
new_stem = f"{slug_title}_视频_{time_suffix}"
|
||||
|
||||
new_path = ensure_unique_path(
|
||||
original_file.with_name(f"{new_stem}{original_file.suffix}")
|
||||
)
|
||||
original_file.rename(new_path)
|
||||
logger.info(
|
||||
f"yt-dlp 下载完成, 标题: {title}, "
|
||||
f"作者: {uploader}, 重命名: {new_path}"
|
||||
)
|
||||
|
||||
return new_path
|
||||
@@ -0,0 +1,9 @@
|
||||
"""handlers 包:消息入口与各平台解析处理。"""
|
||||
|
||||
from . import douyin, sender, universal # noqa: F401
|
||||
from .entry import ( # noqa: F401
|
||||
active_video_handler,
|
||||
auto_video_handler,
|
||||
dispatch_url,
|
||||
match_message,
|
||||
)
|
||||
@@ -0,0 +1,131 @@
|
||||
"""抖音视频/图文解析编排层"""
|
||||
|
||||
import os
|
||||
import re
|
||||
import urllib.parse
|
||||
from pathlib import Path
|
||||
from typing import Optional, Union
|
||||
|
||||
import httpx
|
||||
from nonebot import logger
|
||||
|
||||
from ..fetchers.douyin_api import fetch_douyin_content
|
||||
from ..fetchers.douyin_ssr import MOBILE_UA, fetch_douyin_note_ssr
|
||||
from ..models import DouyinFetchError
|
||||
from ..utils import parse_netscape_cookies
|
||||
from .sender import PendingMedia, _as_paths
|
||||
|
||||
SHORT_LINK_PATTERN = re.compile(r"(v\.douyin\.com/[A-Za-z0-9_\-]+)")
|
||||
|
||||
|
||||
def _get_data_dir() -> str:
|
||||
return os.path.join(os.path.dirname(__file__), "..", "data")
|
||||
|
||||
|
||||
async def _resolve_short_link(url: str) -> Optional[str]:
|
||||
"""短链重定向:httpx 取 Location(最多 3 跳),失败返回 None"""
|
||||
try:
|
||||
async with httpx.AsyncClient(
|
||||
headers={"User-Agent": MOBILE_UA},
|
||||
follow_redirects=False,
|
||||
timeout=10,
|
||||
) as client:
|
||||
current = url
|
||||
for _ in range(3):
|
||||
resp = await client.get(current)
|
||||
if resp.status_code >= 400:
|
||||
return None
|
||||
location = resp.headers.get("Location")
|
||||
if not location:
|
||||
final = str(resp.url)
|
||||
# 200 但仍是短链本身(如 JS 挑战壳页)→ 静默失败,显式记录;
|
||||
# 图文短链若此处失败将退化到 playwright 旧链路
|
||||
if "v.douyin.com" in final:
|
||||
logger.warning(f"短链返回挑战页(未跳转): {final}")
|
||||
return final
|
||||
# Location 可能是相对路径,需拼上当前 URL
|
||||
current = urllib.parse.urljoin(current, location)
|
||||
return current
|
||||
except Exception:
|
||||
logger.warning(f"短链重定向失败: {url}")
|
||||
return None
|
||||
|
||||
|
||||
async def parse_douyin(
|
||||
url: str,
|
||||
) -> tuple[Optional[str], Optional[Union[Path, list[Path]]], bool]:
|
||||
"""解析抖音链接(短链 / 全链接),返回 (title, file_paths, is_image_post)
|
||||
|
||||
分派规则(不能依赖 URL 路径:短链重定向后图文也统一变成
|
||||
iesdouyin.com/share/video/{id} 形态):
|
||||
- note 形态链接 → SSR 静态解析,失败不回退 playwright
|
||||
(其图文链路存在 aweme/post 取到作者其他作品的缺陷)
|
||||
- 其余形态 → 先试 SSR 按内容判定:有 images 即图文 → SSR 秒级下载;
|
||||
真视频 / SSR 失败 → playwright 拦截(保留全部清晰度能力)
|
||||
"""
|
||||
# 优先匹配短链接 v.douyin.com/xxx
|
||||
short = SHORT_LINK_PATTERN.search(url)
|
||||
if short:
|
||||
target_url = "https://" + short.group(1)
|
||||
# 短链先重定向拿到全链接,确定图文/视频类型后分派
|
||||
resolved = await _resolve_short_link(target_url)
|
||||
if resolved:
|
||||
logger.info(f"短链重定向: {target_url} -> {resolved}")
|
||||
target_url = resolved
|
||||
else:
|
||||
# 全链接: www.douyin.com/note/xxx 或 www.douyin.com/video/xxx
|
||||
m = re.search(
|
||||
r"(https?://(?:www\.)?douyin\.com/(?:note|video)/\d+)", url
|
||||
)
|
||||
if not m:
|
||||
return None, None, False
|
||||
target_url = m.group(1)
|
||||
|
||||
try:
|
||||
is_img_post = False
|
||||
cookies_path = os.path.join(_get_data_dir(), "cookies.txt")
|
||||
cookies = parse_netscape_cookies(cookies_path)
|
||||
# 先试 SSR(秒级、免浏览器),失败一律回退 playwright 链路。
|
||||
# 2026-08-13 起抖音对 iesdouyin SSR 端点整体降级(登录态也拿不到
|
||||
# videoInfoRes,只返回 33KB 壳页),note/video 形态统一回退,
|
||||
# 图文作品的"取到作者其他作品"缺陷由 douyin_api 按 aweme_id 精确匹配修复。
|
||||
try:
|
||||
title, file_path = await fetch_douyin_note_ssr(target_url, cookies)
|
||||
except DouyinFetchError:
|
||||
logger.info("SSR 解析失败,回退 playwright 链路")
|
||||
title, file_path = None, None
|
||||
if file_path is None:
|
||||
title, file_path = await fetch_douyin_content(
|
||||
target_url, cookies, 10, False
|
||||
)
|
||||
if isinstance(file_path, list):
|
||||
is_img_post = True
|
||||
return title, file_path, is_img_post
|
||||
except Exception:
|
||||
logger.exception(f"获取直链失败 {target_url}")
|
||||
return None, None, False
|
||||
|
||||
|
||||
async def process_douyin_res(
|
||||
title: str,
|
||||
file_paths: Union[Path, list[Path]],
|
||||
is_private: bool,
|
||||
image_post: bool,
|
||||
plan: str | None = None,
|
||||
) -> tuple[Optional[PendingMedia], Optional[str]]:
|
||||
"""下载已完成 → 打包为待发送媒体(不上传、不发送、不清理)
|
||||
|
||||
多级发送(temp 本地 → S3 链接 → 回退本地)由 send_pending_media 统一处理。
|
||||
"""
|
||||
if not file_paths:
|
||||
return None, None
|
||||
return (
|
||||
PendingMedia(
|
||||
files=_as_paths(file_paths),
|
||||
image_post=image_post,
|
||||
is_private=is_private,
|
||||
plan=plan,
|
||||
title=title,
|
||||
),
|
||||
None,
|
||||
)
|
||||
@@ -0,0 +1,266 @@
|
||||
"""消息入口与分派:自动解析 / 主动解析(文本链接 + QQ小程序/分享卡片)。"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import html
|
||||
import os
|
||||
import re
|
||||
from typing import Optional
|
||||
|
||||
from nonebot import on_message, logger
|
||||
from nonebot.adapters import Event
|
||||
from nonebot.rule import to_me
|
||||
from nonebot_plugin_alconna import UniMessage
|
||||
from nonebot_plugin_alconna.uniseg import get_target
|
||||
|
||||
from ..fetchers.bilibili_content import fetch_bilibili_content, resolve_short_link
|
||||
from ..fetchers.rednote_content import fetch_rednote_content
|
||||
from .douyin import parse_douyin, process_douyin_res
|
||||
from .sender import PendingMedia, send_pending_media
|
||||
from .universal import handle_universal
|
||||
from ..list_proc import AUTO_LINK_KEYWORDS, get_group_auto_link, verify_user
|
||||
|
||||
BASE_DIR = os.path.dirname(os.path.abspath(__file__))
|
||||
FILE_PATH = os.path.join(BASE_DIR, "data", "list.json")
|
||||
|
||||
URL_PATTERN = re.compile(r"(https?://\S+)")
|
||||
XCX_PATTERN = r"QQ小程序(?:&#93;|]|\])"
|
||||
|
||||
VALID_HOSTS = [
|
||||
"b23.tv",
|
||||
"bilibili.com",
|
||||
"youtube.com",
|
||||
"youtu.be",
|
||||
"douyin.com",
|
||||
"v.douyin.com",
|
||||
"iesdouyin.com",
|
||||
"m.douyin.com",
|
||||
"jingxuan.douyin.com",
|
||||
"x.com",
|
||||
"twitter.com",
|
||||
"xiaohongshu.com",
|
||||
"xhslink.com",
|
||||
"xhslink.cn",
|
||||
]
|
||||
|
||||
auto_video_handler = on_message(priority=10, block=False)
|
||||
active_video_handler = on_message(priority=10, block=False, rule=to_me())
|
||||
|
||||
|
||||
async def _check_access(
|
||||
event: Event, *, auto_only: bool = False, msg: str | None = None
|
||||
) -> tuple[bool, str | None]:
|
||||
"""统一权限检查。
|
||||
|
||||
auto_only=True → 自动解析:需白名单 + 开启自动解析,或 auto_link 关键词命中。
|
||||
auto_only=False → 主动触发:需非黑名单,群聊还需白名单。
|
||||
|
||||
Returns:
|
||||
(allowed, plan) — plan 用于 S3 路由,不允许时为 None
|
||||
"""
|
||||
white, black, auto, plan = await verify_user(event)
|
||||
target = get_target(event)
|
||||
|
||||
if target.private:
|
||||
if auto_only:
|
||||
logger.info("权限分析:自动解析不处理私聊")
|
||||
return False, None
|
||||
if black:
|
||||
logger.info(f"权限分析:黑名单用户私聊,不回复: {event.get_user_id()}")
|
||||
return False, None
|
||||
logger.info("权限分析:私聊,直接解析")
|
||||
return True, None
|
||||
|
||||
group_id = str(event.group_id)
|
||||
|
||||
if not white:
|
||||
logger.info(f"权限分析:群 {group_id} 不在白名单,不做处理")
|
||||
return False, None
|
||||
|
||||
if auto_only and not auto:
|
||||
if msg is not None and await _match_auto_link(event, msg):
|
||||
logger.info(f"权限分析:群 {group_id} 未开启自动解析,但自动链接关键词命中")
|
||||
else:
|
||||
logger.info(f"权限分析:群 {group_id} 未开启自动解析")
|
||||
return False, None
|
||||
|
||||
if not auto_only and black:
|
||||
logger.info(f"权限分析:黑名单用户,不回复: {event.get_user_id()}")
|
||||
return False, None
|
||||
|
||||
logger.info(
|
||||
f"权限分析:群 {group_id} 权限通过 — "
|
||||
f"自动解析: {auto}, 方案: {plan or '默认(PLANC)'}"
|
||||
)
|
||||
return True, plan
|
||||
|
||||
|
||||
async def _match_auto_link(event: Event, msg: str) -> bool:
|
||||
"""消息中的 URL 是否命中群配置的 auto_link 关键词。"""
|
||||
keywords = await get_group_auto_link(event)
|
||||
if not keywords:
|
||||
return False
|
||||
urls = URL_PATTERN.findall(msg)
|
||||
if not urls:
|
||||
return False
|
||||
for kw in keywords:
|
||||
domains = AUTO_LINK_KEYWORDS.get(kw, (kw,))
|
||||
if any(any(domain in url for domain in domains) for url in urls):
|
||||
logger.info(f"自动链接:关键词 {kw} 命中消息 {urls}")
|
||||
return True
|
||||
return False
|
||||
|
||||
|
||||
@auto_video_handler.handle()
|
||||
async def handle_auto_video(event: Event):
|
||||
msg = str(event.get_message()).strip()
|
||||
allowed, plan = await _check_access(event, auto_only=True, msg=msg)
|
||||
if allowed:
|
||||
await match_message(event, plan=plan)
|
||||
|
||||
|
||||
@active_video_handler.handle()
|
||||
async def handle_active_video(event: Event):
|
||||
allowed, plan = await _check_access(event, auto_only=False)
|
||||
if allowed:
|
||||
await match_message(event, plan=plan)
|
||||
|
||||
|
||||
async def match_message(event: Event, plan: str | None = None):
|
||||
"""消息匹配与分派:文本链接 / QQ小程序卡片统一走 dispatch_url。"""
|
||||
msg = str(event.get_message()).strip()
|
||||
logger.info(f"消息解析:获取到的消息:{msg}")
|
||||
is_private = get_target(event).private
|
||||
|
||||
message = None
|
||||
public_url = None
|
||||
|
||||
if re.search(XCX_PATTERN, msg) or "CQ:json" in msg or "CQ:share" in msg:
|
||||
logger.info("消息解析:检测到 CQ 卡片")
|
||||
url = await _extract_xcx_url(msg)
|
||||
logger.info(f"消息解析:卡片链接:{url}")
|
||||
if not url or not any(domain in url for domain in VALID_HOSTS):
|
||||
return
|
||||
message, public_url = await dispatch_url(url, is_private, plan=plan)
|
||||
if not message:
|
||||
return
|
||||
else:
|
||||
urls = URL_PATTERN.findall(msg)
|
||||
for url in urls:
|
||||
logger.info(f"消息解析:作品链接:{url}")
|
||||
message, public_url = await dispatch_url(url, is_private, plan=plan)
|
||||
if message:
|
||||
break
|
||||
if not message:
|
||||
return
|
||||
|
||||
if isinstance(message, PendingMedia):
|
||||
ok, pub = await send_pending_media(message)
|
||||
if pub:
|
||||
await UniMessage.text(f"{pub}").send()
|
||||
if not ok:
|
||||
await UniMessage.text(f"媒体发送失败:{url}").send()
|
||||
else:
|
||||
if public_url:
|
||||
await UniMessage.text(f"{public_url}").send()
|
||||
await message.send()
|
||||
|
||||
|
||||
async def dispatch_url(
|
||||
url: str,
|
||||
is_private: bool,
|
||||
plan: str | None = None,
|
||||
) -> tuple[Optional[UniMessage], Optional[str]]:
|
||||
"""按平台分派解析(文本链接与小程序卡片共用)。"""
|
||||
url = url.rstrip(",。!?、;:)】》\"')")
|
||||
|
||||
if "b23.tv" in url or "bili2233.cn" in url:
|
||||
resolved = await resolve_short_link(url)
|
||||
if resolved:
|
||||
logger.info(f"b23 短链重定向: {url} -> {resolved}")
|
||||
url = resolved
|
||||
|
||||
if "douyin.com" in url or "v.douyin.com" in url or "iesdouyin.com" in url:
|
||||
await UniMessage.text("检测到链接,正在处理,请稍候...").send()
|
||||
try:
|
||||
title, parsed_path, image_post = await parse_douyin(url)
|
||||
return await process_douyin_res(
|
||||
title, parsed_path, is_private, image_post, plan=plan
|
||||
)
|
||||
except Exception:
|
||||
await UniMessage.text(f"无法解析到媒体:{url}").send()
|
||||
logger.exception(f"媒体解析:无法解析到媒体:{url}")
|
||||
return None, None
|
||||
|
||||
if any(
|
||||
kw in url
|
||||
for kw in (
|
||||
"bilibili.com/opus",
|
||||
"bilibili.com/dynamic",
|
||||
"t.bilibili.com",
|
||||
"bilibili.com/read",
|
||||
)
|
||||
):
|
||||
await UniMessage.text("检测到链接,正在处理,请稍候...").send()
|
||||
try:
|
||||
title, parsed_path = await fetch_bilibili_content(url)
|
||||
if not title and parsed_path is None:
|
||||
await UniMessage.text(f"无法解析到内容:{url}").send()
|
||||
return None, None
|
||||
if parsed_path == []:
|
||||
await UniMessage.text(title).send()
|
||||
return None, None
|
||||
return await process_douyin_res(
|
||||
title, parsed_path, is_private,
|
||||
isinstance(parsed_path, list), plan=plan,
|
||||
)
|
||||
except Exception:
|
||||
await UniMessage.text(f"无法解析到内容:{url}").send()
|
||||
logger.exception(f"B站内容解析:无法解析:{url}")
|
||||
return None, None
|
||||
|
||||
if "xiaohongshu.com" in url or "xhslink." in url:
|
||||
await UniMessage.text("检测到链接,正在处理,请稍候...").send()
|
||||
try:
|
||||
title, parsed_path = await fetch_rednote_content(url)
|
||||
if not title and parsed_path is None:
|
||||
await UniMessage.text(f"无法解析到内容:{url}").send()
|
||||
return None, None
|
||||
if parsed_path == []:
|
||||
await UniMessage.text(title).send()
|
||||
return None, None
|
||||
return await process_douyin_res(
|
||||
title, parsed_path, is_private,
|
||||
isinstance(parsed_path, list), plan=plan,
|
||||
)
|
||||
except Exception:
|
||||
await UniMessage.text(f"无法解析到内容:{url}").send()
|
||||
logger.exception(f"小红书解析:无法解析:{url}")
|
||||
return None, None
|
||||
|
||||
if any(domain in url for domain in VALID_HOSTS):
|
||||
await UniMessage.text("检测到链接,正在处理,请稍候...").send()
|
||||
try:
|
||||
return await handle_universal(url, is_private, plan=plan)
|
||||
except Exception as e:
|
||||
logger.exception(e)
|
||||
await UniMessage.text("下载过程中出现错误。").send()
|
||||
return None, None
|
||||
|
||||
return None, None
|
||||
|
||||
|
||||
async def _extract_xcx_url(msg: str) -> Optional[str]:
|
||||
"""从 CQ 卡片消息中提取跳转 URL(保留 query 参数)。"""
|
||||
match = re.search(r'"qqdocurl":"(.*?)"', msg)
|
||||
if not match:
|
||||
match = re.search(r'"jumpUrl":"(.*?)"', msg)
|
||||
if not match:
|
||||
logger.warning("未找到 qqdocurl/jumpUrl 字段")
|
||||
return None
|
||||
|
||||
raw_url = match.group(1)
|
||||
unescaped = html.unescape(raw_url)
|
||||
cleaned_url = unescaped.replace(r"\/", "/")
|
||||
logger.info(f"卡片提取链接: {cleaned_url}")
|
||||
return cleaned_url
|
||||
@@ -0,0 +1,107 @@
|
||||
"""媒体多级发送 — temp 本地文件优先,失败逐级降级
|
||||
|
||||
发送策略(2026-08-22 用户需求):
|
||||
1. temp 本地文件直接发送(最快,不经 S3)
|
||||
2. 失败 → 上传本地 S3,用预签名链接发送
|
||||
3. 再失败 → 回退 temp 本地文件再发一次
|
||||
|
||||
temp 下的文件发送成功后也不清理(用户手动处理 data/temp)。
|
||||
"""
|
||||
|
||||
from dataclasses import dataclass
|
||||
from pathlib import Path
|
||||
from typing import Optional, Union
|
||||
|
||||
from nonebot import logger
|
||||
from nonebot_plugin_alconna import UniMessage
|
||||
|
||||
from ..storage.s3 import upload_with_plan
|
||||
|
||||
|
||||
@dataclass
|
||||
class PendingMedia:
|
||||
"""待发送媒体:本地文件 + 上传元数据(发送前不做任何上传/清理)"""
|
||||
|
||||
files: list[Path]
|
||||
image_post: bool = False
|
||||
is_private: bool = False
|
||||
plan: Optional[str] = None
|
||||
title: str = ""
|
||||
|
||||
|
||||
def _as_paths(file_paths: Union[Path, list[Path]]) -> list[Path]:
|
||||
if isinstance(file_paths, list):
|
||||
return [Path(p) for p in file_paths]
|
||||
return [Path(file_paths)]
|
||||
|
||||
|
||||
def _build_local_msg(files: list[Path], image_post: bool) -> UniMessage:
|
||||
"""本地文件版消息(mp4 → 视频,其余 → 图片)"""
|
||||
msg = UniMessage()
|
||||
for fp in files:
|
||||
if fp.suffix.lower() == ".mp4":
|
||||
msg.video(path=fp)
|
||||
else:
|
||||
msg.image(path=fp)
|
||||
return msg
|
||||
|
||||
|
||||
def _build_s3_msg(
|
||||
media: PendingMedia,
|
||||
) -> tuple[UniMessage, Optional[str]]:
|
||||
"""上传本地 S3 并构建链接版消息,返回 (message, public_url)"""
|
||||
msg = UniMessage()
|
||||
public_url = None
|
||||
for fp in media.files:
|
||||
local_url, pub = upload_with_plan(
|
||||
fp,
|
||||
plan=media.plan,
|
||||
is_private=media.is_private,
|
||||
title=media.title,
|
||||
image_post=media.image_post,
|
||||
)
|
||||
if not local_url:
|
||||
raise RuntimeError(f"上传本地 S3 失败: {fp}")
|
||||
if pub:
|
||||
public_url = pub
|
||||
if fp.suffix.lower() == ".mp4":
|
||||
msg.video(url=local_url)
|
||||
else:
|
||||
msg.image(url=local_url)
|
||||
return msg, public_url
|
||||
|
||||
|
||||
async def send_pending_media(media: PendingMedia) -> tuple[bool, Optional[str]]:
|
||||
"""多级发送,返回 (是否成功, public_url)
|
||||
|
||||
public_url 仅在走 S3 链接发送成功时返回(调用方决定是否发文字)。
|
||||
temp 文件发送成功后保留(用户手动清理 data/temp)。
|
||||
"""
|
||||
if not media.files:
|
||||
return False, None
|
||||
|
||||
# ── 1. temp 本地文件直接发送 ──────────────────────────────
|
||||
try:
|
||||
await _build_local_msg(media.files, media.image_post).send()
|
||||
logger.info("媒体发送成功(temp 本地文件直达)")
|
||||
return True, None
|
||||
except Exception as e:
|
||||
logger.warning(f"temp 本地文件发送失败,切换本地 S3 链接: {e}")
|
||||
|
||||
# ── 2. 上传本地 S3 → 链接发送 ─────────────────────────────
|
||||
try:
|
||||
s3_msg, public_url = _build_s3_msg(media)
|
||||
await s3_msg.send()
|
||||
logger.info("媒体发送成功(本地 S3 链接)")
|
||||
return True, public_url
|
||||
except Exception as e:
|
||||
logger.warning(f"本地 S3 链接发送失败,回退 temp 本地文件: {e}")
|
||||
|
||||
# ── 3. 回退:temp 本地文件再发一次 ────────────────────────
|
||||
try:
|
||||
await _build_local_msg(media.files, media.image_post).send()
|
||||
logger.info("媒体发送成功(回退 temp 本地文件)")
|
||||
return True, None
|
||||
except Exception as e:
|
||||
logger.exception(f"回退发送失败: {e}")
|
||||
return False, None
|
||||
@@ -0,0 +1,40 @@
|
||||
"""通用平台视频解析编排层 — B站 / YouTube / Twitter 等"""
|
||||
|
||||
from pathlib import Path
|
||||
from typing import Optional
|
||||
|
||||
from nonebot import logger
|
||||
from nonebot_plugin_alconna import UniMessage
|
||||
|
||||
from ..fetchers.video_downloader import download_video
|
||||
from .sender import PendingMedia, _as_paths
|
||||
|
||||
|
||||
async def handle_universal(
|
||||
url: str,
|
||||
is_private: bool,
|
||||
plan: str | None = None,
|
||||
) -> tuple[Optional[PendingMedia], Optional[str]]:
|
||||
"""
|
||||
下载通用平台视频 → 打包待发送媒体(上传/发送由 sender 多级处理)
|
||||
|
||||
Returns:
|
||||
(PendingMedia, public_url) — None 表示下载失败
|
||||
"""
|
||||
video_file = await download_video(url)
|
||||
if not video_file:
|
||||
await UniMessage.text("视频下载失败。").send()
|
||||
return None, None
|
||||
|
||||
logger.info(f"文件路径:{video_file}")
|
||||
|
||||
return (
|
||||
PendingMedia(
|
||||
files=_as_paths(video_file),
|
||||
image_post=False,
|
||||
is_private=is_private,
|
||||
plan=plan,
|
||||
title="title",
|
||||
),
|
||||
None,
|
||||
)
|
||||
@@ -0,0 +1,364 @@
|
||||
import json
|
||||
import random
|
||||
import asyncio
|
||||
import os
|
||||
from typing import List
|
||||
|
||||
from nonebot.adapters.onebot.v11 import MessageSegment, PokeNotifyEvent, Message
|
||||
from nonebot.plugin.on import on_notice, on_command
|
||||
from nonebot import on_type, on_message, logger
|
||||
from nonebot.adapters import Event
|
||||
from nonebot.matcher import Matcher
|
||||
from nonebot.params import CommandArg
|
||||
from nonebot.rule import to_me
|
||||
from nonebot_plugin_alconna import UniMessage, get_target
|
||||
|
||||
from hexi.plugins.nonebot_plugin_hexi_core.custom_utils import check_admin
|
||||
# from hexi.plugins.nonebot_plugin_hexi_core.MessageUtils import send_poke
|
||||
|
||||
BASE_DIR = os.path.dirname(os.path.abspath(__file__))
|
||||
FILE_PATH = os.path.join(BASE_DIR, "data", "list.json")
|
||||
|
||||
USER_DATA: dict[str, str] = {}
|
||||
_LOADED = False
|
||||
LOCK = asyncio.Lock()
|
||||
|
||||
add_black_list = on_command("添加黑名单",rule=check_admin)
|
||||
add_white_list = on_command("添加白名单",rule=check_admin)
|
||||
add_auto_list = on_command("添加自动名单",rule=check_admin)
|
||||
set_plan_cmd = on_command("设置方案", rule=check_admin)
|
||||
set_auto_link_cmd = on_command("设置自动链接", rule=check_admin)
|
||||
|
||||
# 自动链接关键词 → 域名匹配表(auto_link 配置项使用)
|
||||
AUTO_LINK_KEYWORDS = {
|
||||
"xhs": ("xiaohongshu.com", "xhslink.com", "xhslink.cn"),
|
||||
"bilibili": ("bilibili.com", "b23.tv", "bili2233.cn"),
|
||||
"b23": ("bilibili.com", "b23.tv", "bili2233.cn"),
|
||||
"douyin": ("douyin.com", "v.douyin.com", "iesdouyin.com",
|
||||
"m.douyin.com", "jingxuan.douyin.com"),
|
||||
"yt": ("youtube.com", "youtu.be"),
|
||||
"youtube": ("youtube.com", "youtu.be"),
|
||||
"x": ("x.com", "twitter.com"),
|
||||
"twitter": ("x.com", "twitter.com"),
|
||||
}
|
||||
|
||||
|
||||
@add_black_list.handle()
|
||||
async def handle_add_black(event: Event, matcher: Matcher):
|
||||
input_id = event.get_message().extract_plain_text().strip()
|
||||
cmd = matcher.state["_prefix"]["command"][0]
|
||||
input_id = input_id.replace(cmd, "").strip()
|
||||
ok = await _add_to_blacklist(input_id)
|
||||
if ok:
|
||||
await UniMessage.text(f"{input_id} 已加入黑名单").send()
|
||||
else:
|
||||
await UniMessage.text(f"{input_id} 已在黑名单中").send()
|
||||
|
||||
|
||||
@add_white_list.handle()
|
||||
async def handle_add_white(event: Event, matcher: Matcher):
|
||||
input_id = event.get_message().extract_plain_text().strip()
|
||||
cmd = matcher.state["_prefix"]["command"][0]
|
||||
input_id = input_id.replace(cmd, "").strip()
|
||||
ok = await add_white_user(input_id)
|
||||
if ok:
|
||||
await UniMessage.text(f"群 {input_id} 已加入白名单").send()
|
||||
else:
|
||||
await UniMessage.text(f"群 {input_id} 已在白名单中").send()
|
||||
|
||||
|
||||
@add_auto_list.handle()
|
||||
async def handle_add_auto(event: Event, matcher: Matcher):
|
||||
input_id = event.get_message().extract_plain_text().strip()
|
||||
cmd = matcher.state["_prefix"]["command"][0]
|
||||
input_id = input_id.replace(cmd, "").strip()
|
||||
ok = await add_auto_user(input_id)
|
||||
if ok:
|
||||
await UniMessage.text(f"群 {input_id} 已开启自动解析").send()
|
||||
else:
|
||||
await UniMessage.text(f"群 {input_id} 已开启自动解析,无需重复设置").send()
|
||||
|
||||
|
||||
@set_plan_cmd.handle()
|
||||
async def handle_set_plan(event: Event, matcher: Matcher):
|
||||
input_text = event.get_message().extract_plain_text().strip()
|
||||
cmd = matcher.state["_prefix"]["command"][0]
|
||||
args = input_text.replace(cmd, "").strip().split()
|
||||
if len(args) != 2:
|
||||
await UniMessage.text("用法:设置方案 <群号> A/B").send()
|
||||
return
|
||||
group_id, plan = args[0], args[1].upper()
|
||||
ok = await set_group_plan(group_id, plan)
|
||||
if ok:
|
||||
await UniMessage.text(f"群 {group_id} 存储方案已设为 {plan}").send()
|
||||
elif plan not in ("A", "B"):
|
||||
await UniMessage.text("方案必须是 A 或 B").send()
|
||||
else:
|
||||
await UniMessage.text(f"群 {group_id} 不在白名单中,请先添加白名单").send()
|
||||
|
||||
@set_auto_link_cmd.handle()
|
||||
async def handle_set_auto_link(event: Event, matcher: Matcher):
|
||||
input_text = event.get_message().extract_plain_text().strip()
|
||||
cmd = matcher.state["_prefix"]["command"][0]
|
||||
args = input_text.replace(cmd, "").strip().split()
|
||||
if len(args) < 2:
|
||||
await UniMessage.text(
|
||||
f"用法:设置自动链接 <群号> <关键词...>(关键词:{'/'.join(AUTO_LINK_KEYWORDS)})"
|
||||
).send()
|
||||
return
|
||||
group_id = args[0]
|
||||
keywords = args[1:]
|
||||
ok = await set_group_auto_link(group_id, keywords)
|
||||
if ok:
|
||||
if keywords:
|
||||
await UniMessage.text(
|
||||
f"群 {group_id} 自动链接关键词已设为: {'、'.join(keywords)}\n"
|
||||
"匹配到对应平台链接时,即使未开启自动解析也会自动下载"
|
||||
).send()
|
||||
else:
|
||||
await UniMessage.text(f"群 {group_id} 的自动链接关键词已清空").send()
|
||||
else:
|
||||
await UniMessage.text(f"群 {group_id} 不在白名单中,请先添加白名单").send()
|
||||
|
||||
|
||||
async def _safe_write_json(data, file_path: str):
|
||||
tmp_path = file_path + ".tmp"
|
||||
|
||||
# 在线程池中执行耗时的文件写入
|
||||
await asyncio.to_thread(_write_json_sync, data, tmp_path)
|
||||
|
||||
# os.replace 是轻量级系统调用,通常很快,可直接在主线程执行
|
||||
# (也可放 to_thread,但一般没必要)
|
||||
os.replace(tmp_path, file_path)
|
||||
|
||||
def _write_json_sync(data, tmp_path: str):
|
||||
"""同步写入函数,供 to_thread 调用"""
|
||||
with open(tmp_path, "w", encoding="utf-8") as f:
|
||||
json.dump(data, f, ensure_ascii=False, indent=2)
|
||||
|
||||
|
||||
def _migrate_to_v2(data: dict) -> tuple[dict, bool]:
|
||||
"""将旧格式(平铺数组)转换为新格式(group-centric map)"""
|
||||
if "groups" in data:
|
||||
return data, False # 已是 v2
|
||||
|
||||
white = data.get("WHITE_LIST", [])
|
||||
auto_list = data.get("AUTO_ANALYSIS", [])
|
||||
pa = data.get("PLANA", [])
|
||||
pb = data.get("PLANB", [])
|
||||
|
||||
groups: dict[str, dict] = {}
|
||||
for gid in white:
|
||||
entry: dict = {}
|
||||
if gid in auto_list:
|
||||
entry["auto"] = True
|
||||
if gid in pa:
|
||||
entry["plan"] = "A"
|
||||
elif gid in pb:
|
||||
entry["plan"] = "B"
|
||||
groups[gid] = entry
|
||||
|
||||
return {
|
||||
"groups": groups,
|
||||
"blacklist": data.get("BLACK_LIST", []),
|
||||
}, True
|
||||
|
||||
async def load_list(file_path: str = FILE_PATH):
|
||||
global USER_DATA, _LOADED
|
||||
|
||||
if _LOADED:
|
||||
return
|
||||
|
||||
async with LOCK:
|
||||
if _LOADED:
|
||||
return
|
||||
|
||||
exists = await asyncio.to_thread(os.path.exists, file_path)
|
||||
if not exists:
|
||||
USER_DATA = {"groups": {}, "blacklist": []}
|
||||
await _safe_write_json(USER_DATA, file_path)
|
||||
else:
|
||||
content = await asyncio.to_thread(_read_file_sync, file_path)
|
||||
raw = json.loads(content)
|
||||
USER_DATA, migrated = _migrate_to_v2(raw)
|
||||
if migrated:
|
||||
await _safe_write_json(USER_DATA, file_path)
|
||||
logger.info("list.json 已从旧格式迁移为新 group-centric 格式")
|
||||
|
||||
_LOADED = True
|
||||
|
||||
|
||||
def _read_file_sync(file_path: str) -> str:
|
||||
with open(file_path, "r", encoding="utf-8") as f:
|
||||
return f.read()
|
||||
|
||||
async def verify_user(event: Event) -> tuple[bool, bool, bool, str | None]:
|
||||
"""
|
||||
验证用户/群权限
|
||||
|
||||
Returns:
|
||||
is_white — 群在白名单
|
||||
is_black — 用户在黑名单
|
||||
is_auto — 群开启自动解析
|
||||
plan — 存储方案 "A" / "B" / None
|
||||
"""
|
||||
await load_list()
|
||||
user_id = event.get_user_id()
|
||||
|
||||
groups = USER_DATA.get("groups", {})
|
||||
blacklist = USER_DATA.get("blacklist", [])
|
||||
|
||||
is_black = user_id in blacklist
|
||||
plan = None
|
||||
is_auto = False
|
||||
|
||||
if get_target(event).private:
|
||||
logger.debug(f"[DEBUG] private chat user_id: {repr(user_id)}")
|
||||
return True, is_black, False, None
|
||||
|
||||
group_id = str(event.group_id)
|
||||
group_config = groups.get(group_id, {})
|
||||
is_white = group_id in groups
|
||||
is_auto = group_config.get("auto", False)
|
||||
plan = group_config.get("plan")
|
||||
|
||||
logger.debug(f"[DEBUG] user_id: {repr(user_id)} group_id: {group_id}")
|
||||
logger.debug(f"[DEBUG] is_white: {is_white} is_black: {is_black} "
|
||||
f"is_auto: {is_auto} plan: {plan}")
|
||||
|
||||
return is_white, is_black, is_auto, plan
|
||||
|
||||
|
||||
async def _add_to_blacklist(new_id: str) -> bool:
|
||||
await load_list()
|
||||
async with LOCK:
|
||||
lst: list = USER_DATA.get("blacklist", [])
|
||||
if new_id in lst:
|
||||
return False
|
||||
lst.append(new_id)
|
||||
USER_DATA["blacklist"] = lst
|
||||
await _safe_write_json(USER_DATA, FILE_PATH)
|
||||
return True
|
||||
|
||||
|
||||
async def add_white_user(new_id: str) -> bool:
|
||||
"""添加群到白名单(groups map)"""
|
||||
await load_list()
|
||||
async with LOCK:
|
||||
groups: dict = USER_DATA.get("groups", {})
|
||||
if new_id in groups:
|
||||
return False
|
||||
groups[new_id] = {}
|
||||
USER_DATA["groups"] = groups
|
||||
await _safe_write_json(USER_DATA, FILE_PATH)
|
||||
return True
|
||||
|
||||
|
||||
async def add_auto_user(new_id: str) -> bool:
|
||||
"""设置群自动解析(群不在白名单则自动加入)"""
|
||||
await load_list()
|
||||
async with LOCK:
|
||||
groups: dict = USER_DATA.get("groups", {})
|
||||
if new_id not in groups:
|
||||
groups[new_id] = {}
|
||||
if groups[new_id].get("auto"):
|
||||
return False
|
||||
groups[new_id]["auto"] = True
|
||||
USER_DATA["groups"] = groups
|
||||
await _safe_write_json(USER_DATA, FILE_PATH)
|
||||
return True
|
||||
|
||||
|
||||
async def set_group_plan(group_id: str, plan: str) -> bool:
|
||||
"""设置群的存储方案(A 或 B),群必须在白名单中"""
|
||||
if plan not in ("A", "B"):
|
||||
return False
|
||||
await load_list()
|
||||
async with LOCK:
|
||||
groups: dict = USER_DATA.get("groups", {})
|
||||
if group_id not in groups:
|
||||
return False
|
||||
groups[group_id]["plan"] = plan
|
||||
USER_DATA["groups"] = groups
|
||||
await _safe_write_json(USER_DATA, FILE_PATH)
|
||||
return True
|
||||
|
||||
|
||||
async def get_group_auto_link(event: Event) -> list[str]:
|
||||
"""获取群配置的自动链接关键词(auto_link),私聊返回空列表"""
|
||||
await load_list()
|
||||
if get_target(event).private:
|
||||
return []
|
||||
group_id = str(event.group_id)
|
||||
return USER_DATA.get("groups", {}).get(group_id, {}).get("auto_link", [])
|
||||
|
||||
|
||||
async def set_group_auto_link(group_id: str, keywords: list[str]) -> bool:
|
||||
"""设置群的自动链接关键词(auto_link),群必须在白名单中
|
||||
|
||||
keywords 为空列表时清空该配置。
|
||||
"""
|
||||
await load_list()
|
||||
async with LOCK:
|
||||
groups: dict = USER_DATA.get("groups", {})
|
||||
if group_id not in groups:
|
||||
return False
|
||||
if keywords:
|
||||
groups[group_id]["auto_link"] = keywords
|
||||
else:
|
||||
groups[group_id].pop("auto_link", None)
|
||||
USER_DATA["groups"] = groups
|
||||
await _safe_write_json(USER_DATA, FILE_PATH)
|
||||
return True
|
||||
|
||||
|
||||
# ── Web「分组配置」同步读写(list.json) ─────────────────────────────
|
||||
def get_group_config_sync() -> dict:
|
||||
"""Web 读取用:确保已加载并返回 list.json 的完整结构(dict)。"""
|
||||
global USER_DATA, _LOADED
|
||||
if not _LOADED:
|
||||
if os.path.exists(FILE_PATH):
|
||||
try:
|
||||
raw = json.loads(_read_file_sync(FILE_PATH))
|
||||
USER_DATA, _ = _migrate_to_v2(raw)
|
||||
except Exception: # noqa: BLE001
|
||||
USER_DATA = {"groups": {}, "blacklist": []}
|
||||
else:
|
||||
USER_DATA = {"groups": {}, "blacklist": []}
|
||||
_LOADED = True
|
||||
return USER_DATA
|
||||
|
||||
|
||||
def set_group_config_sync(new_data: dict) -> bool:
|
||||
"""Web 保存用:校验结构 → 更新内存 → 同步写回 list.json。
|
||||
|
||||
只保留合法字段(auto/plan/auto_link/blacklist),避免脏数据。
|
||||
"""
|
||||
global USER_DATA
|
||||
if not isinstance(new_data, dict):
|
||||
return False
|
||||
groups = new_data.get("groups", {})
|
||||
blacklist = new_data.get("blacklist", [])
|
||||
if not isinstance(groups, dict) or not isinstance(blacklist, list):
|
||||
return False
|
||||
|
||||
norm_groups: dict[str, dict] = {}
|
||||
for gid, entry in groups.items():
|
||||
if not isinstance(entry, dict):
|
||||
continue
|
||||
e: dict = {}
|
||||
if entry.get("auto"):
|
||||
e["auto"] = True
|
||||
plan = str(entry.get("plan", "")).upper()
|
||||
if plan in ("A", "B"):
|
||||
e["plan"] = plan
|
||||
if isinstance(entry.get("auto_link"), list):
|
||||
e["auto_link"] = [str(x) for x in entry["auto_link"]]
|
||||
norm_groups[str(gid)] = e
|
||||
|
||||
norm = {"groups": norm_groups, "blacklist": [str(x) for x in blacklist]}
|
||||
USER_DATA = norm
|
||||
_write_json_sync(norm, FILE_PATH + ".tmp")
|
||||
os.replace(FILE_PATH + ".tmp", FILE_PATH)
|
||||
return True
|
||||
|
||||
@@ -0,0 +1,10 @@
|
||||
class DouyinFetchError(Exception):
|
||||
"""抖音数据抓取自定义异常"""
|
||||
|
||||
pass
|
||||
|
||||
|
||||
class ContentFetchError(Exception):
|
||||
"""多平台内容(B站动态/小红书笔记等)抓取通用异常"""
|
||||
|
||||
pass
|
||||
@@ -0,0 +1,258 @@
|
||||
"""统一 S3 存储模块 — 合并本地局域网 S3 和公网 MinIO"""
|
||||
|
||||
from pathlib import Path
|
||||
from time import strftime, localtime
|
||||
|
||||
import boto3
|
||||
from botocore.config import Config
|
||||
from nonebot import logger
|
||||
|
||||
from hexi.web_config import get_effective_value
|
||||
|
||||
# 该插件模块名,用于读取统一配置值库(Web 修改后生效)
|
||||
_PLUGIN_ID = "hexi.plugins.nonebot_plugin_video_analysis"
|
||||
|
||||
|
||||
def _ev(name: str, default):
|
||||
"""读取统一配置生效值(用户值库 > getter > 默认),来源无关。"""
|
||||
return get_effective_value(_PLUGIN_ID, name, default)
|
||||
|
||||
|
||||
# ============================= 本地局域网 PLANA_S3 配置 =============================
|
||||
LOCAL_S3_PLANA_ENDPOINT = "http://192.168.2.15:5246"
|
||||
LOCAL_S3_PLANA_ACCESS_KEY = "vS8Lf7UoiS5cxL+mz+CH"
|
||||
LOCAL_S3_PLANA_SECRET_KEY = "Fi3phPW10CsZWuiQCerK4gxaIIJp41w5SQNlnDRb"
|
||||
LOCAL_S3_PLANA_BUCKET = "PLANA"
|
||||
|
||||
# ============================= 本地局域网 PLANB_S3 配置 =============================
|
||||
LOCAL_S3_PLANB_ENDPOINT = "http://192.168.2.15:5246"
|
||||
LOCAL_S3_PLANB_ACCESS_KEY = "vS8Lf7UoiS5cxL+mz+CH"
|
||||
LOCAL_S3_PLANB_SECRET_KEY = "Fi3phPW10CsZWuiQCerK4gxaIIJp41w5SQNlnDRb"
|
||||
LOCAL_S3_PLANB_BUCKET = "PLANB"
|
||||
|
||||
# ============================= 本地局域网 S3 配置 =============================
|
||||
LOCAL_S3_PLANC_ENDPOINT = "http://192.168.2.15:5246"
|
||||
LOCAL_S3_PLANC_ACCESS_KEY = "vS8Lf7UoiS5cxL+mz+CH"
|
||||
LOCAL_S3_PLANC_SECRET_KEY = "Fi3phPW10CsZWuiQCerK4gxaIIJp41w5SQNlnDRb"
|
||||
LOCAL_S3_PLANC_BUCKET = "PLANC"
|
||||
|
||||
# ============================= 公网 MinIO 配置 =============================
|
||||
PUBLIC_S3_ENDPOINT = "s3.sansenhoshi.top"
|
||||
PUBLIC_S3_ACCESS_KEY = "JDMynACSjPaN8JRwriwS"
|
||||
PUBLIC_S3_SECRET_KEY = "JRl4bIGxeiqwqvfeBTlAWQbUKMpNNHSZPm6ne93j"
|
||||
PUBLIC_S3_BUCKET = "s-file-trans"
|
||||
PUBLIC_S3_REGION = "bot"
|
||||
PUBLIC_S3_SECURE = True
|
||||
PUBLIC_S3_DOMAIN = f"https://{PUBLIC_S3_ENDPOINT}/{PUBLIC_S3_BUCKET}"
|
||||
|
||||
# ============================= 客户端实例(懒加载) =============================
|
||||
_local_s3_plana = None
|
||||
_local_s3_planb = None
|
||||
_local_s3_planc = None
|
||||
_public_s3 = None
|
||||
|
||||
|
||||
class S3Client:
|
||||
"""通用 S3 客户端"""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
endpoint_url: str,
|
||||
access_key: str,
|
||||
secret_key: str,
|
||||
bucket: str,
|
||||
region: str = "us-east-1",
|
||||
secure: bool = False,
|
||||
):
|
||||
self.bucket = bucket
|
||||
self.client = boto3.client(
|
||||
"s3",
|
||||
endpoint_url=endpoint_url,
|
||||
aws_access_key_id=access_key,
|
||||
aws_secret_access_key=secret_key,
|
||||
region_name=region,
|
||||
verify=secure,
|
||||
config=Config(
|
||||
signature_version="s3v4",
|
||||
s3={"addressing_style": "path"},
|
||||
retries={"max_attempts": 3},
|
||||
connect_timeout=10,
|
||||
read_timeout=60,
|
||||
),
|
||||
)
|
||||
|
||||
def upload_and_get_url(
|
||||
self, local_path: str, key: str, expire: int = 3600
|
||||
) -> str:
|
||||
self.client.upload_file(local_path, self.bucket, key)
|
||||
url = self.client.generate_presigned_url(
|
||||
"get_object",
|
||||
Params={"Bucket": self.bucket, "Key": key},
|
||||
ExpiresIn=expire,
|
||||
)
|
||||
return url
|
||||
|
||||
def upload_public(self, local_path: str, key: str) -> str:
|
||||
"""上传并设为公开读"""
|
||||
self.client.upload_file(
|
||||
local_path,
|
||||
self.bucket,
|
||||
key,
|
||||
ExtraArgs={"ACL": "public-read"},
|
||||
)
|
||||
logger.info(f"文件 {key} 已成功上传至云端S3")
|
||||
return f"{_ev('PUBLIC_S3_DOMAIN', PUBLIC_S3_DOMAIN)}/{key}"
|
||||
|
||||
def delete(self, key: str) -> bool:
|
||||
try:
|
||||
self.client.delete_object(Bucket=self.bucket, Key=key)
|
||||
logger.info(f"成功删除 S3 文件: {key}")
|
||||
return True
|
||||
except Exception:
|
||||
logger.exception(f"删除 S3 文件失败: {key}")
|
||||
return False
|
||||
|
||||
|
||||
def _get_local_s3_plana() -> S3Client:
|
||||
global _local_s3_plana
|
||||
if _local_s3_plana is None:
|
||||
_local_s3_plana = S3Client(
|
||||
_ev("LOCAL_S3_PLANA_ENDPOINT", LOCAL_S3_PLANA_ENDPOINT),
|
||||
_ev("LOCAL_S3_PLANA_ACCESS_KEY", LOCAL_S3_PLANA_ACCESS_KEY),
|
||||
_ev("LOCAL_S3_PLANA_SECRET_KEY", LOCAL_S3_PLANA_SECRET_KEY),
|
||||
_ev("LOCAL_S3_PLANA_BUCKET", LOCAL_S3_PLANA_BUCKET),
|
||||
)
|
||||
return _local_s3_plana
|
||||
|
||||
|
||||
def _get_local_s3_planb() -> S3Client:
|
||||
global _local_s3_planb
|
||||
if _local_s3_planb is None:
|
||||
_local_s3_planb = S3Client(
|
||||
_ev("LOCAL_S3_PLANB_ENDPOINT", LOCAL_S3_PLANB_ENDPOINT),
|
||||
_ev("LOCAL_S3_PLANB_ACCESS_KEY", LOCAL_S3_PLANB_ACCESS_KEY),
|
||||
_ev("LOCAL_S3_PLANB_SECRET_KEY", LOCAL_S3_PLANB_SECRET_KEY),
|
||||
_ev("LOCAL_S3_PLANB_BUCKET", LOCAL_S3_PLANB_BUCKET),
|
||||
)
|
||||
return _local_s3_planb
|
||||
|
||||
|
||||
def _get_local_s3_planc() -> S3Client:
|
||||
global _local_s3_planc
|
||||
if _local_s3_planc is None:
|
||||
_local_s3_planc = S3Client(
|
||||
_ev("LOCAL_S3_PLANC_ENDPOINT", LOCAL_S3_PLANC_ENDPOINT),
|
||||
_ev("LOCAL_S3_PLANC_ACCESS_KEY", LOCAL_S3_PLANC_ACCESS_KEY),
|
||||
_ev("LOCAL_S3_PLANC_SECRET_KEY", LOCAL_S3_PLANC_SECRET_KEY),
|
||||
_ev("LOCAL_S3_PLANC_BUCKET", LOCAL_S3_PLANC_BUCKET),
|
||||
)
|
||||
return _local_s3_planc
|
||||
|
||||
|
||||
def _get_public_s3() -> S3Client:
|
||||
global _public_s3
|
||||
if _public_s3 is None:
|
||||
secure = _ev("PUBLIC_S3_SECURE", PUBLIC_S3_SECURE)
|
||||
endpoint = _ev("PUBLIC_S3_ENDPOINT", PUBLIC_S3_ENDPOINT)
|
||||
_public_s3 = S3Client(
|
||||
f"{'https' if secure else 'http'}://{endpoint}",
|
||||
_ev("PUBLIC_S3_ACCESS_KEY", PUBLIC_S3_ACCESS_KEY),
|
||||
_ev("PUBLIC_S3_SECRET_KEY", PUBLIC_S3_SECRET_KEY),
|
||||
_ev("PUBLIC_S3_BUCKET", PUBLIC_S3_BUCKET),
|
||||
region=_ev("PUBLIC_S3_REGION", PUBLIC_S3_REGION),
|
||||
secure=secure,
|
||||
)
|
||||
return _public_s3
|
||||
|
||||
|
||||
# ============================= 对外接口 =============================
|
||||
|
||||
def upload_to_public_s3(file_path: str | Path) -> tuple[str, str]:
|
||||
"""
|
||||
上传到公网 MinIO,返回 (公网URL, file_key)
|
||||
|
||||
私聊场景使用,生成可公网访问的链接
|
||||
"""
|
||||
file_path = Path(file_path)
|
||||
current_day = strftime("%Y-%m-%d", localtime())
|
||||
file_key = f"{current_day}/{file_path.name}"
|
||||
try:
|
||||
url = _get_public_s3().upload_public(str(file_path), file_key)
|
||||
return url, file_key
|
||||
except Exception:
|
||||
logger.exception("上传到公网 MinIO 失败")
|
||||
return "", ""
|
||||
|
||||
|
||||
def upload_to_local_s3(
|
||||
title: str, image_post: bool, file_path: str | Path, plan: str | None = None
|
||||
) -> str:
|
||||
"""
|
||||
上传到局域网 S3,返回预签名 URL
|
||||
|
||||
plan="A" → PLANA 桶
|
||||
plan="B" → PLANB 桶
|
||||
None → PLANC 桶(默认)
|
||||
"""
|
||||
file_path = Path(file_path)
|
||||
current_day = strftime("%Y-%m-%d", localtime())
|
||||
if image_post:
|
||||
file_key = f"{current_day}/{title}/{file_path.name}"
|
||||
else:
|
||||
file_key = f"{current_day}/{file_path.name}"
|
||||
|
||||
if plan == "A":
|
||||
client = _get_local_s3_plana()
|
||||
elif plan == "B":
|
||||
client = _get_local_s3_planb()
|
||||
else:
|
||||
client = _get_local_s3_planc()
|
||||
|
||||
try:
|
||||
logger.info(f"文件 {file_key} 已成功上传至本地S3")
|
||||
return client.upload_and_get_url(str(file_path), file_key)
|
||||
except Exception:
|
||||
logger.exception("上传到本地 S3 失败")
|
||||
return ""
|
||||
|
||||
|
||||
def delete_from_public_s3(file_key: str) -> bool:
|
||||
"""从公网 MinIO 删除指定 key 的对象"""
|
||||
return _get_public_s3().delete(file_key)
|
||||
|
||||
|
||||
def upload_with_plan(
|
||||
file_path: str | Path,
|
||||
*,
|
||||
plan: str | None = None,
|
||||
is_private: bool = False,
|
||||
title: str = "",
|
||||
image_post: bool = False,
|
||||
) -> tuple[str, str | None]:
|
||||
"""
|
||||
统一上传入口:按 plan 路由到对应的本地桶,并按需上传公网
|
||||
|
||||
plan="A" → PLANA(仅本地)
|
||||
plan="B" → PLANB + 公网
|
||||
私聊 → PLANB + 公网
|
||||
默认 → PLANC(仅本地)
|
||||
|
||||
Returns:
|
||||
(local_url, public_url_or_none)
|
||||
"""
|
||||
# 本地上传
|
||||
if plan == "A":
|
||||
local_url = upload_to_local_s3(title, image_post, file_path, plan="A")
|
||||
elif plan == "B":
|
||||
local_url = upload_to_local_s3(title, image_post, file_path, plan="B")
|
||||
elif is_private:
|
||||
local_url = upload_to_local_s3(title, image_post, file_path, plan="B")
|
||||
else:
|
||||
local_url = upload_to_local_s3(title, image_post, file_path) # PLANC
|
||||
|
||||
# 公网上传(仅 PLANB 或私聊)
|
||||
public_url = None
|
||||
if plan == "B" or is_private:
|
||||
public_url, _ = upload_to_public_s3(file_path)
|
||||
|
||||
return local_url, public_url
|
||||
@@ -0,0 +1,153 @@
|
||||
import re
|
||||
import unicodedata
|
||||
from pathlib import Path
|
||||
from time import strftime, localtime
|
||||
from typing import List
|
||||
|
||||
|
||||
def get_temp_root(sub: str = "") -> Path:
|
||||
"""插件媒体临时目录:data/temp[/sub] — 下载的媒体先进这里
|
||||
|
||||
发送时优先直接用这里的本地文件,失败才走 S3 链接(见 handlers/sender.py)。
|
||||
"""
|
||||
root = Path(__file__).resolve().parent.parent.parent / "data" / "temp"
|
||||
if sub:
|
||||
root = root / sub
|
||||
root.mkdir(parents=True, exist_ok=True)
|
||||
return root
|
||||
|
||||
|
||||
def slugify(text: str, max_length: int = 80) -> str:
|
||||
"""
|
||||
将字符串转换为 URL slug
|
||||
|
||||
规则:
|
||||
1. 去掉 #tag
|
||||
2. Unicode 归一化
|
||||
3. 转小写
|
||||
4. 空白和分隔符替换为 -
|
||||
5. 移除非法字符
|
||||
6. 合并连续 -
|
||||
7. 裁剪长度
|
||||
"""
|
||||
|
||||
# 去掉 #标签
|
||||
text = re.sub(r"#\S+", "", text)
|
||||
|
||||
# Unicode 标准化
|
||||
text = unicodedata.normalize("NFKC", text)
|
||||
|
||||
# 转小写
|
||||
text = text.lower()
|
||||
|
||||
# 空白字符 -> -
|
||||
text = re.sub(r"\s+", "-", text)
|
||||
|
||||
# 允许:中文、字母、数字、-
|
||||
text = re.sub(r"[^\w\-一-鿿]", "", text)
|
||||
|
||||
# 合并多个 -
|
||||
text = re.sub(r"-{2,}", "-", text)
|
||||
|
||||
# 去掉首尾 -
|
||||
text = text.strip("-")
|
||||
|
||||
# 控制长度
|
||||
if len(text) > max_length:
|
||||
text = text[:max_length].rstrip("-")
|
||||
|
||||
return text
|
||||
|
||||
|
||||
def ensure_unique_path(base_path: Path) -> Path:
|
||||
"""
|
||||
确保路径不冲突:如已存在则追加 _2, _3... 后缀
|
||||
适用于文件和目录
|
||||
"""
|
||||
if not base_path.exists():
|
||||
return base_path
|
||||
|
||||
parent = base_path.parent
|
||||
stem = base_path.stem
|
||||
ext = base_path.suffix # 目录无后缀 -> ""
|
||||
|
||||
counter = 2
|
||||
while True:
|
||||
new_path = parent / f"{stem}_{counter}{ext}"
|
||||
if not new_path.exists():
|
||||
return new_path
|
||||
counter += 1
|
||||
|
||||
|
||||
def clean_filename(filename: str, max_length: int = 120) -> str:
|
||||
"""
|
||||
清理文件名并添加时间前缀
|
||||
支持多扩展名,如 .tar.gz
|
||||
"""
|
||||
|
||||
current_time = strftime("%H-%M-%S", localtime())
|
||||
|
||||
p = Path(filename)
|
||||
|
||||
# 主文件名
|
||||
name = p.stem
|
||||
|
||||
# 完整扩展名 (.tar.gz)
|
||||
ext = "".join(p.suffixes)
|
||||
|
||||
# Unicode 标准化
|
||||
name = unicodedata.normalize("NFKC", name)
|
||||
|
||||
# 去掉 #tag
|
||||
name = re.sub(r"#\S+", "", name)
|
||||
|
||||
# 非法字符替换
|
||||
name = re.sub(r'[\\/:*?"<>|]', "_", name)
|
||||
|
||||
# 中英文标点
|
||||
name = re.sub(r"[&'\"。,:?!《》【】|]", "_", name)
|
||||
|
||||
# 空白 -> _
|
||||
name = re.sub(r"\s+", "_", name)
|
||||
|
||||
# 只保留:中文、字母、数字、_
|
||||
name = re.sub(r"[^\w一-鿿_]", "", name)
|
||||
|
||||
# 合并 _
|
||||
name = re.sub(r"_+", "_", name)
|
||||
|
||||
# 去首尾 _
|
||||
name = name.strip("_")
|
||||
|
||||
# 长度控制
|
||||
max_name_length = max_length - len(ext) - len(current_time) - 1
|
||||
if len(name) > max_name_length:
|
||||
name = name[:max_name_length].rstrip("_")
|
||||
|
||||
return f"{current_time}_{name}{ext}"
|
||||
|
||||
|
||||
def parse_netscape_cookies(file_path: str) -> List[dict]:
|
||||
"""解析 Netscape 格式 cookies 文件"""
|
||||
cookies = []
|
||||
with open(file_path, "r", encoding="utf-8") as f:
|
||||
for line in f:
|
||||
line = line.strip()
|
||||
if not line or line.startswith("#"):
|
||||
continue
|
||||
parts = line.split("\t")
|
||||
if len(parts) != 7:
|
||||
continue
|
||||
domain, flag, path, secure, expiry, name, value = parts
|
||||
cookie = {
|
||||
"name": name,
|
||||
"value": value,
|
||||
"domain": domain,
|
||||
"path": path,
|
||||
"secure": secure.upper() == "TRUE",
|
||||
"sameSite": "Lax",
|
||||
}
|
||||
if expiry.isdigit() and int(expiry) > 0:
|
||||
cookie["expires"] = int(expiry)
|
||||
cookies.append(cookie)
|
||||
return cookies
|
||||
Reference in New Issue
Block a user