Files
HeXi/hexi/plugins/nonebot_plugin_video_analysis/handlers/douyin.py
T
sansenhoshi 131b92b319 结构调整
视频解析多图/多媒体结构 消息体适配
2026-09-08 14:25:32 +08:00

132 lines
5.3 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""抖音视频/图文解析编排层"""
import re
import urllib.parse
from pathlib import Path
from typing import Optional, Union
import httpx
from nonebot import logger
from ..services.fetchers.douyin_api import fetch_douyin_content
from ..services.fetchers.douyin_ssr import MOBILE_UA, fetch_douyin_note_ssr
from ..models import DouyinFetchError
from ..utils import get_data_dir, parse_netscape_cookies
from .sender import PendingMedia, _as_paths
SHORT_LINK_PATTERN = re.compile(r"(v\.douyin\.com/[A-Za-z0-9_\-]+)")
async def _resolve_short_link(url: str) -> Optional[str]:
"""短链重定向:httpx 取 Location(最多 3 跳),失败返回 None"""
try:
async with httpx.AsyncClient(
headers={"User-Agent": MOBILE_UA},
follow_redirects=False,
timeout=10,
) as client:
current = url
for _ in range(3):
resp = await client.get(current)
if resp.status_code >= 400:
return None
location = resp.headers.get("Location")
if not location:
final = str(resp.url)
# 200 但仍是短链本身(如 JS 挑战壳页)→ 静默失败,显式记录;
# 图文短链若此处失败将退化到 playwright 旧链路
if "v.douyin.com" in final:
logger.warning(f"短链返回挑战页(未跳转): {final}")
return final
# Location 可能是相对路径,需拼上当前 URL
current = urllib.parse.urljoin(current, location)
return current
except Exception:
logger.warning(f"短链重定向失败: {url}")
return None
async def parse_douyin(
url: str,
) -> tuple[Optional[str], Optional[Union[Path, list[Path]]], bool]:
"""解析抖音链接(短链 / 全链接),返回 (title, file_paths, is_image_post)
分派规则(不能依赖 URL 路径:短链重定向后图文也统一变成
iesdouyin.com/share/video/{id} 形态):
- note 形态链接 → SSR 静态解析,失败不回退 playwright
(其图文链路存在 aweme/post 取到作者其他作品的缺陷)
- 其余形态 → 先试 SSR 按内容判定:有 images 即图文 → SSR 秒级下载;
真视频 / SSR 失败 → playwright 拦截(保留全部清晰度能力)
"""
# 优先匹配短链接 v.douyin.com/xxx
short = SHORT_LINK_PATTERN.search(url)
if short:
target_url = "https://" + short.group(1)
# 短链先重定向拿到全链接,确定图文/视频类型后分派
resolved = await _resolve_short_link(target_url)
if resolved:
logger.info(f"短链重定向: {target_url} -> {resolved}")
target_url = resolved
else:
# 全链接: www.douyin.com/note/xxx、www.douyin.com/video/xxx,
# 以及抖音复制出的 iesdouyin.com/share/video|note/xxx(弱网/无网下 App
# 生成的恢复连接同样会是这个形态)。统一提取出作品 id 后再交给 SSR / playwright。
m = re.search(
r"(https?://(?:www\.)?(?:m\.)?(?:ies)?douyin\.com/(?:share/)?(note|video)/(\d+))",
url,
)
if not m:
return None, None, False
# 统一规范化为 www.douyin.com/{note|video}/{id},避免 iesdouyin 分享页
# 在无登录态/弱网下只返回壳页,直接落到浏览器可正常打开的作品页。
target_url = f"https://www.douyin.com/{m.group(2)}/{m.group(3)}"
try:
is_img_post = False
cookies_path = get_data_dir() / "cookies.txt"
cookies = parse_netscape_cookies(cookies_path)
# 先试 SSR(秒级、免浏览器),失败一律回退 playwright 链路。
# 2026-08-13 起抖音对 iesdouyin SSR 端点整体降级(登录态也拿不到
# videoInfoRes,只返回 33KB 壳页),note/video 形态统一回退,
# 图文作品的"取到作者其他作品"缺陷由 douyin_api 按 aweme_id 精确匹配修复。
try:
title, file_path = await fetch_douyin_note_ssr(target_url, cookies)
except DouyinFetchError:
logger.info("SSR 解析失败,回退 playwright 链路")
title, file_path = None, None
if file_path is None:
title, file_path = await fetch_douyin_content(
target_url, cookies, 10, False
)
if isinstance(file_path, list):
is_img_post = True
return title, file_path, is_img_post
except Exception:
logger.exception(f"获取直链失败 {target_url}")
return None, None, False
async def process_douyin_res(
title: str,
file_paths: Union[Path, list[Path]],
is_private: bool,
image_post: bool,
plan: str | None = None,
) -> tuple[Optional[PendingMedia], Optional[str]]:
"""下载已完成 → 打包为待发送媒体(不上传、不发送、不清理)
多级发送(temp 本地 → S3 链接 → 回退本地)由 send_pending_media 统一处理。
"""
if not file_paths:
return None, None
return (
PendingMedia(
files=_as_paths(file_paths),
image_post=image_post,
is_private=is_private,
plan=plan,
title=title,
),
None,
)