"""抖音视频/图文解析编排层""" import re import urllib.parse from pathlib import Path from typing import Optional, Union import httpx from nonebot import logger from ..services.fetchers.douyin_api import fetch_douyin_content from ..services.fetchers.douyin_ssr import MOBILE_UA, fetch_douyin_note_ssr from ..models import DouyinFetchError from ..utils import get_data_dir, parse_netscape_cookies from .sender import PendingMedia, _as_paths SHORT_LINK_PATTERN = re.compile(r"(v\.douyin\.com/[A-Za-z0-9_\-]+)") async def _resolve_short_link(url: str) -> Optional[str]: """短链重定向:httpx 取 Location(最多 3 跳),失败返回 None""" try: async with httpx.AsyncClient( headers={"User-Agent": MOBILE_UA}, follow_redirects=False, timeout=10, ) as client: current = url for _ in range(3): resp = await client.get(current) if resp.status_code >= 400: return None location = resp.headers.get("Location") if not location: final = str(resp.url) # 200 但仍是短链本身(如 JS 挑战壳页)→ 静默失败,显式记录; # 图文短链若此处失败将退化到 playwright 旧链路 if "v.douyin.com" in final: logger.warning(f"短链返回挑战页(未跳转): {final}") return final # Location 可能是相对路径,需拼上当前 URL current = urllib.parse.urljoin(current, location) return current except Exception: logger.warning(f"短链重定向失败: {url}") return None async def parse_douyin( url: str, ) -> tuple[Optional[str], Optional[Union[Path, list[Path]]], bool]: """解析抖音链接(短链 / 全链接),返回 (title, file_paths, is_image_post) 分派规则(不能依赖 URL 路径:短链重定向后图文也统一变成 iesdouyin.com/share/video/{id} 形态): - note 形态链接 → SSR 静态解析,失败不回退 playwright (其图文链路存在 aweme/post 取到作者其他作品的缺陷) - 其余形态 → 先试 SSR 按内容判定:有 images 即图文 → SSR 秒级下载; 真视频 / SSR 失败 → playwright 拦截(保留全部清晰度能力) """ # 优先匹配短链接 v.douyin.com/xxx short = SHORT_LINK_PATTERN.search(url) if short: target_url = "https://" + short.group(1) # 短链先重定向拿到全链接,确定图文/视频类型后分派 resolved = await _resolve_short_link(target_url) if resolved: logger.info(f"短链重定向: {target_url} -> {resolved}") target_url = resolved else: # 全链接: www.douyin.com/note/xxx 或 www.douyin.com/video/xxx m = re.search( r"(https?://(?:www\.)?douyin\.com/(?:note|video)/\d+)", url ) if not m: return None, None, False target_url = m.group(1) try: is_img_post = False cookies_path = get_data_dir() / "cookies.txt" cookies = parse_netscape_cookies(cookies_path) # 先试 SSR(秒级、免浏览器),失败一律回退 playwright 链路。 # 2026-08-13 起抖音对 iesdouyin SSR 端点整体降级(登录态也拿不到 # videoInfoRes,只返回 33KB 壳页),note/video 形态统一回退, # 图文作品的"取到作者其他作品"缺陷由 douyin_api 按 aweme_id 精确匹配修复。 try: title, file_path = await fetch_douyin_note_ssr(target_url, cookies) except DouyinFetchError: logger.info("SSR 解析失败,回退 playwright 链路") title, file_path = None, None if file_path is None: title, file_path = await fetch_douyin_content( target_url, cookies, 10, False ) if isinstance(file_path, list): is_img_post = True return title, file_path, is_img_post except Exception: logger.exception(f"获取直链失败 {target_url}") return None, None, False async def process_douyin_res( title: str, file_paths: Union[Path, list[Path]], is_private: bool, image_post: bool, plan: str | None = None, ) -> tuple[Optional[PendingMedia], Optional[str]]: """下载已完成 → 打包为待发送媒体(不上传、不发送、不清理) 多级发送(temp 本地 → S3 链接 → 回退本地)由 send_pending_media 统一处理。 """ if not file_paths: return None, None return ( PendingMedia( files=_as_paths(file_paths), image_post=image_post, is_private=is_private, plan=plan, title=title, ), None, )