2026-09-01 13:13:40 +08:00
|
|
|
|
"""抖音视频/图文解析编排层"""
|
|
|
|
|
|
|
|
|
|
|
|
import re
|
|
|
|
|
|
import urllib.parse
|
|
|
|
|
|
from pathlib import Path
|
|
|
|
|
|
from typing import Optional, Union
|
|
|
|
|
|
|
|
|
|
|
|
import httpx
|
|
|
|
|
|
from nonebot import logger
|
|
|
|
|
|
|
2026-09-03 00:44:38 +08:00
|
|
|
|
from ..services.fetchers.douyin_api import fetch_douyin_content
|
|
|
|
|
|
from ..services.fetchers.douyin_ssr import MOBILE_UA, fetch_douyin_note_ssr
|
2026-09-01 13:13:40 +08:00
|
|
|
|
from ..models import DouyinFetchError
|
2026-09-03 00:44:38 +08:00
|
|
|
|
from ..utils import get_data_dir, parse_netscape_cookies
|
2026-09-22 14:23:32 +08:00
|
|
|
|
from ..policy import Policy
|
2026-09-01 13:13:40 +08:00
|
|
|
|
from .sender import PendingMedia, _as_paths
|
|
|
|
|
|
|
|
|
|
|
|
SHORT_LINK_PATTERN = re.compile(r"(v\.douyin\.com/[A-Za-z0-9_\-]+)")
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
async def _resolve_short_link(url: str) -> Optional[str]:
|
|
|
|
|
|
"""短链重定向:httpx 取 Location(最多 3 跳),失败返回 None"""
|
|
|
|
|
|
try:
|
|
|
|
|
|
async with httpx.AsyncClient(
|
|
|
|
|
|
headers={"User-Agent": MOBILE_UA},
|
|
|
|
|
|
follow_redirects=False,
|
|
|
|
|
|
timeout=10,
|
|
|
|
|
|
) as client:
|
|
|
|
|
|
current = url
|
|
|
|
|
|
for _ in range(3):
|
|
|
|
|
|
resp = await client.get(current)
|
|
|
|
|
|
if resp.status_code >= 400:
|
|
|
|
|
|
return None
|
|
|
|
|
|
location = resp.headers.get("Location")
|
|
|
|
|
|
if not location:
|
|
|
|
|
|
final = str(resp.url)
|
|
|
|
|
|
# 200 但仍是短链本身(如 JS 挑战壳页)→ 静默失败,显式记录;
|
|
|
|
|
|
# 图文短链若此处失败将退化到 playwright 旧链路
|
|
|
|
|
|
if "v.douyin.com" in final:
|
|
|
|
|
|
logger.warning(f"短链返回挑战页(未跳转): {final}")
|
|
|
|
|
|
return final
|
|
|
|
|
|
# Location 可能是相对路径,需拼上当前 URL
|
|
|
|
|
|
current = urllib.parse.urljoin(current, location)
|
|
|
|
|
|
return current
|
|
|
|
|
|
except Exception:
|
|
|
|
|
|
logger.warning(f"短链重定向失败: {url}")
|
|
|
|
|
|
return None
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
async def parse_douyin(
|
|
|
|
|
|
url: str,
|
|
|
|
|
|
) -> tuple[Optional[str], Optional[Union[Path, list[Path]]], bool]:
|
|
|
|
|
|
"""解析抖音链接(短链 / 全链接),返回 (title, file_paths, is_image_post)
|
|
|
|
|
|
|
|
|
|
|
|
分派规则(不能依赖 URL 路径:短链重定向后图文也统一变成
|
|
|
|
|
|
iesdouyin.com/share/video/{id} 形态):
|
|
|
|
|
|
- note 形态链接 → SSR 静态解析,失败不回退 playwright
|
|
|
|
|
|
(其图文链路存在 aweme/post 取到作者其他作品的缺陷)
|
|
|
|
|
|
- 其余形态 → 先试 SSR 按内容判定:有 images 即图文 → SSR 秒级下载;
|
|
|
|
|
|
真视频 / SSR 失败 → playwright 拦截(保留全部清晰度能力)
|
|
|
|
|
|
"""
|
|
|
|
|
|
# 优先匹配短链接 v.douyin.com/xxx
|
|
|
|
|
|
short = SHORT_LINK_PATTERN.search(url)
|
|
|
|
|
|
if short:
|
|
|
|
|
|
target_url = "https://" + short.group(1)
|
|
|
|
|
|
# 短链先重定向拿到全链接,确定图文/视频类型后分派
|
|
|
|
|
|
resolved = await _resolve_short_link(target_url)
|
|
|
|
|
|
if resolved:
|
|
|
|
|
|
logger.info(f"短链重定向: {target_url} -> {resolved}")
|
|
|
|
|
|
target_url = resolved
|
|
|
|
|
|
else:
|
2026-09-08 14:25:32 +08:00
|
|
|
|
# 全链接: www.douyin.com/note/xxx、www.douyin.com/video/xxx,
|
|
|
|
|
|
# 以及抖音复制出的 iesdouyin.com/share/video|note/xxx(弱网/无网下 App
|
|
|
|
|
|
# 生成的恢复连接同样会是这个形态)。统一提取出作品 id 后再交给 SSR / playwright。
|
2026-09-01 13:13:40 +08:00
|
|
|
|
m = re.search(
|
2026-09-08 14:25:32 +08:00
|
|
|
|
r"(https?://(?:www\.)?(?:m\.)?(?:ies)?douyin\.com/(?:share/)?(note|video)/(\d+))",
|
|
|
|
|
|
url,
|
2026-09-01 13:13:40 +08:00
|
|
|
|
)
|
|
|
|
|
|
if not m:
|
|
|
|
|
|
return None, None, False
|
2026-09-08 14:25:32 +08:00
|
|
|
|
# 统一规范化为 www.douyin.com/{note|video}/{id},避免 iesdouyin 分享页
|
|
|
|
|
|
# 在无登录态/弱网下只返回壳页,直接落到浏览器可正常打开的作品页。
|
|
|
|
|
|
target_url = f"https://www.douyin.com/{m.group(2)}/{m.group(3)}"
|
2026-09-01 13:13:40 +08:00
|
|
|
|
|
|
|
|
|
|
try:
|
|
|
|
|
|
is_img_post = False
|
2026-09-03 00:44:38 +08:00
|
|
|
|
cookies_path = get_data_dir() / "cookies.txt"
|
2026-09-01 13:13:40 +08:00
|
|
|
|
cookies = parse_netscape_cookies(cookies_path)
|
|
|
|
|
|
# 先试 SSR(秒级、免浏览器),失败一律回退 playwright 链路。
|
|
|
|
|
|
# 2026-08-13 起抖音对 iesdouyin SSR 端点整体降级(登录态也拿不到
|
|
|
|
|
|
# videoInfoRes,只返回 33KB 壳页),note/video 形态统一回退,
|
|
|
|
|
|
# 图文作品的"取到作者其他作品"缺陷由 douyin_api 按 aweme_id 精确匹配修复。
|
|
|
|
|
|
try:
|
|
|
|
|
|
title, file_path = await fetch_douyin_note_ssr(target_url, cookies)
|
|
|
|
|
|
except DouyinFetchError:
|
|
|
|
|
|
logger.info("SSR 解析失败,回退 playwright 链路")
|
|
|
|
|
|
title, file_path = None, None
|
|
|
|
|
|
if file_path is None:
|
|
|
|
|
|
title, file_path = await fetch_douyin_content(
|
|
|
|
|
|
target_url, cookies, 10, False
|
|
|
|
|
|
)
|
|
|
|
|
|
if isinstance(file_path, list):
|
|
|
|
|
|
is_img_post = True
|
|
|
|
|
|
return title, file_path, is_img_post
|
|
|
|
|
|
except Exception:
|
|
|
|
|
|
logger.exception(f"获取直链失败 {target_url}")
|
|
|
|
|
|
return None, None, False
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
async def process_douyin_res(
|
|
|
|
|
|
title: str,
|
|
|
|
|
|
file_paths: Union[Path, list[Path]],
|
|
|
|
|
|
is_private: bool,
|
|
|
|
|
|
image_post: bool,
|
2026-09-22 14:23:32 +08:00
|
|
|
|
policy: Policy | None = None,
|
|
|
|
|
|
platform: str | None = None,
|
2026-09-01 13:13:40 +08:00
|
|
|
|
) -> tuple[Optional[PendingMedia], Optional[str]]:
|
|
|
|
|
|
"""下载已完成 → 打包为待发送媒体(不上传、不发送、不清理)
|
|
|
|
|
|
|
2026-09-22 14:23:32 +08:00
|
|
|
|
多级发送(temp 本地 → S3 链接 → 回退本地)与群文件上传由
|
|
|
|
|
|
send_pending_media 统一按 policy 处理;platform 用于群文件限定平台。
|
2026-09-01 13:13:40 +08:00
|
|
|
|
"""
|
|
|
|
|
|
if not file_paths:
|
|
|
|
|
|
return None, None
|
|
|
|
|
|
return (
|
|
|
|
|
|
PendingMedia(
|
|
|
|
|
|
files=_as_paths(file_paths),
|
|
|
|
|
|
image_post=image_post,
|
|
|
|
|
|
is_private=is_private,
|
2026-09-22 14:23:32 +08:00
|
|
|
|
policy=policy,
|
|
|
|
|
|
platform=platform,
|
2026-09-01 13:13:40 +08:00
|
|
|
|
title=title,
|
|
|
|
|
|
),
|
|
|
|
|
|
None,
|
|
|
|
|
|
)
|