Files
HeXi/hexi/plugins/nonebot_plugin_video_analysis/fetchers/douyin_api.py
T
sansenhoshiandClaude b61d09f09f Add HeXi bot codebase: custom plugins, web frontends, tests
- hexi core: message handling, rate limiting, cooldown, plugin manager
- Custom plugins: BF stats, daily check-in, quotes, persona cards, etc.
- Community plugins vendored under hexi/plugins with local fixes
- Web admin frontends (learning-chat, persona-admin), unified hexi/web
- Tests for rate_limit/cooldown/memes/persona; poetry.lock

Co-Authored-By: Claude <noreply@anthropic.com>
2026-09-01 13:13:40 +08:00

426 lines
15 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""抖音内容抓取 — 浏览器自动化 + API 拦截 + 媒体下载"""
import asyncio
import subprocess
from pathlib import Path
from typing import Dict, List, Optional
import httpx
from nonebot import logger
from playwright.async_api import async_playwright
from ..models import DouyinFetchError
from ..utils import ensure_unique_path, get_temp_root
from .douyin_parser import (
ParsedDouyinContent,
extract_trailing_digits,
is_animated_note,
parse_animated_note_videos,
parse_douyin_response,
parse_note_images,
parse_ssr_page,
parse_video_urls,
)
def merge_video_audio(video_file: Path, audio_file: Path, out_file: Path):
"""分轨视频与音频合并"""
subprocess.run(
[
"ffmpeg",
"-i",
str(video_file),
"-i",
str(audio_file),
"-c:v",
"copy",
"-c:a",
"copy",
"-movflags",
"faststart",
"-y",
str(out_file),
],
check=True,
)
# ============================= 主入口 =============================
async def fetch_douyin_content(
douyin_url: str,
cookies: List[dict],
wait_seconds: int = 10,
headless: bool = True,
) -> tuple[str | None, Path | List[Path] | None]:
"""
获取抖音内容(视频或图文)
返回:
(title, content)
- 视频: (title, Path)
- 图文: (title, List[Path])
- 失败: (None, None)
"""
if not cookies:
raise DouyinFetchError("cookies 为空")
tmp_root = get_temp_root("douyin")
api_response: Optional[dict] = None
api_response_favorite: Optional[dict] = None
aweme_id = 0
# 目标作品 id(从入口 URL 提取):aweme/post 返回的是作者作品列表,
# 按此 id 精确匹配要解析的作品,避免取到作者的其他作品
target_aweme_id = extract_trailing_digits(douyin_url) or ""
referer_url = None
async with async_playwright() as p:
browser = await p.chromium.launch(
executable_path="C:/Program Files/Google/Chrome/Application/chrome.exe",
headless=headless,
args=[
"--autoplay-policy=no-user-gesture-required",
"--disable-features=AutoplayDisableSuppression",
],
)
context = await browser.new_context(
viewport={"width": 1280, "height": 720},
device_scale_factor=2,
user_agent=(
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) "
"AppleWebKit/537.36 (KHTML, like Gecko) "
"Chrome/122.0.0.0 Safari/537.36"
),
locale="zh-CN",
)
await context.add_cookies(cookies)
page = await context.new_page()
async def handle_response(response):
nonlocal api_response, api_response_favorite, aweme_id, referer_url
try:
if (
"https://www.douyin.com/note" in response.url
or "https://www.douyin.com/video" in response.url
):
referer_url = response.url
logger.info(f"作品链接:{referer_url}")
aweme_id = extract_trailing_digits(response.url)
logger.info(f"作品id:{aweme_id}")
if "aweme/v1/web/aweme/detail" in response.url:
if not api_response:
logger.info(f"捕获到视频 API: {response.url}")
ct = response.headers.get("content-type", "")
if "json" in ct:
api_response = await response.json()
logger.info("API 响应已捕获")
return
else:
logger.info("API 已有捕获")
return
elif "aweme/v1/web/aweme/post" in response.url:
logger.info(f"捕获到图文 API: {response.url}")
ct = response.headers.get("content-type", "")
if "json" in ct:
data = await response.json()
aweme_list = data.get("aweme_list", [])
if aweme_list:
# 列表是作者的全部作品,优先按目标作品 id 精确匹配
# (未登录/风控时目标作品可能不在列表里)
match_id = aweme_id or target_aweme_id
target_aweme = None
if match_id:
target_aweme = next(
(
a
for a in aweme_list
if str(a.get("aweme_id")) == str(match_id)
),
None,
)
if not target_aweme:
target_aweme = aweme_list[0]
if target_aweme.get("images"):
logger.info("找到正确的图文 API 响应(包含图片数据)")
data["aweme_list"] = [target_aweme]
api_response = data
return
else:
logger.warning(
"此响应不包含有效的图片数据,等待下一个请求"
)
except Exception as e:
logger.warning(f"处理响应失败: {e}")
page.on("response", handle_response)
await page.goto(douyin_url, wait_until="domcontentloaded")
max_wait = wait_seconds
for i in range(max_wait):
if api_response:
logger.info(f"成功在第 {i + 1} 秒捕获 API 响应")
break
await page.wait_for_timeout(1000)
# 截取页面 HTML(在关闭浏览器前),用于 SSR 回退解析
page_html = await page.content() if not api_response else None
await browser.close()
if not api_response and page_html:
api_response = parse_ssr_page(page_html)
if api_response:
logger.info("通过 SSR 页面回退解析获取到图文数据")
if not api_response:
raise DouyinFetchError("无法捕获 API 响应,请检查网络或 URL")
parsed = parse_douyin_response(api_response, referer_url)
headers = {
"Referer": "https://www.douyin.com/",
"User-Agent": (
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) "
"AppleWebKit/537.36 (KHTML, like Gecko) "
"Chrome/122.0.0.0 Safari/537.36"
),
}
if parsed.media_type == "视频":
content = await _process_video(
api_response, tmp_root, parsed.file_name, aweme_id, headers
)
return parsed.file_name, content
elif parsed.media_type == "图片":
# 先解析 images 列表,区分纯动图和图文/图+视频
images_urls, video_url = parse_note_images(
api_response, api_response_favorite, aweme_id
)
if images_urls:
# 有图片(纯图文 或 图+视频混合作品)
content = await _process_note_with_parsed(
images_urls,
video_url,
tmp_root,
parsed.file_name,
headers,
)
else:
# 纯动图(所有项都是视频)
content = await _process_animated_note(
api_response, tmp_root, parsed.file_name, headers
)
return parsed.file_name, content
return None, None
# ============================= 视频下载 =============================
async def _process_video(
api_response: dict,
tmp_root: Path,
file_name: str,
aweme_id: str,
headers: Dict[str, str],
) -> Path:
"""处理视频内容,返回本地文件路径"""
groups = parse_video_urls(api_response)
best_group = None
for g in groups.values():
if g["FULL"] or (g["VIDEO"] and g["AUDIO"]):
best_group = g
break
if not best_group:
raise DouyinFetchError("没有可用的视频组合")
# 完整视频 — 流式下载
if best_group["FULL"]:
full = best_group["FULL"]
logger.info(f"找到完整视频,数量: {len(full)}")
best = max(full, key=lambda x: x["br"])
logger.info(f"选择码率: {best['br']} - {best['url'][:60]}...")
output_path = ensure_unique_path(tmp_root / f"{file_name}.mp4")
async with httpx.AsyncClient(headers=headers) as client:
async with client.stream("GET", best["url"]) as resp:
resp.raise_for_status()
with open(output_path, "wb") as f:
async for chunk in resp.aiter_bytes(8192):
f.write(chunk)
logger.info(f"视频下载完成: {output_path}")
return output_path
# 分轨视频 — 分别下载后合并
logger.info(
f"使用分轨模式,视频数: {len(best_group['VIDEO'])}, "
f"音频数: {len(best_group['AUDIO'])}"
)
video = max(best_group["VIDEO"], key=lambda x: x["br"])
audio = max(best_group["AUDIO"], key=lambda x: x["br"])
logger.info(f"选择视频码率: {video['br']}")
logger.info(f"选择音频码率: {audio['br']}")
video_path = tmp_root / f"{file_name}_v.mp4"
audio_path = tmp_root / f"{file_name}_a.mp4"
output_path = ensure_unique_path(tmp_root / f"{file_name}.mp4")
async with httpx.AsyncClient(headers=headers) as client:
logger.info("开始下载视频...")
async with client.stream("GET", video["url"]) as v:
v.raise_for_status()
with open(video_path, "wb") as f:
async for chunk in v.aiter_bytes(8192):
f.write(chunk)
logger.info("开始下载音频...")
async with client.stream("GET", audio["url"]) as a:
a.raise_for_status()
with open(audio_path, "wb") as f:
async for chunk in a.aiter_bytes(8192):
f.write(chunk)
logger.info("合并视频和音频...")
merge_video_audio(video_path, audio_path, output_path)
video_path.unlink()
audio_path.unlink()
logger.info(f"视频下载完成: {output_path}")
return output_path
# ============================= 图文下载 =============================
async def _process_note_with_parsed(
images_urls: List[List[str]],
video_url: Optional[str],
tmp_root: Path,
file_name: str,
headers: Dict[str, str],
) -> List[Path]:
"""根据已解析的图片/视频 URL 列表,并行下载"""
note_dir = ensure_unique_path(tmp_root / file_name)
note_dir.mkdir(parents=True, exist_ok=True)
logger.info(f"图文保存目录: {note_dir}")
async def _download_one(idx: int, url: str, ext: str = "") -> Optional[Path]:
if not ext:
ext = _infer_extension(url)
filename = f"{idx:03d}{ext}"
filepath = note_dir / filename
try:
async with httpx.AsyncClient(headers=headers) as client:
async with client.stream("GET", url) as resp:
resp.raise_for_status()
with open(filepath, "wb") as f:
async for chunk in resp.aiter_bytes(8192):
f.write(chunk)
logger.info(f"已保存: {filename}")
return filepath
except Exception as e:
logger.error(f"下载 {idx} 失败: {e}")
return None
tasks = [
_download_one(i, url_list[0])
for i, url_list in enumerate(images_urls, 1)
]
# 图+视频混合作品:视频追加到下载任务
if video_url:
next_idx = len(images_urls) + 1
tasks.append(_download_one(next_idx, video_url, ext=".mp4"))
results = await asyncio.gather(*tasks)
saved_paths: List[Path] = [p for p in results if p is not None]
logger.info(f"图文下载完成,共 {len(saved_paths)} 个文件")
return saved_paths
async def _process_note(
api_response: dict,
api_response_favorite: dict,
tmp_root: Path,
file_name: str,
aweme_id: str,
headers: Dict[str, str],
) -> List[Path]:
"""处理图文内容,并行下载所有图片;图+视频混合作品同时下载视频"""
images_urls, video_url = parse_note_images(
api_response, api_response_favorite, aweme_id
)
logger.info(f"解析到的图片链接:{images_urls}")
if not images_urls and not video_url:
raise DouyinFetchError("未找到图文链接")
return await _process_note_with_parsed(
images_urls, video_url, tmp_root, file_name, headers
)
# ============================= 动图下载 =============================
async def _process_animated_note(
api_response: dict,
tmp_root: Path,
file_name: str,
headers: Dict[str, str],
) -> List[Path]:
"""处理动图内容(media_type=42),并行下载所有无声 mp4 视频"""
video_urls = parse_animated_note_videos(api_response)
logger.info(f"解析到的动图视频链接: {video_urls}")
note_dir = ensure_unique_path(tmp_root / file_name)
note_dir.mkdir(parents=True, exist_ok=True)
logger.info(f"动图保存目录: {note_dir}")
async def _download_one(idx: int, url: str) -> Optional[Path]:
filename = f"{idx:03d}.mp4"
filepath = note_dir / filename
try:
async with httpx.AsyncClient(headers=headers) as client:
async with client.stream("GET", url) as resp:
resp.raise_for_status()
with open(filepath, "wb") as f:
async for chunk in resp.aiter_bytes(8192):
f.write(chunk)
logger.info(f"动图视频已保存: {filename}")
return filepath
except Exception as e:
logger.error(f"下载动图视频 {idx} 失败: {e}")
return None
tasks = [
_download_one(i, url)
for i, url in enumerate(video_urls, 1)
]
results = await asyncio.gather(*tasks)
saved_paths: List[Path] = [p for p in results if p is not None]
logger.info(f"动图下载完成,共 {len(saved_paths)} 个视频")
return saved_paths
def _infer_extension(url: str) -> str:
"""从 URL 推断文件扩展名"""
extensions = [".webp", ".jpg", ".jpeg", ".png", ".gif", ".avif"]
url_lower = url.lower()
for ext in extensions:
if ext in url_lower:
return ext
return ".jpg"