426 lines
15 KiB
Python
426 lines
15 KiB
Python
"""抖音内容抓取 — 浏览器自动化 + API 拦截 + 媒体下载"""
|
||||
|
|
|
|||
|
|
import asyncio
|
|||
|
|
import subprocess
|
|||
|
|
from pathlib import Path
|
|||
|
|
from typing import Dict, List, Optional
|
|||
|
|
|
|||
|
|
import httpx
|
|||
|
|
from nonebot import logger
|
|||
|
|
from playwright.async_api import async_playwright
|
|||
|
|
|
|||
|
|
from ..models import DouyinFetchError
|
|||
|
|
from ..utils import ensure_unique_path, get_temp_root
|
|||
|
|
from .douyin_parser import (
|
|||
|
|
ParsedDouyinContent,
|
|||
|
|
extract_trailing_digits,
|
|||
|
|
is_animated_note,
|
|||
|
|
parse_animated_note_videos,
|
|||
|
|
parse_douyin_response,
|
|||
|
|
parse_note_images,
|
|||
|
|
parse_ssr_page,
|
|||
|
|
parse_video_urls,
|
|||
|
|
)
|
|||
|
|
|
|||
|
|
|
|||
|
|
def merge_video_audio(video_file: Path, audio_file: Path, out_file: Path):
|
|||
|
|
"""分轨视频与音频合并"""
|
|||
|
|
subprocess.run(
|
|||
|
|
[
|
|||
|
|
"ffmpeg",
|
|||
|
|
"-i",
|
|||
|
|
str(video_file),
|
|||
|
|
"-i",
|
|||
|
|
str(audio_file),
|
|||
|
|
"-c:v",
|
|||
|
|
"copy",
|
|||
|
|
"-c:a",
|
|||
|
|
"copy",
|
|||
|
|
"-movflags",
|
|||
|
|
"faststart",
|
|||
|
|
"-y",
|
|||
|
|
str(out_file),
|
|||
|
|
],
|
|||
|
|
check=True,
|
|||
|
|
)
|
|||
|
|
|
|||
|
|
|
|||
|
|
# ============================= 主入口 =============================
|
|||
|
|
|
|||
|
|
|
|||
|
|
async def fetch_douyin_content(
|
|||
|
|
douyin_url: str,
|
|||
|
|
cookies: List[dict],
|
|||
|
|
wait_seconds: int = 10,
|
|||
|
|
headless: bool = True,
|
|||
|
|
) -> tuple[str | None, Path | List[Path] | None]:
|
|||
|
|
"""
|
|||
|
|
获取抖音内容(视频或图文)
|
|||
|
|
|
|||
|
|
返回:
|
|||
|
|
(title, content)
|
|||
|
|
- 视频: (title, Path)
|
|||
|
|
- 图文: (title, List[Path])
|
|||
|
|
- 失败: (None, None)
|
|||
|
|
"""
|
|||
|
|
if not cookies:
|
|||
|
|
raise DouyinFetchError("cookies 为空")
|
|||
|
|
|
|||
|
|
tmp_root = get_temp_root("douyin")
|
|||
|
|
api_response: Optional[dict] = None
|
|||
|
|
api_response_favorite: Optional[dict] = None
|
|||
|
|
aweme_id = 0
|
|||
|
|
# 目标作品 id(从入口 URL 提取):aweme/post 返回的是作者作品列表,
|
|||
|
|
# 按此 id 精确匹配要解析的作品,避免取到作者的其他作品
|
|||
|
|
target_aweme_id = extract_trailing_digits(douyin_url) or ""
|
|||
|
|
referer_url = None
|
|||
|
|
|
|||
|
|
async with async_playwright() as p:
|
|||
|
|
browser = await p.chromium.launch(
|
|||
|
|
executable_path="C:/Program Files/Google/Chrome/Application/chrome.exe",
|
|||
|
|
headless=headless,
|
|||
|
|
args=[
|
|||
|
|
"--autoplay-policy=no-user-gesture-required",
|
|||
|
|
"--disable-features=AutoplayDisableSuppression",
|
|||
|
|
],
|
|||
|
|
)
|
|||
|
|
|
|||
|
|
context = await browser.new_context(
|
|||
|
|
viewport={"width": 1280, "height": 720},
|
|||
|
|
device_scale_factor=2,
|
|||
|
|
user_agent=(
|
|||
|
|
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) "
|
|||
|
|
"AppleWebKit/537.36 (KHTML, like Gecko) "
|
|||
|
|
"Chrome/122.0.0.0 Safari/537.36"
|
|||
|
|
),
|
|||
|
|
locale="zh-CN",
|
|||
|
|
)
|
|||
|
|
|
|||
|
|
await context.add_cookies(cookies)
|
|||
|
|
page = await context.new_page()
|
|||
|
|
|
|||
|
|
async def handle_response(response):
|
|||
|
|
nonlocal api_response, api_response_favorite, aweme_id, referer_url
|
|||
|
|
try:
|
|||
|
|
if (
|
|||
|
|
"https://www.douyin.com/note" in response.url
|
|||
|
|
or "https://www.douyin.com/video" in response.url
|
|||
|
|
):
|
|||
|
|
referer_url = response.url
|
|||
|
|
logger.info(f"作品链接:{referer_url}")
|
|||
|
|
aweme_id = extract_trailing_digits(response.url)
|
|||
|
|
logger.info(f"作品id:{aweme_id}")
|
|||
|
|
|
|||
|
|
if "aweme/v1/web/aweme/detail" in response.url:
|
|||
|
|
if not api_response:
|
|||
|
|
logger.info(f"捕获到视频 API: {response.url}")
|
|||
|
|
ct = response.headers.get("content-type", "")
|
|||
|
|
if "json" in ct:
|
|||
|
|
api_response = await response.json()
|
|||
|
|
logger.info("API 响应已捕获")
|
|||
|
|
return
|
|||
|
|
else:
|
|||
|
|
logger.info("API 已有捕获")
|
|||
|
|
return
|
|||
|
|
|
|||
|
|
elif "aweme/v1/web/aweme/post" in response.url:
|
|||
|
|
logger.info(f"捕获到图文 API: {response.url}")
|
|||
|
|
ct = response.headers.get("content-type", "")
|
|||
|
|
if "json" in ct:
|
|||
|
|
data = await response.json()
|
|||
|
|
aweme_list = data.get("aweme_list", [])
|
|||
|
|
if aweme_list:
|
|||
|
|
# 列表是作者的全部作品,优先按目标作品 id 精确匹配
|
|||
|
|
# (未登录/风控时目标作品可能不在列表里)
|
|||
|
|
match_id = aweme_id or target_aweme_id
|
|||
|
|
target_aweme = None
|
|||
|
|
if match_id:
|
|||
|
|
target_aweme = next(
|
|||
|
|
(
|
|||
|
|
a
|
|||
|
|
for a in aweme_list
|
|||
|
|
if str(a.get("aweme_id")) == str(match_id)
|
|||
|
|
),
|
|||
|
|
None,
|
|||
|
|
)
|
|||
|
|
if not target_aweme:
|
|||
|
|
target_aweme = aweme_list[0]
|
|||
|
|
if target_aweme.get("images"):
|
|||
|
|
logger.info("找到正确的图文 API 响应(包含图片数据)")
|
|||
|
|
data["aweme_list"] = [target_aweme]
|
|||
|
|
api_response = data
|
|||
|
|
return
|
|||
|
|
else:
|
|||
|
|
logger.warning(
|
|||
|
|
"此响应不包含有效的图片数据,等待下一个请求"
|
|||
|
|
)
|
|||
|
|
except Exception as e:
|
|||
|
|
logger.warning(f"处理响应失败: {e}")
|
|||
|
|
|
|||
|
|
page.on("response", handle_response)
|
|||
|
|
|
|||
|
|
await page.goto(douyin_url, wait_until="domcontentloaded")
|
|||
|
|
|
|||
|
|
max_wait = wait_seconds
|
|||
|
|
for i in range(max_wait):
|
|||
|
|
if api_response:
|
|||
|
|
logger.info(f"成功在第 {i + 1} 秒捕获 API 响应")
|
|||
|
|
break
|
|||
|
|
await page.wait_for_timeout(1000)
|
|||
|
|
|
|||
|
|
# 截取页面 HTML(在关闭浏览器前),用于 SSR 回退解析
|
|||
|
|
page_html = await page.content() if not api_response else None
|
|||
|
|
|
|||
|
|
await browser.close()
|
|||
|
|
|
|||
|
|
if not api_response and page_html:
|
|||
|
|
api_response = parse_ssr_page(page_html)
|
|||
|
|
if api_response:
|
|||
|
|
logger.info("通过 SSR 页面回退解析获取到图文数据")
|
|||
|
|
|
|||
|
|
if not api_response:
|
|||
|
|
raise DouyinFetchError("无法捕获 API 响应,请检查网络或 URL")
|
|||
|
|
|
|||
|
|
parsed = parse_douyin_response(api_response, referer_url)
|
|||
|
|
|
|||
|
|
headers = {
|
|||
|
|
"Referer": "https://www.douyin.com/",
|
|||
|
|
"User-Agent": (
|
|||
|
|
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) "
|
|||
|
|
"AppleWebKit/537.36 (KHTML, like Gecko) "
|
|||
|
|
"Chrome/122.0.0.0 Safari/537.36"
|
|||
|
|
),
|
|||
|
|
}
|
|||
|
|
|
|||
|
|
if parsed.media_type == "视频":
|
|||
|
|
content = await _process_video(
|
|||
|
|
api_response, tmp_root, parsed.file_name, aweme_id, headers
|
|||
|
|
)
|
|||
|
|
return parsed.file_name, content
|
|||
|
|
|
|||
|
|
elif parsed.media_type == "图片":
|
|||
|
|
# 先解析 images 列表,区分纯动图和图文/图+视频
|
|||
|
|
images_urls, video_url = parse_note_images(
|
|||
|
|
api_response, api_response_favorite, aweme_id
|
|||
|
|
)
|
|||
|
|
if images_urls:
|
|||
|
|
# 有图片(纯图文 或 图+视频混合作品)
|
|||
|
|
content = await _process_note_with_parsed(
|
|||
|
|
images_urls,
|
|||
|
|
video_url,
|
|||
|
|
tmp_root,
|
|||
|
|
parsed.file_name,
|
|||
|
|
headers,
|
|||
|
|
)
|
|||
|
|
else:
|
|||
|
|
# 纯动图(所有项都是视频)
|
|||
|
|
content = await _process_animated_note(
|
|||
|
|
api_response, tmp_root, parsed.file_name, headers
|
|||
|
|
)
|
|||
|
|
return parsed.file_name, content
|
|||
|
|
|
|||
|
|
return None, None
|
|||
|
|
|
|||
|
|
|
|||
|
|
# ============================= 视频下载 =============================
|
|||
|
|
|
|||
|
|
|
|||
|
|
async def _process_video(
|
|||
|
|
api_response: dict,
|
|||
|
|
tmp_root: Path,
|
|||
|
|
file_name: str,
|
|||
|
|
aweme_id: str,
|
|||
|
|
headers: Dict[str, str],
|
|||
|
|
) -> Path:
|
|||
|
|
"""处理视频内容,返回本地文件路径"""
|
|||
|
|
groups = parse_video_urls(api_response)
|
|||
|
|
|
|||
|
|
best_group = None
|
|||
|
|
for g in groups.values():
|
|||
|
|
if g["FULL"] or (g["VIDEO"] and g["AUDIO"]):
|
|||
|
|
best_group = g
|
|||
|
|
break
|
|||
|
|
|
|||
|
|
if not best_group:
|
|||
|
|
raise DouyinFetchError("没有可用的视频组合")
|
|||
|
|
|
|||
|
|
# 完整视频 — 流式下载
|
|||
|
|
if best_group["FULL"]:
|
|||
|
|
full = best_group["FULL"]
|
|||
|
|
logger.info(f"找到完整视频,数量: {len(full)}")
|
|||
|
|
best = max(full, key=lambda x: x["br"])
|
|||
|
|
logger.info(f"选择码率: {best['br']} - {best['url'][:60]}...")
|
|||
|
|
|
|||
|
|
output_path = ensure_unique_path(tmp_root / f"{file_name}.mp4")
|
|||
|
|
async with httpx.AsyncClient(headers=headers) as client:
|
|||
|
|
async with client.stream("GET", best["url"]) as resp:
|
|||
|
|
resp.raise_for_status()
|
|||
|
|
with open(output_path, "wb") as f:
|
|||
|
|
async for chunk in resp.aiter_bytes(8192):
|
|||
|
|
f.write(chunk)
|
|||
|
|
logger.info(f"视频下载完成: {output_path}")
|
|||
|
|
return output_path
|
|||
|
|
|
|||
|
|
# 分轨视频 — 分别下载后合并
|
|||
|
|
logger.info(
|
|||
|
|
f"使用分轨模式,视频数: {len(best_group['VIDEO'])}, "
|
|||
|
|
f"音频数: {len(best_group['AUDIO'])}"
|
|||
|
|
)
|
|||
|
|
video = max(best_group["VIDEO"], key=lambda x: x["br"])
|
|||
|
|
audio = max(best_group["AUDIO"], key=lambda x: x["br"])
|
|||
|
|
logger.info(f"选择视频码率: {video['br']}")
|
|||
|
|
logger.info(f"选择音频码率: {audio['br']}")
|
|||
|
|
|
|||
|
|
video_path = tmp_root / f"{file_name}_v.mp4"
|
|||
|
|
audio_path = tmp_root / f"{file_name}_a.mp4"
|
|||
|
|
output_path = ensure_unique_path(tmp_root / f"{file_name}.mp4")
|
|||
|
|
|
|||
|
|
async with httpx.AsyncClient(headers=headers) as client:
|
|||
|
|
logger.info("开始下载视频...")
|
|||
|
|
async with client.stream("GET", video["url"]) as v:
|
|||
|
|
v.raise_for_status()
|
|||
|
|
with open(video_path, "wb") as f:
|
|||
|
|
async for chunk in v.aiter_bytes(8192):
|
|||
|
|
f.write(chunk)
|
|||
|
|
|
|||
|
|
logger.info("开始下载音频...")
|
|||
|
|
async with client.stream("GET", audio["url"]) as a:
|
|||
|
|
a.raise_for_status()
|
|||
|
|
with open(audio_path, "wb") as f:
|
|||
|
|
async for chunk in a.aiter_bytes(8192):
|
|||
|
|
f.write(chunk)
|
|||
|
|
|
|||
|
|
logger.info("合并视频和音频...")
|
|||
|
|
merge_video_audio(video_path, audio_path, output_path)
|
|||
|
|
video_path.unlink()
|
|||
|
|
audio_path.unlink()
|
|||
|
|
|
|||
|
|
logger.info(f"视频下载完成: {output_path}")
|
|||
|
|
return output_path
|
|||
|
|
|
|||
|
|
|
|||
|
|
# ============================= 图文下载 =============================
|
|||
|
|
|
|||
|
|
|
|||
|
|
async def _process_note_with_parsed(
|
|||
|
|
images_urls: List[List[str]],
|
|||
|
|
video_url: Optional[str],
|
|||
|
|
tmp_root: Path,
|
|||
|
|
file_name: str,
|
|||
|
|
headers: Dict[str, str],
|
|||
|
|
) -> List[Path]:
|
|||
|
|
"""根据已解析的图片/视频 URL 列表,并行下载"""
|
|||
|
|
note_dir = ensure_unique_path(tmp_root / file_name)
|
|||
|
|
note_dir.mkdir(parents=True, exist_ok=True)
|
|||
|
|
logger.info(f"图文保存目录: {note_dir}")
|
|||
|
|
|
|||
|
|
async def _download_one(idx: int, url: str, ext: str = "") -> Optional[Path]:
|
|||
|
|
if not ext:
|
|||
|
|
ext = _infer_extension(url)
|
|||
|
|
filename = f"{idx:03d}{ext}"
|
|||
|
|
filepath = note_dir / filename
|
|||
|
|
try:
|
|||
|
|
async with httpx.AsyncClient(headers=headers) as client:
|
|||
|
|
async with client.stream("GET", url) as resp:
|
|||
|
|
resp.raise_for_status()
|
|||
|
|
with open(filepath, "wb") as f:
|
|||
|
|
async for chunk in resp.aiter_bytes(8192):
|
|||
|
|
f.write(chunk)
|
|||
|
|
logger.info(f"已保存: {filename}")
|
|||
|
|
return filepath
|
|||
|
|
except Exception as e:
|
|||
|
|
logger.error(f"下载 {idx} 失败: {e}")
|
|||
|
|
return None
|
|||
|
|
|
|||
|
|
tasks = [
|
|||
|
|
_download_one(i, url_list[0])
|
|||
|
|
for i, url_list in enumerate(images_urls, 1)
|
|||
|
|
]
|
|||
|
|
|
|||
|
|
# 图+视频混合作品:视频追加到下载任务
|
|||
|
|
if video_url:
|
|||
|
|
next_idx = len(images_urls) + 1
|
|||
|
|
tasks.append(_download_one(next_idx, video_url, ext=".mp4"))
|
|||
|
|
|
|||
|
|
results = await asyncio.gather(*tasks)
|
|||
|
|
saved_paths: List[Path] = [p for p in results if p is not None]
|
|||
|
|
|
|||
|
|
logger.info(f"图文下载完成,共 {len(saved_paths)} 个文件")
|
|||
|
|
return saved_paths
|
|||
|
|
|
|||
|
|
|
|||
|
|
async def _process_note(
|
|||
|
|
api_response: dict,
|
|||
|
|
api_response_favorite: dict,
|
|||
|
|
tmp_root: Path,
|
|||
|
|
file_name: str,
|
|||
|
|
aweme_id: str,
|
|||
|
|
headers: Dict[str, str],
|
|||
|
|
) -> List[Path]:
|
|||
|
|
"""处理图文内容,并行下载所有图片;图+视频混合作品同时下载视频"""
|
|||
|
|
images_urls, video_url = parse_note_images(
|
|||
|
|
api_response, api_response_favorite, aweme_id
|
|||
|
|
)
|
|||
|
|
logger.info(f"解析到的图片链接:{images_urls}")
|
|||
|
|
|
|||
|
|
if not images_urls and not video_url:
|
|||
|
|
raise DouyinFetchError("未找到图文链接")
|
|||
|
|
|
|||
|
|
return await _process_note_with_parsed(
|
|||
|
|
images_urls, video_url, tmp_root, file_name, headers
|
|||
|
|
)
|
|||
|
|
|
|||
|
|
|
|||
|
|
# ============================= 动图下载 =============================
|
|||
|
|
|
|||
|
|
|
|||
|
|
async def _process_animated_note(
|
|||
|
|
api_response: dict,
|
|||
|
|
tmp_root: Path,
|
|||
|
|
file_name: str,
|
|||
|
|
headers: Dict[str, str],
|
|||
|
|
) -> List[Path]:
|
|||
|
|
"""处理动图内容(media_type=42),并行下载所有无声 mp4 视频"""
|
|||
|
|
video_urls = parse_animated_note_videos(api_response)
|
|||
|
|
logger.info(f"解析到的动图视频链接: {video_urls}")
|
|||
|
|
|
|||
|
|
note_dir = ensure_unique_path(tmp_root / file_name)
|
|||
|
|
note_dir.mkdir(parents=True, exist_ok=True)
|
|||
|
|
logger.info(f"动图保存目录: {note_dir}")
|
|||
|
|
|
|||
|
|
async def _download_one(idx: int, url: str) -> Optional[Path]:
|
|||
|
|
filename = f"{idx:03d}.mp4"
|
|||
|
|
filepath = note_dir / filename
|
|||
|
|
try:
|
|||
|
|
async with httpx.AsyncClient(headers=headers) as client:
|
|||
|
|
async with client.stream("GET", url) as resp:
|
|||
|
|
resp.raise_for_status()
|
|||
|
|
with open(filepath, "wb") as f:
|
|||
|
|
async for chunk in resp.aiter_bytes(8192):
|
|||
|
|
f.write(chunk)
|
|||
|
|
logger.info(f"动图视频已保存: {filename}")
|
|||
|
|
return filepath
|
|||
|
|
except Exception as e:
|
|||
|
|
logger.error(f"下载动图视频 {idx} 失败: {e}")
|
|||
|
|
return None
|
|||
|
|
|
|||
|
|
tasks = [
|
|||
|
|
_download_one(i, url)
|
|||
|
|
for i, url in enumerate(video_urls, 1)
|
|||
|
|
]
|
|||
|
|
results = await asyncio.gather(*tasks)
|
|||
|
|
saved_paths: List[Path] = [p for p in results if p is not None]
|
|||
|
|
|
|||
|
|
logger.info(f"动图下载完成,共 {len(saved_paths)} 个视频")
|
|||
|
|
return saved_paths
|
|||
|
|
|
|||
|
|
|
|||
|
|
def _infer_extension(url: str) -> str:
|
|||
|
|
"""从 URL 推断文件扩展名"""
|
|||
|
|
extensions = [".webp", ".jpg", ".jpeg", ".png", ".gif", ".avif"]
|
|||
|
|
url_lower = url.lower()
|
|||
|
|
for ext in extensions:
|
|||
|
|
if ext in url_lower:
|
|||
|
|
return ext
|
|||
|
|
return ".jpg"
|