- hexi core: message handling, rate limiting, cooldown, plugin manager - Custom plugins: BF stats, daily check-in, quotes, persona cards, etc. - Community plugins vendored under hexi/plugins with local fixes - Web admin frontends (learning-chat, persona-admin), unified hexi/web - Tests for rate_limit/cooldown/memes/persona; poetry.lock Co-Authored-By: Claude <noreply@anthropic.com>
426 lines
15 KiB
Python
426 lines
15 KiB
Python
"""抖音内容抓取 — 浏览器自动化 + API 拦截 + 媒体下载"""
|
||
|
||
import asyncio
|
||
import subprocess
|
||
from pathlib import Path
|
||
from typing import Dict, List, Optional
|
||
|
||
import httpx
|
||
from nonebot import logger
|
||
from playwright.async_api import async_playwright
|
||
|
||
from ..models import DouyinFetchError
|
||
from ..utils import ensure_unique_path, get_temp_root
|
||
from .douyin_parser import (
|
||
ParsedDouyinContent,
|
||
extract_trailing_digits,
|
||
is_animated_note,
|
||
parse_animated_note_videos,
|
||
parse_douyin_response,
|
||
parse_note_images,
|
||
parse_ssr_page,
|
||
parse_video_urls,
|
||
)
|
||
|
||
|
||
def merge_video_audio(video_file: Path, audio_file: Path, out_file: Path):
|
||
"""分轨视频与音频合并"""
|
||
subprocess.run(
|
||
[
|
||
"ffmpeg",
|
||
"-i",
|
||
str(video_file),
|
||
"-i",
|
||
str(audio_file),
|
||
"-c:v",
|
||
"copy",
|
||
"-c:a",
|
||
"copy",
|
||
"-movflags",
|
||
"faststart",
|
||
"-y",
|
||
str(out_file),
|
||
],
|
||
check=True,
|
||
)
|
||
|
||
|
||
# ============================= 主入口 =============================
|
||
|
||
|
||
async def fetch_douyin_content(
|
||
douyin_url: str,
|
||
cookies: List[dict],
|
||
wait_seconds: int = 10,
|
||
headless: bool = True,
|
||
) -> tuple[str | None, Path | List[Path] | None]:
|
||
"""
|
||
获取抖音内容(视频或图文)
|
||
|
||
返回:
|
||
(title, content)
|
||
- 视频: (title, Path)
|
||
- 图文: (title, List[Path])
|
||
- 失败: (None, None)
|
||
"""
|
||
if not cookies:
|
||
raise DouyinFetchError("cookies 为空")
|
||
|
||
tmp_root = get_temp_root("douyin")
|
||
api_response: Optional[dict] = None
|
||
api_response_favorite: Optional[dict] = None
|
||
aweme_id = 0
|
||
# 目标作品 id(从入口 URL 提取):aweme/post 返回的是作者作品列表,
|
||
# 按此 id 精确匹配要解析的作品,避免取到作者的其他作品
|
||
target_aweme_id = extract_trailing_digits(douyin_url) or ""
|
||
referer_url = None
|
||
|
||
async with async_playwright() as p:
|
||
browser = await p.chromium.launch(
|
||
executable_path="C:/Program Files/Google/Chrome/Application/chrome.exe",
|
||
headless=headless,
|
||
args=[
|
||
"--autoplay-policy=no-user-gesture-required",
|
||
"--disable-features=AutoplayDisableSuppression",
|
||
],
|
||
)
|
||
|
||
context = await browser.new_context(
|
||
viewport={"width": 1280, "height": 720},
|
||
device_scale_factor=2,
|
||
user_agent=(
|
||
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) "
|
||
"AppleWebKit/537.36 (KHTML, like Gecko) "
|
||
"Chrome/122.0.0.0 Safari/537.36"
|
||
),
|
||
locale="zh-CN",
|
||
)
|
||
|
||
await context.add_cookies(cookies)
|
||
page = await context.new_page()
|
||
|
||
async def handle_response(response):
|
||
nonlocal api_response, api_response_favorite, aweme_id, referer_url
|
||
try:
|
||
if (
|
||
"https://www.douyin.com/note" in response.url
|
||
or "https://www.douyin.com/video" in response.url
|
||
):
|
||
referer_url = response.url
|
||
logger.info(f"作品链接:{referer_url}")
|
||
aweme_id = extract_trailing_digits(response.url)
|
||
logger.info(f"作品id:{aweme_id}")
|
||
|
||
if "aweme/v1/web/aweme/detail" in response.url:
|
||
if not api_response:
|
||
logger.info(f"捕获到视频 API: {response.url}")
|
||
ct = response.headers.get("content-type", "")
|
||
if "json" in ct:
|
||
api_response = await response.json()
|
||
logger.info("API 响应已捕获")
|
||
return
|
||
else:
|
||
logger.info("API 已有捕获")
|
||
return
|
||
|
||
elif "aweme/v1/web/aweme/post" in response.url:
|
||
logger.info(f"捕获到图文 API: {response.url}")
|
||
ct = response.headers.get("content-type", "")
|
||
if "json" in ct:
|
||
data = await response.json()
|
||
aweme_list = data.get("aweme_list", [])
|
||
if aweme_list:
|
||
# 列表是作者的全部作品,优先按目标作品 id 精确匹配
|
||
# (未登录/风控时目标作品可能不在列表里)
|
||
match_id = aweme_id or target_aweme_id
|
||
target_aweme = None
|
||
if match_id:
|
||
target_aweme = next(
|
||
(
|
||
a
|
||
for a in aweme_list
|
||
if str(a.get("aweme_id")) == str(match_id)
|
||
),
|
||
None,
|
||
)
|
||
if not target_aweme:
|
||
target_aweme = aweme_list[0]
|
||
if target_aweme.get("images"):
|
||
logger.info("找到正确的图文 API 响应(包含图片数据)")
|
||
data["aweme_list"] = [target_aweme]
|
||
api_response = data
|
||
return
|
||
else:
|
||
logger.warning(
|
||
"此响应不包含有效的图片数据,等待下一个请求"
|
||
)
|
||
except Exception as e:
|
||
logger.warning(f"处理响应失败: {e}")
|
||
|
||
page.on("response", handle_response)
|
||
|
||
await page.goto(douyin_url, wait_until="domcontentloaded")
|
||
|
||
max_wait = wait_seconds
|
||
for i in range(max_wait):
|
||
if api_response:
|
||
logger.info(f"成功在第 {i + 1} 秒捕获 API 响应")
|
||
break
|
||
await page.wait_for_timeout(1000)
|
||
|
||
# 截取页面 HTML(在关闭浏览器前),用于 SSR 回退解析
|
||
page_html = await page.content() if not api_response else None
|
||
|
||
await browser.close()
|
||
|
||
if not api_response and page_html:
|
||
api_response = parse_ssr_page(page_html)
|
||
if api_response:
|
||
logger.info("通过 SSR 页面回退解析获取到图文数据")
|
||
|
||
if not api_response:
|
||
raise DouyinFetchError("无法捕获 API 响应,请检查网络或 URL")
|
||
|
||
parsed = parse_douyin_response(api_response, referer_url)
|
||
|
||
headers = {
|
||
"Referer": "https://www.douyin.com/",
|
||
"User-Agent": (
|
||
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) "
|
||
"AppleWebKit/537.36 (KHTML, like Gecko) "
|
||
"Chrome/122.0.0.0 Safari/537.36"
|
||
),
|
||
}
|
||
|
||
if parsed.media_type == "视频":
|
||
content = await _process_video(
|
||
api_response, tmp_root, parsed.file_name, aweme_id, headers
|
||
)
|
||
return parsed.file_name, content
|
||
|
||
elif parsed.media_type == "图片":
|
||
# 先解析 images 列表,区分纯动图和图文/图+视频
|
||
images_urls, video_url = parse_note_images(
|
||
api_response, api_response_favorite, aweme_id
|
||
)
|
||
if images_urls:
|
||
# 有图片(纯图文 或 图+视频混合作品)
|
||
content = await _process_note_with_parsed(
|
||
images_urls,
|
||
video_url,
|
||
tmp_root,
|
||
parsed.file_name,
|
||
headers,
|
||
)
|
||
else:
|
||
# 纯动图(所有项都是视频)
|
||
content = await _process_animated_note(
|
||
api_response, tmp_root, parsed.file_name, headers
|
||
)
|
||
return parsed.file_name, content
|
||
|
||
return None, None
|
||
|
||
|
||
# ============================= 视频下载 =============================
|
||
|
||
|
||
async def _process_video(
|
||
api_response: dict,
|
||
tmp_root: Path,
|
||
file_name: str,
|
||
aweme_id: str,
|
||
headers: Dict[str, str],
|
||
) -> Path:
|
||
"""处理视频内容,返回本地文件路径"""
|
||
groups = parse_video_urls(api_response)
|
||
|
||
best_group = None
|
||
for g in groups.values():
|
||
if g["FULL"] or (g["VIDEO"] and g["AUDIO"]):
|
||
best_group = g
|
||
break
|
||
|
||
if not best_group:
|
||
raise DouyinFetchError("没有可用的视频组合")
|
||
|
||
# 完整视频 — 流式下载
|
||
if best_group["FULL"]:
|
||
full = best_group["FULL"]
|
||
logger.info(f"找到完整视频,数量: {len(full)}")
|
||
best = max(full, key=lambda x: x["br"])
|
||
logger.info(f"选择码率: {best['br']} - {best['url'][:60]}...")
|
||
|
||
output_path = ensure_unique_path(tmp_root / f"{file_name}.mp4")
|
||
async with httpx.AsyncClient(headers=headers) as client:
|
||
async with client.stream("GET", best["url"]) as resp:
|
||
resp.raise_for_status()
|
||
with open(output_path, "wb") as f:
|
||
async for chunk in resp.aiter_bytes(8192):
|
||
f.write(chunk)
|
||
logger.info(f"视频下载完成: {output_path}")
|
||
return output_path
|
||
|
||
# 分轨视频 — 分别下载后合并
|
||
logger.info(
|
||
f"使用分轨模式,视频数: {len(best_group['VIDEO'])}, "
|
||
f"音频数: {len(best_group['AUDIO'])}"
|
||
)
|
||
video = max(best_group["VIDEO"], key=lambda x: x["br"])
|
||
audio = max(best_group["AUDIO"], key=lambda x: x["br"])
|
||
logger.info(f"选择视频码率: {video['br']}")
|
||
logger.info(f"选择音频码率: {audio['br']}")
|
||
|
||
video_path = tmp_root / f"{file_name}_v.mp4"
|
||
audio_path = tmp_root / f"{file_name}_a.mp4"
|
||
output_path = ensure_unique_path(tmp_root / f"{file_name}.mp4")
|
||
|
||
async with httpx.AsyncClient(headers=headers) as client:
|
||
logger.info("开始下载视频...")
|
||
async with client.stream("GET", video["url"]) as v:
|
||
v.raise_for_status()
|
||
with open(video_path, "wb") as f:
|
||
async for chunk in v.aiter_bytes(8192):
|
||
f.write(chunk)
|
||
|
||
logger.info("开始下载音频...")
|
||
async with client.stream("GET", audio["url"]) as a:
|
||
a.raise_for_status()
|
||
with open(audio_path, "wb") as f:
|
||
async for chunk in a.aiter_bytes(8192):
|
||
f.write(chunk)
|
||
|
||
logger.info("合并视频和音频...")
|
||
merge_video_audio(video_path, audio_path, output_path)
|
||
video_path.unlink()
|
||
audio_path.unlink()
|
||
|
||
logger.info(f"视频下载完成: {output_path}")
|
||
return output_path
|
||
|
||
|
||
# ============================= 图文下载 =============================
|
||
|
||
|
||
async def _process_note_with_parsed(
|
||
images_urls: List[List[str]],
|
||
video_url: Optional[str],
|
||
tmp_root: Path,
|
||
file_name: str,
|
||
headers: Dict[str, str],
|
||
) -> List[Path]:
|
||
"""根据已解析的图片/视频 URL 列表,并行下载"""
|
||
note_dir = ensure_unique_path(tmp_root / file_name)
|
||
note_dir.mkdir(parents=True, exist_ok=True)
|
||
logger.info(f"图文保存目录: {note_dir}")
|
||
|
||
async def _download_one(idx: int, url: str, ext: str = "") -> Optional[Path]:
|
||
if not ext:
|
||
ext = _infer_extension(url)
|
||
filename = f"{idx:03d}{ext}"
|
||
filepath = note_dir / filename
|
||
try:
|
||
async with httpx.AsyncClient(headers=headers) as client:
|
||
async with client.stream("GET", url) as resp:
|
||
resp.raise_for_status()
|
||
with open(filepath, "wb") as f:
|
||
async for chunk in resp.aiter_bytes(8192):
|
||
f.write(chunk)
|
||
logger.info(f"已保存: {filename}")
|
||
return filepath
|
||
except Exception as e:
|
||
logger.error(f"下载 {idx} 失败: {e}")
|
||
return None
|
||
|
||
tasks = [
|
||
_download_one(i, url_list[0])
|
||
for i, url_list in enumerate(images_urls, 1)
|
||
]
|
||
|
||
# 图+视频混合作品:视频追加到下载任务
|
||
if video_url:
|
||
next_idx = len(images_urls) + 1
|
||
tasks.append(_download_one(next_idx, video_url, ext=".mp4"))
|
||
|
||
results = await asyncio.gather(*tasks)
|
||
saved_paths: List[Path] = [p for p in results if p is not None]
|
||
|
||
logger.info(f"图文下载完成,共 {len(saved_paths)} 个文件")
|
||
return saved_paths
|
||
|
||
|
||
async def _process_note(
|
||
api_response: dict,
|
||
api_response_favorite: dict,
|
||
tmp_root: Path,
|
||
file_name: str,
|
||
aweme_id: str,
|
||
headers: Dict[str, str],
|
||
) -> List[Path]:
|
||
"""处理图文内容,并行下载所有图片;图+视频混合作品同时下载视频"""
|
||
images_urls, video_url = parse_note_images(
|
||
api_response, api_response_favorite, aweme_id
|
||
)
|
||
logger.info(f"解析到的图片链接:{images_urls}")
|
||
|
||
if not images_urls and not video_url:
|
||
raise DouyinFetchError("未找到图文链接")
|
||
|
||
return await _process_note_with_parsed(
|
||
images_urls, video_url, tmp_root, file_name, headers
|
||
)
|
||
|
||
|
||
# ============================= 动图下载 =============================
|
||
|
||
|
||
async def _process_animated_note(
|
||
api_response: dict,
|
||
tmp_root: Path,
|
||
file_name: str,
|
||
headers: Dict[str, str],
|
||
) -> List[Path]:
|
||
"""处理动图内容(media_type=42),并行下载所有无声 mp4 视频"""
|
||
video_urls = parse_animated_note_videos(api_response)
|
||
logger.info(f"解析到的动图视频链接: {video_urls}")
|
||
|
||
note_dir = ensure_unique_path(tmp_root / file_name)
|
||
note_dir.mkdir(parents=True, exist_ok=True)
|
||
logger.info(f"动图保存目录: {note_dir}")
|
||
|
||
async def _download_one(idx: int, url: str) -> Optional[Path]:
|
||
filename = f"{idx:03d}.mp4"
|
||
filepath = note_dir / filename
|
||
try:
|
||
async with httpx.AsyncClient(headers=headers) as client:
|
||
async with client.stream("GET", url) as resp:
|
||
resp.raise_for_status()
|
||
with open(filepath, "wb") as f:
|
||
async for chunk in resp.aiter_bytes(8192):
|
||
f.write(chunk)
|
||
logger.info(f"动图视频已保存: {filename}")
|
||
return filepath
|
||
except Exception as e:
|
||
logger.error(f"下载动图视频 {idx} 失败: {e}")
|
||
return None
|
||
|
||
tasks = [
|
||
_download_one(i, url)
|
||
for i, url in enumerate(video_urls, 1)
|
||
]
|
||
results = await asyncio.gather(*tasks)
|
||
saved_paths: List[Path] = [p for p in results if p is not None]
|
||
|
||
logger.info(f"动图下载完成,共 {len(saved_paths)} 个视频")
|
||
return saved_paths
|
||
|
||
|
||
def _infer_extension(url: str) -> str:
|
||
"""从 URL 推断文件扩展名"""
|
||
extensions = [".webp", ".jpg", ".jpeg", ".png", ".gif", ".avif"]
|
||
url_lower = url.lower()
|
||
for ext in extensions:
|
||
if ext in url_lower:
|
||
return ext
|
||
return ".jpg"
|