Files
HeXi/hexi/plugins/nonebot_plugin_picfinder_take/services/image.py
T
sansenhoshiandClaude 9371a28e35 refactor: restructure plugins per MTSS standard, consolidate assets to res/
Per docs/plugin-audit-report.md (plugins normalized to
Trigger(handlers) → Service(services) → Model(repository/models) + utils):

- Split monolithic __init__.py into handlers/services/utils across
  dailywife, deer_pipe, dice, galgame_card, helldivers_tools,
  huoziyinshua, learning_chat, makeaquote, mc_server_status,
  ncm_saying, picfinder_take, picstatus, random_jm_code, regif,
  steam_info, video_analysis, group_tools
- Move static assets under res/: deer_pipe font/img, makeaquote font,
  helldivers img/templates, huoziyinshua HuoZiYinShua
- Add config.py + register_config_items to ncm_saying, random_jm_code,
  group_tools; learning_chat unified config bridge
- Remove deprecated: voice_trans plugin, bf_bot/test.py, dead code in
  dailywife/deer_pipe, empty dirs, debug scripts under helldivers temp
- Disable brash_general_supercredits_tools (stub comment only)
- bot.py: optional stdout/stderr redirect to log file for Web log viewer,
  force ANSI colorize on non-TTY sinks
- Move runtime data (jm_code.json) out of plugin dir into hexi/data
- Docs: plugin-audit-report.md; README reflects removed plugins

Co-Authored-By: Claude <noreply@anthropic.com>
2026-09-03 00:44:38 +08:00

866 lines
31 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
import asyncio
import base64
import os
import re
from io import BytesIO
from random import randint
from traceback import format_exc
import cloudscraper
import httpx
from PIL import Image
from lxml import etree
from nonebot.log import logger
from playwright.async_api import async_playwright
from ..config import (
ASCII2D_COOKIES_FILE,
ASCII_RESULT_NUM,
HOST_CUSTOM,
SAUCENAO_RESULT_NUM,
THUMB_ON,
proxies,
)
def pic2b64(pic) -> str:
"""PIL Image 转 base64:// 字符串"""
buf = BytesIO()
pic.save(buf, format="PNG")
return "base64://" + base64.b64encode(buf.getvalue()).decode()
def _img_cq(base64_str: str) -> str:
"""生成图片 CQ 码字符串(用于嵌入搜索报告文本)"""
return f"[CQ:image,file={base64_str}]"
def _make_client(timeout: float) -> httpx.AsyncClient:
"""构造 httpx 客户端(httpx 0.28 移除了 AsyncClient 的 proxies 参数,转为 mounts)"""
http_proxy = proxies.get("http")
https_proxy = proxies.get("https")
if http_proxy or https_proxy:
mounts = {}
if http_proxy:
mounts["http://"] = httpx.HTTPTransport(proxy=http_proxy)
if https_proxy:
mounts["https://"] = httpx.HTTPTransport(proxy=https_proxy)
return httpx.AsyncClient(mounts=mounts, timeout=timeout)
return httpx.AsyncClient(timeout=timeout)
async def get_pic(address):
async with _make_client(20) as client:
resp = await client.get(address)
return resp.content
def randcolor():
return (randint(0, 255), randint(0, 255), randint(0, 255))
def ats_pic(img):
if img.mode != "RGB":
img = img.convert("RGB")
width = img.size[0] - 1 # 长度
height = img.size[1] - 1 # 宽度
img.putpixel((0, 0), randcolor())
img.putpixel((0, height), randcolor())
img.putpixel((width, 0), randcolor())
img.putpixel((width, height), randcolor())
return img
async def check_screenshot(bot, file, imgurl):
async with _make_client(20) as client:
pichead = await client.head(imgurl)
if pichead.headers["Content-Type"] == "image/gif":
logger.info("gif pic, not likely a screen shot")
return 0
try:
resp = await client.get(imgurl)
image = Image.open(BytesIO(resp.content))
except Exception:
logger.info("download failed")
return 0
cord = image.size[0] / image.size[1]
height = image.size[1]
logger.info(cord)
if cord > 0.565:
logger.info("too short, not likely a screen shot")
return 0
if cord < 0.2:
logger.info("too long, might be long screen shot")
return 2
logger.info("size checked, next ocr")
try:
ocr_result = await bot.call_api(".ocr_image", image=file)
except Exception:
logger.info("ocr failed")
return False
flag = 0
for result in ocr_result["texts"]:
key1 = re.search("[0-9]{1,2}:[0-9]{2}", result["text"]) # 时间
key2 = re.search("移动|联通|电信", result["text"])
key3 = re.search("4G|5G", result["text"])
key4 = re.search("[0-9]{1,2}%", result["text"]) # 电量
key5 = re.search("[0-9]{0,3}[\\\/][0-9]{0,3}", result["text"]) # 页数
if key2 or key3 or key4:
logger.info(str(result))
loc = result["coordinates"][2]["y"]
if int(loc) < (int(height) / 19):
flag = 1
if key1 or key5:
logger.info(str(result))
loc = result["coordinates"][2]["y"]
if int(loc) < (int(height) / 19) or int(loc) > (int(height) * 18 / 20):
flag = 1
if flag:
break
if flag:
return 1
else:
return 0
def sauces_info(sauce):
service_name = ""
info = ""
try:
if sauce["header"]["index_id"] == 0:
service_name = "H-Magazines"
title = sauce["data"]["title"]
part = sauce["data"]["part"]
date = sauce["data"]["date"]
info = f"{title}-{part}/{date}"
# index 1 "h-anime" disabled
elif sauce["header"]["index_id"] == 2:
service_name = "H-Game CG"
company = sauce["data"]["company"]
title = sauce["data"]["title"]
info = f"[{company}] {title}"
# index 3 "ddb-objects" disabled
# index 4 "ddb-samples" disabled
elif sauce["header"]["index_id"] == 5 or sauce["header"]["index_id"] == 6:
service_name = "pixiv"
author_name = sauce["data"]["member_name"]
title = sauce["data"]["title"]
info = f"「{title}」/「{author_name}」"
# index 6 "pixiv historical" with 5
# index 7 "anime" disabled
elif sauce["header"]["index_id"] == 8:
service_name = "nico nico seiga"
author_name = sauce["data"]["member_name"]
title = sauce["data"]["title"]
info = f"「{title}」/「{author_name}」"
elif sauce["header"]["index_id"] == 9:
service_name = "Danbooru"
creator = sauce["data"]["creator"]
material = sauce["data"]["material"]
info = f"[{creator}]({material})"
elif sauce["header"]["index_id"] == 10:
service_name = "drawr Images"
author_name = sauce["data"]["member_name"]
title = sauce["data"]["title"]
info = f"「{title}」/「{author_name}」"
elif sauce["header"]["index_id"] == 11:
service_name = "Nijie Images"
author_name = sauce["data"]["member_name"]
title = sauce["data"]["title"]
info = f"「{title}」/「{author_name}」"
elif sauce["header"]["index_id"] == 12:
service_name = "Yande.re"
creator = sauce["data"]["creator"]
material = sauce["data"]["material"]
info = f"[{creator}]({material})"
# index 13 "animeop" disabled
# index 14 "IMDb" disabled
# index 15 "Shutterstock" disabled
elif sauce["header"]["index_id"] == 16:
service_name = "FAKKU"
creator = sauce["data"]["creator"]
source = sauce["data"]["source"]
info = f"[{creator}]({source})"
# index 17 reserved
elif sauce["header"]["index_id"] == 18 or sauce["header"]["index_id"] == 38:
service_name = "H-Misc (ehentai)"
eng_name = sauce["data"]["eng_name"]
jp_name = sauce["data"]["jp_name"]
info = f"{jp_name}" if jp_name else f"{eng_name}"
elif sauce["header"]["index_id"] == 19:
service_name = "2D-Market"
creator = sauce["data"]["creator"]
source = sauce["data"]["source"]
info = f"[{creator}]({source})"
elif sauce["header"]["index_id"] == 20:
service_name = "MediBang"
member_name = sauce["data"]["member_name"]
title = sauce["data"]["title"]
info = f"「{title}」/「{member_name}」"
elif sauce["header"]["index_id"] == 21:
service_name = "Anime"
title = sauce["data"]["source"]
year = sauce["data"]["year"]
part = sauce["data"]["part"]
est_time = sauce["data"]["est_time"]
time = est_time.split("/")[0]
info = f"《{title}》/{year}\n第{part}集,{time}"
elif sauce["header"]["index_id"] == 22:
service_name = "H-Anime"
title = sauce["data"]["source"]
year = sauce["data"]["year"]
part = sauce["data"]["part"]
est_time = sauce["data"]["est_time"]
time = est_time.split("/")[0]
info = f"《{title}》/{year}\n第{part}集,{time}"
elif sauce["header"]["index_id"] == 23:
service_name = "IMDb-Movies"
title = sauce["data"]["source"]
year = sauce["data"]["year"]
est_time = sauce["data"]["est_time"]
time = est_time.split("/")[0]
info = f"《{title}》/{year},{time}"
elif sauce["header"]["index_id"] == 24:
service_name = "IMDb-Shows"
title = sauce["data"]["source"]
year = sauce["data"]["year"]
part = sauce["data"]["part"]
est_time = sauce["data"]["est_time"]
time = est_time.split("/")[0]
info = f"《{title}》/{year}\n第{part}集,{time}"
elif sauce["header"]["index_id"] == 25:
service_name = "Gelbooru"
creator = sauce["data"]["creator"]
material = sauce["data"]["material"]
info = f"[{creator}]({material})"
elif sauce["header"]["index_id"] == 26:
service_name = "Konachan"
creator = sauce["data"]["creator"]
material = sauce["data"]["material"]
info = f"[{creator}]({material})"
elif sauce["header"]["index_id"] == 27:
service_name = "Sankaku Channel"
creator = sauce["data"]["creator"]
material = sauce["data"]["material"]
info = f"[{creator}]({material})"
elif sauce["header"]["index_id"] == 28:
service_name = "Anime-Pictures.net"
creator = sauce["data"]["creator"]
material = sauce["data"]["material"]
info = f"[{creator}]({material})"
elif sauce["header"]["index_id"] == 29:
service_name = "e621.net"
creator = sauce["data"]["creator"]
material = sauce["data"]["material"]
info = f"[{creator}]({material})"
elif sauce["header"]["index_id"] == 30:
service_name = "Idol Complex"
creator = sauce["data"]["creator"]
material = sauce["data"]["material"]
info = f"[{creator}]({material})"
elif sauce["header"]["index_id"] == 31:
service_name = "bcy.net Illust"
author_name = sauce["data"]["member_name"]
title = sauce["data"]["title"]
info = f"「{title}」/「{author_name}」"
elif sauce["header"]["index_id"] == 32:
service_name = "bcy.net Cosplay"
author_name = sauce["data"]["member_name"]
title = sauce["data"]["title"]
info = f"「{title}」/「{author_name}」"
elif sauce["header"]["index_id"] == 33:
service_name = "PortalGraphics.net"
member_name = sauce["data"]["member_name"]
title = sauce["data"]["title"]
info = f"「{title}」/「{member_name}」"
elif sauce["header"]["index_id"] == 34:
service_name = "deviantArt"
author_name = sauce["data"]["author_name"]
title = sauce["data"]["title"]
info = f"「{title}」/「{author_name}」"
elif sauce["header"]["index_id"] == 35:
service_name = "Pawoo.net"
illust_id = sauce["data"]["pawoo_id"]
author_name = sauce["data"]["pawoo_user_display_name"]
info = f"「{illust_id}」/「{author_name}」"
elif sauce["header"]["index_id"] == 36:
service_name = "Madokami (Manga)"
source = sauce["data"]["source"]
part = sauce["data"]["part"]
info = part if source in part else f"{source}-{part}"
elif sauce["header"]["index_id"] == 37 or sauce["header"]["index_id"] == 371:
service_name = "MangaDex"
artist = sauce["data"]["artist"]
author = sauce["data"]["author"]
source = sauce["data"]["source"]
part = sauce["data"]["part"]
info_a = f"[{artist}]" if artist == author else f"[{artist}({author})]"
info_b = part if source in part else f"{source}-{part}"
info = info_a + info_b
# index 38 "H-Misc (ehentai)" with 18
elif sauce["header"]["index_id"] == 39:
service_name = "Artstation"
author_name = sauce["data"]["author_name"]
title = sauce["data"]["title"]
info = f"「{title}」/「{author_name}」"
elif sauce["header"]["index_id"] == 40:
service_name = "FurAffinity"
author_name = sauce["data"]["author_name"]
title = sauce["data"]["title"]
info = f"「{title}」/「{author_name}」"
elif sauce["header"]["index_id"] == 41:
service_name = "Twitter"
author_name = sauce["data"]["twitter_user_handle"]
time = sauce["data"]["created_at"]
info = f"「{time[0:10]}」/「{author_name}」"
elif sauce["header"]["index_id"] == 42:
service_name = "Furry Network"
author_name = sauce["data"]["author_name"]
title = sauce["data"]["title"]
info = f"「{title}」/「{author_name}」"
elif sauce["header"]["index_id"] == 43:
service_name = "Kemono"
service = sauce["data"]["service_name"]
author_name = sauce["data"]["user_name"]
title = sauce["data"]["title"]
info = f"「{title}」/「({service}){author_name}」"
elif sauce["header"]["index_id"] == 44:
service_name = "Skeb"
creator_name = sauce["data"]["creator_name"]
creator = sauce["data"]["creator"]
info = f"[{creator_name}]({creator})"
else:
index = sauce["header"]["index_id"]
service_name = f"Index #{index}"
info = "no info"
except Exception as e:
index = sauce["header"]["index_id"]
service_name = f"Index #{index}"
info = "no info"
logger.info(format_exc())
return service_name, info
class SauceNAO:
def __init__(
self,
api_key,
output_type=2,
testmode=0,
dbmask=None,
dbmaski=None,
db=999,
numres=3,
shortlimit=20,
longlimit=300,
):
params = dict()
params["api_key"] = api_key
params["output_type"] = output_type
params["testmode"] = testmode
params["dbmask"] = dbmask
params["dbmaski"] = dbmaski
params["db"] = db
params["numres"] = numres
self.params = params
self.host = HOST_CUSTOM["SAUCENAO"] or "https://saucenao.com"
self.header = "————>saucenao<————"
async def get_sauce(self, image_url):
logger.debug(f"Now starting get the SauceNAO data:{image_url}")
# 过滤 None 参数(SauceNAO 文件上传遇 dbmask=None 编码会返回 500)
params = {k: v for k, v in self.params.items() if v is not None}
# 优先本地下载图片后以文件形式提交,避免 SauceNAO 服务器抓取 QQ 图床失败
try:
async with _make_client(20) as client:
resp = await client.get(image_url)
resp.raise_for_status()
image_bytes = resp.content
submit_ok = True
except Exception as e:
logger.error(f"图片本地下载失败({e}), 退回 URL 提交")
submit_ok = False
if submit_ok:
params.pop("url", None)
async with _make_client(30) as client:
response = await client.post(
f"{self.host}/search.php",
params=params,
files={"file": ("search_image.jpg", image_bytes, "image/jpeg")},
)
else:
params["url"] = image_url
async with _make_client(15) as client:
response = await client.get(f"{self.host}/search.php", params=params)
if response.status_code != 200:
raise RuntimeError(
f"SauceNAO 请求失败: HTTP {response.status_code}, 响应: {response.text[:100]!r}"
)
return response.json()
async def get_view(self, sauce) -> str:
sauces = await self.get_sauce(sauce)
repass = ""
simimax = 0
index = 1
for sauce in sauces["results"]:
try:
url = (
sauce["data"]["ext_urls"][0].replace("\\", "").strip()
if "ext_urls" in sauce["data"]
else "no link"
)
similarity = sauce["header"]["similarity"]
if not similarity.replace(".", "").isdigit():
similarity = 0
simimax = float(similarity) if float(similarity) > simimax else simimax
thumbnail_url = sauce["header"]["thumbnail"]
if THUMB_ON:
try:
thumbnail_image = _img_cq(
pic2b64(
ats_pic(
Image.open(BytesIO(await get_pic(thumbnail_url)))
)
)
)
except Exception as e:
logger.info(format_exc())
thumbnail_image = "[预览图下载失败]"
else:
thumbnail_image = ""
service_name, info = sauces_info(sauce)
putline = (
f"搜索结果{index}:\n\n{thumbnail_image}\n\n平台:\n\n{service_name}:\n\n"
f"相似度:{similarity}%\n\n信息:\n\n{info}\n\n相关:\n\n{url}"
)
if repass:
repass = "\n\n\n".join([repass, putline])
else:
repass = putline
except Exception as e:
logger.info(format_exc())
pass
index += 1
return [repass, simimax]
headers = {
"Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,image/webp,image/apng,*/*;q=0.8,application/signed-exchange;v=b3;q=0.9",
"Accept-Encoding": "gzip, deflate",
"Accept-Language": "zh-CN,zh;q=0.9",
"Cache-Control": "max-age=0",
"Connection": "keep-alive",
"Origin": "https://ascii2d.net",
"Referer": "https://ascii2d.net/",
"Sec-Fetch-Dest": "document",
"Sec-Fetch-Mode": "navigate",
"Sec-Fetch-Site": "same-origin",
"Sec-Fetch-User": "?1",
"Upgrade-Insecure-Requests": "1",
"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/80.0.3987.163 Safari/537.36",
}
class ascii2d:
def __init__(self, num=2):
self.num = num
self.host = HOST_CUSTOM["ASCII"] or "https://ascii2d.net"
self.header = "————>ascii2d<————"
self.scraper = cloudscraper.create_scraper()
async def get_search_data(self, url: str, data=None):
if data is not None:
html = data
else:
html = await get_html_ascii2d(url)
all_data = html.xpath('//div[@class="row item-box"]')
info = []
for data in all_data[1 : self.num + 1]:
try:
title = ""
member = ""
if not data.xpath('.//img[@loading="lazy"]/@src'):
continue
thumb_url = data.xpath('.//img[@loading="lazy"]/@src')[0].strip()
thumb_url = f"{self.host}{thumb_url}"
if not data.xpath('.//div[@class="detail-box gray-link"]/h6'):
data2 = (
data.xpath('.//div[@class="external"]')[0]
if data.xpath('.//div[@class="external"]')
else data
)
info_url = (
data2.xpath(".//a/@href")[0].strip()
if data2.xpath(".//a/@rel")
else "no link"
)
tag = "外部登录" if info_url == "no link" else info_url.split("/")[2]
else:
data2 = data.xpath('.//div[@class="detail-box gray-link"]/h6')[0]
info_url = data2.xpath(".//a/@href")[0].strip()
tag = (
data2.xpath("./small/text()") or data2.xpath(".//a/text()")
)[0].strip()
if tag == "pixiv" or tag == "twitter":
title = data2.xpath(".//a//text()")[0]
member = data2.xpath(".//a//text()")[1]
title = f"「{title}」/「{member}」"
elif tag == "外部登录":
title = data2.text.replace("\n", "") if data2.text else ""
else:
title = data2.text.replace("\n", "")
title = f"「{title}」"
info.append([info_url, tag, thumb_url, title])
except Exception as e:
logger.info(format_exc())
logger.error(e)
continue
return info
async def add_repass(self, tag: str, data, thumbnails: dict | None = None):
po = "——{}——".format(tag)
index = 1
for line in data:
if THUMB_ON:
try:
thumb_content = (thumbnails or {}).get(line[2])
if thumb_content is None:
thumb_content = self.scraper.get(
line[2], timeout=20, proxies=proxies
).content
thumbnail_image = _img_cq(
pic2b64(ats_pic(Image.open(BytesIO(thumb_content))))
)
except Exception as e:
logger.info(format_exc())
thumbnail_image = "[预览图下载失败]"
else:
thumbnail_image = ""
putline = (
f"搜索结果{index}:\n\n{thumbnail_image}\n\n平台:\n\n{line[1]}\n\n"
f"信息:\n\n{line[3]}\n\n相关:\n\n{line[0]}"
)
po = "\n\n\n".join([po, putline])
index += 1
return po
async def get_view(self, ascii2d) -> str:
putline1 = ""
putline2 = ""
url_index = f"{self.host}/search/url/{ascii2d}"
logger.debug(f"Now starting get the {url_index}")
try:
html_index = await get_html_ascii2d(url_index)
if html_index is None:
logger.error("ascii2d 页面获取失败(Cloudflare challenge 未通过)")
return [putline1, putline2]
except Exception as e:
logger.info(format_exc())
logger.error(f"ascii2d get html data failed: {e}")
return [putline1, putline2]
neet_div = html_index.xpath(
'//div[@class="detail-link pull-xs-right hidden-sm-down gray-link"]'
)
if neet_div:
a_url_foot = neet_div[0].xpath("./span/a/@href")
url2 = f"{self.host}{a_url_foot[1]}"
color = await self.get_search_data("", data=html_index)
bovw = await self.get_search_data(url2)
# 用系统浏览器会话批量下载缩略图(带 CF cookie,避免被 Cloudflare 拦截)
thumb_urls = [line[2] for line in color + bovw]
thumbnails = await fetch_thumbs_playwright(thumb_urls)
if color:
putline1 = await self.add_repass("色调检索", color, thumbnails)
if bovw:
putline2 = await self.add_repass("特征检索", bovw, thumbnails)
return [putline1, putline2]
async def get_image_data_sauce(image_url: str, api_key: str):
if type(image_url) == list:
image_url = image_url[0]
logger.info("Loading Image Search Container……")
NAO = SauceNAO(api_key, numres=SAUCENAO_RESULT_NUM)
logger.debug("Loading all view……")
repass = ""
simimax = 0
# 网络抖动/免费key限流时重试,退避间隔同时规避 SauceNAO 的 4 秒限流
for attempt in range(3):
try:
result = await NAO.get_view(image_url)
if result:
header = NAO.header
simimax = result[1]
repass = "\n".join([header, result[0]])
break
except Exception as e:
logger.error(f"SauceNAO 第{attempt + 1}次尝试失败: {e}")
if attempt < 2:
await asyncio.sleep(2 * (attempt + 1))
else:
return ["SauceNAO搜索失败……", 0]
return [repass, simimax]
async def get_image_data_ascii(image_url: str):
if type(image_url) == list:
image_url = image_url[0]
logger.info("Loading Image Search Container……")
ii2d = ascii2d(ASCII_RESULT_NUM)
logger.debug("Loading all view……")
repass1 = ""
repass2 = ""
try:
putline = await ii2d.get_view(image_url)
if putline:
header = ii2d.header
if putline[0]:
repass1 = "\n".join([header, putline[0]])
if putline[1]:
repass2 = "\n".join([header, putline[1]])
except Exception as e:
logger.error(format_exc())
return ["ascii2d搜索失败……", ""]
return [repass1, repass2]
# 系统浏览器可执行文件(与获取 cookie 的浏览器同款 TLS 指纹)
CHROME_PATHS = [
"C:/Program Files/Google/Chrome/Application/chrome.exe",
"C:/Program Files (x86)/Google/Chrome/Application/chrome.exe",
"C:/Program Files (x86)/Microsoft/Edge/Application/msedge.exe",
"C:/Program Files/Microsoft/Edge/Application/msedge.exe",
]
# 专用浏览器 profile 目录(cf_clearance 持久化,首次过验证后长期有效)
CHROME_PROFILE_DIR = os.path.join(
os.path.dirname(os.path.dirname(os.path.abspath(__file__))), "data", "chrome_profile"
)
def _parse_netscape_cookies(file_path: str) -> list[dict]:
"""解析 Netscape 格式 cookies 文件(与 nonebot_plugin_video_analysis 同款)"""
cookies = []
with open(file_path, "r", encoding="utf-8") as f:
for line in f:
line = line.strip()
if not line or line.startswith("#"):
continue
parts = line.split("\t")
if len(parts) != 7:
continue
domain, flag, path, secure, expiry, name, value = parts
cookie = {
"name": name,
"value": value,
"domain": domain,
"path": path,
"secure": secure.upper() == "TRUE",
"sameSite": "Lax",
}
if expiry.isdigit() and int(expiry) > 0:
cookie["expires"] = int(expiry)
cookies.append(cookie)
return cookies
def _load_ascii2d_cookies():
"""读取配置的 ascii2d cookie 文件,未配置或不存在返回 None"""
if not ASCII2D_COOKIES_FILE or not os.path.isfile(ASCII2D_COOKIES_FILE):
return None
return _parse_netscape_cookies(ASCII2D_COOKIES_FILE)
def _save_netscape_cookies(cookies: list[dict], file_path: str):
"""把浏览器回传的 cookies 写回 Netscape 格式文件(实现 cf_clearance 自愈)"""
lines = ["# Netscape HTTP Cookie File"]
for c in cookies:
domain = c.get("domain", "")
include_subdomains = "TRUE" if domain.startswith(".") else "FALSE"
path = c.get("path", "/")
secure = "TRUE" if c.get("secure") else "FALSE"
try:
expires = int(c.get("expires") or 0)
except (TypeError, ValueError):
expires = 0
name = c.get("name", "")
value = c.get("value", "")
lines.append(
f"{domain}\t{include_subdomains}\t{path}\t{secure}\t{expires}\t{name}\t{value}"
)
with open(file_path, "w", encoding="utf-8") as f:
f.write("\n".join(lines))
async def get_html_ascii2d(url):
"""获取 ascii2d 页面 HTML:配置了 cookie 文件时注入系统浏览器直过验证"""
return await get_html_content(url, cookies=_load_ascii2d_cookies())
async def fetch_thumbs_playwright(urls: list[str]) -> dict:
"""用系统浏览器会话批量下载缩略图(带 CF cookie,避免 Cloudflare 拦截)"""
if not urls:
return {}
result = {}
async with async_playwright() as p:
launch_kwargs = {
"headless": False,
"args": [
"--window-position=-32000,-32000",
"--disable-blink-features=AutomationControlled",
],
}
for path in CHROME_PATHS:
if os.path.isfile(path):
launch_kwargs["executable_path"] = path
break
context = await p.chromium.launch_persistent_context(
CHROME_PROFILE_DIR, **launch_kwargs
)
cookies = _load_ascii2d_cookies()
if cookies:
await context.add_cookies(cookies)
try:
for u in urls:
try:
resp = await context.request.get(u, timeout=20000)
if resp.status == 200 and "image" in resp.headers.get(
"content-type", ""
):
result[u] = await resp.body()
except Exception as e:
logger.debug(f"缩略图下载失败 {u}: {e}")
finally:
await context.close()
return result
async def get_html_content(url, cookies=None):
async with async_playwright() as p:
# 用系统 Chrome + 固定 profile 目录(cf_clearance 持久化,与浏览器获取 cookie 时同款 TLS 指纹)
# 不指定 UA:使用系统 Chrome 真实 UA,否则 cf_clearance(与 UA 绑定)会失效
# headless 会被 Cloudflare 检测拒绝,用有头模式并将窗口移出屏幕避免打扰
launch_kwargs = {
"headless": False,
"args": [
"--window-position=-32000,-32000",
"--disable-blink-features=AutomationControlled",
],
}
for path in CHROME_PATHS:
if os.path.isfile(path):
launch_kwargs["executable_path"] = path
break
context = await p.chromium.launch_persistent_context(
CHROME_PROFILE_DIR, **launch_kwargs
)
await context.add_init_script(
"Object.defineProperty(navigator, 'webdriver', {get: () => undefined});"
)
if cookies:
await context.add_cookies(cookies)
page = context.pages[0] if context.pages else await context.new_page()
try:
await page.goto(url, wait_until="domcontentloaded")
# 轮询等待 Cloudflare challenge 自动通过(最多约 20 秒)
passed = False
for _ in range(8):
await page.wait_for_timeout(2500)
title = await page.title()
if "Just a moment" not in title:
passed = True
break
html = etree.HTML(await page.content())
# 浏览器可能刷新了 cf_clearance,回写文件实现自愈
if passed and ASCII2D_COOKIES_FILE:
try:
_save_netscape_cookies(
await context.cookies(), ASCII2D_COOKIES_FILE
)
logger.info("ascii2d cookies 已刷新回写")
except Exception as e:
logger.error(f"ascii2d cookies 回写失败: {e}")
logger.info(f"获取到的html内容:{html}")
return html
except Exception as e:
logger.info(f"Error fetching page: {e}")
return None
finally:
await context.close()