diff --git a/.gitignore b/.gitignore index 3effee7..d5a76c1 100644 --- a/.gitignore +++ b/.gitignore @@ -78,7 +78,10 @@ bg.jpg /.ai/ /.claude/ /hexi/config/ -/CLAUDE.md + +# ---- 开发文档(插件内 CLAUDE.md / DESIGN.md,含根目录,不入库) ---- +**/CLAUDE.md +**/DESIGN.md # helldivers 图标素材&生成脚本(不入库) /dev/docs/HD2/ diff --git a/hexi/core/message_utils.py b/hexi/core/message_utils.py index 324e16e..d9dd539 100644 --- a/hexi/core/message_utils.py +++ b/hexi/core/message_utils.py @@ -1,17 +1,26 @@ +import mimetypes +from pathlib import Path + from nonebot import get_bot from nonebot import require +from nonebot.adapters import Event from nonebot.adapters.onebot.v11 import Bot, Message, MessageEvent, MessageSegment from nonebot.log import logger +from nonebot_plugin_alconna import UniMessage +from nonebot_plugin_alconna.uniseg import Receipt, Target require("nonebot_plugin_htmlrender") from nonebot_plugin_htmlrender import md_to_pic +# 「处理中」占位动图(任务驱动通用回复用) +THINKING_GIF = Path(__file__).parents[1] / "resource" / "imgs" / "thinking.gif" + async def send_markdown( - bot: Bot, - event: MessageEvent, - markdown: str, - fallback_text: str | None = None, + bot: Bot, + event: MessageEvent, + markdown: str, + fallback_text: str | None = None, ) -> None: """发送 Markdown @@ -62,9 +71,9 @@ def get_reply_message(event: MessageEvent) -> Message: def build_forward_nodes( - self_id: int, - entries: list[tuple[str, str]], - sender_name: str | None = None, + self_id: int, + entries: list[tuple[str, str]], + sender_name: str | None = None, ) -> list[dict]: """构造合并转发节点列表 @@ -86,11 +95,11 @@ def build_forward_nodes( async def send_forward_msg( - bot: Bot, - event: MessageEvent, - entries: list[tuple[str, str]], - fallback: str, - sender_name: str | None = None, + bot: Bot, + event: MessageEvent, + entries: list[tuple[str, str]], + fallback: str, + sender_name: str | None = None, ): """群聊发送合并转发消息,私聊或发送失败时回退纯文本 @@ -108,3 +117,65 @@ async def send_forward_msg( except Exception as e: logger.warning(f"合并转发发送失败,回退文本: {e}") await bot.send(event, fallback) + + +async def common_proc_reply( + message_id: str | None = None, + *, + target: Event | Target | None = None, + image: str | Path = THINKING_GIF, + text: str | None = None, +) -> Receipt | None: + """任务驱动通用回复:引用指定消息并回一张「处理中」占位图(UniMessage 版) + + 引用哪条消息 / 发到哪: + - 事件处理中(有上下文):不传参数即引用当前触发消息,发回当前会话; + - 传 message_id 则引用指定消息(仍需事件上下文或用 target 指定会话); + - 后台任务(无事件上下文):必须显式传 target(Event 或 alconna Target), + 如 `Target("872490448")`(群)/`Target("2931589710", private=True)`(私聊); + 此时不传 message_id 即为不带引用的普通发送。 + + Args: + message_id: 要引用的消息 id,默认当前事件消息 + target: 发送目标(Event/Target);仅后台任务需要 + image: 占位图路径(或 http(s) 链接),默认 hexi/resource/imgs/thinking.gif + text: 附加文本,为空只发图 + + Returns: + alconna Receipt: 可 `.recall()` 撤回占位图、`.msg_ids` 取消息 id + (是否支持编辑见 `.editable`);发送失败返回 None + """ + + def build() -> UniMessage: + # 每次重建:alconna 的 reply_to 会把引用段插入消息自身,降级重发需干净副本 + # 本地文件转 base64 内联发送:后端(NapCat)读的是它自己 CWD 下的路径, + # file:/// 形式不可靠;base64:// 由 bot 直接携带字节,OneBot 实现通用支持 + if isinstance(image, str) and image.startswith(("http://", "https://")): + img = UniMessage.image(url=image) + else: + img = UniMessage.image( + raw=Path(image).read_bytes(), + mimetype=mimetypes.guess_type(str(image))[0], + ) + return UniMessage.text(f"{text} ") + img if text else img + + if message_id: + reply_to: str | bool = message_id + elif target is None: + reply_to = True # 从当前事件上下文取被引用消息 + else: + reply_to = False # 后台任务且未指定消息 id → 普通发送 + + try: + return await build().send(target=target, reply_to=reply_to) + except Exception as e: + if not reply_to: + logger.error(f"占位图发送失败: {e}") + return None + logger.warning(f"占位回复引用发送失败,降级为不带引用: {e}") + + try: + return await build().send(target=target) + except Exception as e: + logger.error(f"占位图发送失败: {e}") + return None diff --git a/hexi/plugins/nonebot_plugin_galgame_card/CLAUDE.md b/hexi/plugins/nonebot_plugin_galgame_card/CLAUDE.md deleted file mode 100644 index f657e6d..0000000 --- a/hexi/plugins/nonebot_plugin_galgame_card/CLAUDE.md +++ /dev/null @@ -1,81 +0,0 @@ -# CLAUDE.md — 群聊人设卡插件开发文档 - -本文件是**本插件内开发**的唯一入口文档;需要项目全局信息(Poetry 命令、启动方式、测试约定)时再查项目根目录的 CLAUDE.md。详细设计定稿见同目录 `DESIGN.md`。 - -## 插件概述 - -基于群聊语料蒸馏群成员的形象风格,生成 galgame 风格人物卡(九段画像)。 - -- **目标**:生成"人的画像",不是关系网分析;样貌参考为虚构,永远带标注 -- **隐私硬约束**:只采集 opt-in 成员;敏感数据本地正则替换为占位符,**绝不经过 LLM** -- **维度隔离**:每个 `(group_id, user_id)` 是独立人设,跨群不混 - -## 目录结构 - -``` -nonebot_plugin_galgame_card/ -├── __init__.py 入口层:on_message 采集器(鉴权/采样/调治理落库) -├── models.py 数据层:五张表(persona_group/user/chat_log/impression/summary) -├── repository.py 数据层:仓储(来源无关,所有读写唯一入口) -├── processor.py 治理层:纯函数(五维提取/噪声/脱敏),可单测 -├── config.py 插件配置(仅图片识别开关等;Web 鉴权统一走 /hub) -├── web_hub.py Web API 子应用(挂载到 /api/galgame_card,auth=hexi.web_hub.web_auth) -├── DESIGN.md 设计定稿(数据模型/流水线/九段协议/证据纪律) -└── CLAUDE.md 本文件 -``` - -**Web 管理后台**:已嵌入统一管理台 `/hub/`,前端由 `hexi/web` 渲染,API 挂载到 `/api/galgame_card`,鉴权与 /hub 共用 `hexi.web_hub.web_auth`(OAuth2 + SQLite)。功能:群开关、参与者增删、语料/印象/画像浏览与删除、清空群数据。改后端需重启 bot。 - -## 核心设计(速览,细节见 DESIGN.md) - -1. **两级闸门**:群开关 `persona_group.enabled`(默认关)+ 个人 opt-in `persona_user`,都过才采集 -2. **两级流水线**:语料 →(攒够 N 条)→ 印象(LLM 自然语言,增量中间层)→(攒够 M 条)→ 画像(九段 markdown,版本化) -3. **消息五维**:内容 / 谁发的 / 发给谁(回复/@)/ 几点发的 / 被回复内容快照——只存治理后纯文本 -4. **脱敏四层**(`desensitize()`):明确模式(最长优先排序防截胡)→ 定位式(同条关键词+值)→ 跨条语境(关键词在附近消息)→ 兜底(保守替换) -5. **九段画像协议**:身份印象/性格特征/说话风格/口头禅语录/兴趣话题/相处模式/时间画像/样貌参考(虚构)/不确定信息 -6. **总结路径不进消息 handler**:LLM 调用只在调度器触发,防延迟/限流 - -## 开发命令 - -```bash -# 跑本插件测试(治理层纯函数,9+ 个用例) -poetry run pytest tests/test_persona_processor.py - -# 全量测试 -poetry run pytest - -# 启动/重启验证:PyCharm 的 "start bot" 运行配置(勿用 bat 脚本) -``` - -## 数据库 - -- orm 默认库:`data/nonebot_plugin_orm/db.sqlite3`(不是 `hexi/data/data.db`) -- 建表:bot 启动时 orm 自动 create_all(新表加在 `models.py` 里即可,重启生效) -- 配置键是 `SQLALCHEMY_DATABASE_URL`(本插件未设置,走默认库) - -## 开发注意事项(踩过的坑) - -0. **⚠️ orm 启动自动同步会清空表数据**:`.env` 里 `ALEMBIC_STARTUP_CHECK=false` 时,nonebot_plugin_orm 每次启动都 autogenerate 同步数据库模式,**模型一有变更(改 models.py)就会重建表、清空全部数据**(2026-08-11 实测踩坑,全表被清)。已在本插件 `__init__.py` 导入期把 `migrate.sync` 替换为安全空操作。**今后 schema 演进只准通过 `repository.ensure_schema()` 显式 ALTER**,改完 models.py 后要在重启前手动执行对应 ALTER(或加进 ensure_schema)。 -1. **`on_message` 必须 `block=False`**:否则事件流被拦截,群里其他插件全废 -2. **只收群消息**:handler 参数注解 `GroupMessageEvent`(类型注解即过滤器) -3. **脱敏正则排序**:身份证/银行卡必须在手机号之前,否则手机号截胡身份证数字段 -4. **定位式替换只替换值、保留关键词**:`密码是 xyz789` → `密码是 [密码]`,关键词不能丢 -5. **测试不能裸 import 包**:`__init__.py` 触发 NoneBot 初始化,用 importlib 按路径加载 processor(见 `tests/test_persona_processor.py`,与 test_rate_limit 同款) -6. **总结/印象生成**:绝不在消息 handler 里调 LLM,走调度层(apscheduler) -7. **仓储并发**:`add_summary` 版本自增有并发撞 UNIQUE 风险,调度层加锁保护 - -## 当前进度 - -- ✅ 数据层:五表 + 仓储(含群开关、滚动淘汰、版本自增) -- ✅ 采集层:消息路径(监控→鉴权→治理→脱敏→采样→落库) -- ✅ Web 管理后台:嵌入 /hub(/api/galgame_card + hexi/web 页面;群开关、参与者、数据浏览/清理) -- ⏳ QQ 命令集:`开启人设采集` / `加入人设` / `退出人设` / `查看人设`(Web 已覆盖同等功能,QQ 命令可选做) -- ⏳ 总结路径:LLM 客户端、印象生成、九段画像生成、调度触发 -- ⏳ 呈现层:人物卡展示/图片渲染 - -## 待决策点 - -- 触发阈值(印象 ≥50 条新语料 / 画像 ≥5 条新印象,⏳ 待调) -- 脱敏兜底位数(裸数字 ≥6 位默认替换 `[账号]`,保守优先;误杀多可提到 8 位,动 `BARE_DIGITS_RE`) -- 密保答案场景(中文值正则误杀率高,方案待定) -- 命令名与权限(超管/群主) diff --git a/hexi/plugins/nonebot_plugin_galgame_card/DESIGN.md b/hexi/plugins/nonebot_plugin_galgame_card/DESIGN.md deleted file mode 100644 index c175479..0000000 --- a/hexi/plugins/nonebot_plugin_galgame_card/DESIGN.md +++ /dev/null @@ -1,194 +0,0 @@ -# 群聊人设卡(Galgame 风格人物卡构建器)设计文档 - -> 本文档固化设计决策,作为各层实现的唯一依据。标注 ⏳ 的为草案/待定项。 - -## 1. 定位 - -**目标**:基于群聊发言语料,蒸馏群成员的形象风格,生成 galgame 风格的人物卡(人设)。 - -**非目标**: -- 不做关系网分析("相处模式"只是画像的一个段落,不是独立产品) -- 不做真实身份推断(年龄/职业/住址等现实信息只能进"不确定信息"段) -- 不生成真实样貌(样貌参考为虚构,永远带标注) - -**消费方**:① 展示给人看(人物卡);② 可选:作为扮演 prompt 注入(MaiBot 兼容方向)。 - -## 2. 整体架构 - -功能分层:采集 → 治理 → 存储 → 调度 → 分析 → 呈现。 - -两级流水线(借鉴 MaiBot 印象机制): - -``` -群聊消息 ──五维落库──> 语料 ──(攒够 N 条)──> 印象(LLM 自然语言) ──(攒够 M 条)──> 画像(九段协议) ──> 版本化快照 -``` - -- **语料 → 印象**:每次对"上次印象之后的新语料"生成一段自然语言印象(话题/氛围/互动),存 `persona_impression`。印象是增量中间产物,画像不重读全部原文。 -- **印象 → 画像**:从印象集 + 规则统计(@/回复 互动、活跃时段)生成九段人物卡。 -- 画像每次生成都是新版本(version +1),永久留档可对比。 - -## 3. 数据层(已实现 ✅) - -五张表,前缀 `persona_`: - -### persona_group —— 群采集开关(数据来源总闸门) - -| 字段 | 类型 | 说明 | -|---|---|---| -| group_id | BigInteger PK | 群号 | -| enabled | Boolean 默认 False | 该群是否开启采集(默认关,需显式开启) | -| updated_at | DateTime | 最后变更时间 | - -群关闭 → 该群所有人一律不采集;群开启后,个人还需 opt-in(两级闸门)。 - -### persona_user —— 参与者名单(按群 opt-in) - -| 字段 | 类型 | 说明 | -|---|---|---| -| user_id | BigInteger PK | 参与人 QQ | -| group_id | BigInteger PK | 所在群 | -| joined_at | DateTime | 加入时间 | - -### persona_chat_log —— 采集语料(五维 + 发言段链条) - -| 字段 | 类型 | 说明 | -|---|---|---| -| id | Integer PK 自增 | 全局有序,印象覆盖区间用它表示 | -| user_id | BigInteger | 谁发的 | -| group_id | BigInteger | | -| nickname | String(64) | 群昵称快照 | -| content | Text | 内容(纯文本,已脱敏截断) | -| target_user_id | BigInteger NULL | 发给谁(ev.reply 的 sender / @ 对象;无则 NULL=群聊漫谈) | -| target_inherited | Boolean | target 是否从发言段链条继承(对上一句的解释/补充仍算发给同一对象) | -| follows_id | Integer NULL | 发言段链条:同一说话人的上一条语料 id(间隔 ≤ 5 分钟) | -| reply_to_content | Text NULL | 被回复内容快照(对方不在语料里也能知道他在回应什么) | -| created_at | DateTime | 几点发的 | - -索引:`(user_id, group_id, created_at)`。滚动保留:单用户单群上限 3000 条(⏳ 常量待定)。 - -**发言段链条**(解决"不带 @/回复 的后续补充丢目标"):`@B 借我玩玩` → 下一条 `我的意思是借号不是借人`(无显式目标)继承 target=B 并打 `target_inherited` 标记;分析层窗口组装可沿 `follows_id` 回溯整段发言。 - -### persona_impression —— 印象(两级流水线中间产物) - -| 字段 | 类型 | 说明 | -|---|---|---| -| id | Integer PK 自增 | | -| user_id / group_id | BigInteger | | -| content | Text | LLM 生成的自然语言印象 | -| cover_from_id / cover_to_id | Integer | 覆盖的语料 id 区间(增量依据) | -| model | String(64) | 生成模型 | -| created_at | DateTime | | - -### persona_image —— 图片识别结果缓存(⏳ 多模态预留,未启用) - -| 字段 | 类型 | 说明 | -|---|---|---| -| hash | String(64) PK | 图片 hash(对应 chat_log.image_hashes) | -| description | Text | 多模态识别结果(表情包梗/截图内容) | -| model | String(64) | 识别模型 | -| recognized_at | DateTime | | - -**预留接口**:`vision.py`(BaseImageRecognizer,当前为 Noop 占位)。将来接入多模态 LLM 后:异步后台识别(绝不在消息路径同步调)、同一 hash 只识别一次(缓存复用)、失败不影响采集。识别描述供分析层窗口组装喂给总结 LLM。 - -### persona_summary —— 画像快照(版本化) - -| 字段 | 类型 | 说明 | -|---|---|---| -| id | Integer PK 自增 | | -| user_id / group_id | BigInteger | | -| version | Integer | 每次生成 +1,同人同群唯一 | -| card_text | Text | 九段 markdown 画像原文 | -| corpus_count | Integer | 语料覆盖条数(元信息) | -| impression_count | Integer | 使用的印象条数(元信息) | -| model | String(64) | 生成模型 | -| created_at | DateTime | | - -约束:`UNIQUE(user_id, group_id, version)`(并发写入由调度层锁保护)。 - -## 4. 九段画像协议(已定稿 ✅) - -格式:markdown 固定标题 + 有界 bullet,代码可解析、人可编辑、LLM 可生成、可注入 prompt。 - -``` -# 人物卡 · {主称呼} -语料 {N} 条 · 时间跨度 {start}-{end} · 版本 v{n} · 生成于 {date} - -## 身份印象 ≤4 条 群内可见的:自称方式、群角色(吐槽役)、常用昵称 -## 性格特征 ≤6 条 毒舌但心软 / 重度拖延 / 嘴硬 -## 说话风格 ≤5 条 爱用"草"开头、句尾 wwww、先吐槽再给结论 -## 口头禅语录 ≤6 条 带原文引用:"有一说一,这个图确实带" -## 兴趣话题 ≤5 条 明日方舟(资深);聊工作→抱怨、聊感情→回避 -## 相处模式 ≤4 条 对小B互怼最多,对新人客气(一句话式,非关系网) -## 时间画像 ≤3 条 深夜 22-02 点活跃,白天潜水 -## 样貌参考 ≤3 条 🎨 虚构标注:基于气质的参考描述 -## 不确定信息 ≤3 条 疑是学生(语料出现"上课"),未证实 -``` - -- 段内 bullet 为纯文本,可带原文引用(口头禅语录段必须带原文) -- 样貌参考段**必须**带 🎨 虚构标注与设计依据("基于 XX 气质") -- 现实身份信息只能出现在"不确定信息"段 -- ⏳ 每段具体生成约束(prompt 细则)属分析层,待细化 - -## 5. 证据纪律(已定稿 ✅) - -三层防编造(借鉴 MaiBot): - -1. 印象 prompt 明文约束:"不要添加语料中没有依据的新事实" -2. 规则统计(互动对象、活跃时段)优先于 LLM 分类结果 -3. LLM 分类结果默认降级进"不确定信息"段——**模型说的不算稳定真相** - -指纹缓存(⏳):证据(印象集 + 统计)hash 未变则不重新生成画像。 - -## 6. 采集与治理(✅ 消息路径已实现,⏳ 阈值待调) - -实现位置:`__init__.py`(on_message 入口 + 鉴权 + 采样)+ `processor.py`(纯函数治理,可单测)。 - -**两级闸门**:群开关 `persona_group.enabled`(群级,默认关)+ 个人 opt-in `persona_user`(个人,群内开启才生效)。两条同时满足才采集。 - -- 群开关控制:`开启人设采集 @群` / `关闭人设采集`(⏳ 命令名待定,权限:超管/群主) -- 只采集 `persona_user` 名单内成员(opt-in),退出即停 - -**脱敏(硬约束:本地正则完成,绝不经过 LLM——LLM 只接触脱敏后文本)**: -敏感值**替换为占位符**而非丢弃整条,保留对话语境(如"借号"互动是人格素材,凭证不是)。 -实现:`processor.desensitize()`,四层,覆盖场景:借号/验证码代收/密码口令/兑换码卡密/密保/收入/联系方式变体/位置。 - -| 规则层 | 模式 | 占位符 | -|---|---|---| -| 明确模式(按最长优先排序,防截胡) | 身份证 → 银行卡 → 邮箱 → 手机号(含分隔变体) → IP → wxid → 坐标 → 车牌 | `[身份证]` `[银行卡]` `[邮箱]` `[手机号]` `[IP]` `[微信号]` `[坐标]` `[车牌]` | -| 定位式(同条"关键词+值") | `密码是 xyz789` `账号 abc123` `激活码 ABCDE-1` `验证码是 123456` `VX: xxx` `月薪 25000`(连接词支持 是/为/冒号/空格) | `[密码]` `[账号]` `[兑换码]` `[验证码]` `[微信号]` `[收入]` | -| 跨条语境 | 关键词在附近消息(如"收下验证码"→ 下一条 `123456`)→ 本条值按语境类型替换 | `[验证码]` `[密码]` `[兑换码]` | -| 兜底 | 裸长数字串 ≥6 位 → `[账号]`;字母+数字混合 ≥6 位 → `[密码]`(保守替换,宁误杀不放过) | `[账号]` `[密码]` | - -语境来源:本群最近 3 条已治理文本(复用复读检测的内存窗口)。 -待扩展场景(⏳):密保答案("你妈妈的名字")、QQ 号文本、代充代练语境。 - -- 噪声过滤:纯表情图(无文字)、复读、命令/签到、长链接轰炸 -- 长文截断(约 200 字/条) -- 采样:连续刷屏 5 秒内只记 1 条 - -## 7. 调度与触发(⏳ 草案) - -| 方式 | 条件 | -|---|---| -| 手动 | 管理员 `生成人物卡 @xxx`(强制,无视阈值) | -| 印象 | 新语料 ≥ 50 条(⏳)且距上次印象 ≥ 24h | -| 画像 | 新印象 ≥ 5 条(⏳)或语料显著增长;首次需语料 ≥ 200 条 | -| 防重入 | 同人同群生成中加锁 | - -## 8. 入口层命令集(⏳ 草案) - -`加入人设` `退出人设`(opt-in 控制)、`查看人设 @xxx`(展示九段卡)、`生成人设 @xxx`(管理员强制)。 - -## 9. 借鉴与不借鉴 MaiBot(已定稿 ✅) - -**借鉴**:两级流水线(印象机制)、九段协议格式(段落文本协议)、证据纪律三层、指纹缓存、防串人(证据绑定 user_id)。 - -**不借鉴**:向量库 + BM25 双路召回 + PPR(语料量级 SQL 直查即可)、完整 A_memorix 记忆系统、md5 person_id(QQ 号即 id)。 - -## 10. 开发阶段 - -- **Phase 1**:数据层(四表 + 仓储)✅ 本文档落盘时完成 -- **Phase 2**:治理层(采集过滤 + opt-in 命令) -- **Phase 3**:分析层(印象/画像生成,LLM 客户端) -- **Phase 4**:调度层(触发/锁)+ 呈现层(人物卡展示) -- ⏳ 后续可选:galgame 风格卡面图片渲染(协议文本为渲染源) diff --git a/hexi/plugins/nonebot_plugin_group_daily_analysis/__init__.py b/hexi/plugins/nonebot_plugin_group_daily_analysis/__init__.py index 494be5c..e3f32a2 100644 --- a/hexi/plugins/nonebot_plugin_group_daily_analysis/__init__.py +++ b/hexi/plugins/nonebot_plugin_group_daily_analysis/__init__.py @@ -17,6 +17,8 @@ from nonebot.params import CommandArg from nonebot.plugin import PluginMetadata from nonebot.rule import to_me +from ...core import message_utils + require("nonebot_plugin_alconna") from nonebot_plugin_alconna import UniMessage # noqa: E402 @@ -31,7 +33,6 @@ from .service import get_services, register_bot_adapter # noqa: E402 from .renderer import html_render # noqa: E402 from .templates import list_templates, template_exists # noqa: E402 - # ---------- 指令定义(全部要求 @ 机器人) ---------- analysis_cmd = on_command( @@ -192,7 +193,7 @@ async def _(bot: Bot, event: GroupMessageEvent): group_id = str(event.group_id) days = _parse_days(event) - await UniMessage.text("正在启动分析引擎,正在拉取最近消息...").send() + await message_utils.common_proc_reply(event.message_id) try: svc = get_services() @@ -347,6 +348,7 @@ async def _(bot: Bot, event: GroupMessageEvent): f"用户称号: {'开' if cm.get_user_title_analysis_enabled() else '关'}", f"金句分析: {'开' if cm.get_golden_quote_analysis_enabled() else '关'}", f"聊天质量: {'开' if cm.get_chat_quality_analysis_enabled() else '关'}", + f"本地记录: {'开' if cm.get_use_local_history() else '关'}", "用法:设置分析 [参数] [值],如:设置分析 天数 3", ] await UniMessage.text(chr(10).join(lines)).send() @@ -362,8 +364,12 @@ _SETTING_MAP = { "称号": ("set_user_title_analysis_enabled", str, "用户称号已{}"), "金句": ("set_golden_quote_analysis_enabled", str, "金句分析已{}"), "聊天质量": ("set_chat_quality_analysis_enabled", str, "聊天质量已{}"), + "本地记录": ("set_use_local_history", str, "本地记录已{}"), } +# 走开/关布尔语义的键(值为 _BOOL_WORDS 里的词) +_BOOL_KEYS = {"话题", "称号", "金句", "聊天质量", "本地记录"} + _BOOL_WORDS = {"开": True, "on": True, "true": True, "关": False, "off": False, "false": False} @@ -375,7 +381,9 @@ async def _(bot: Bot, event: GroupMessageEvent, args: tuple = CommandArg()): tokens = _cmd_tokens(args) if len(tokens) < 2: await UniMessage.text( - "用法:设置分析 [参数] [值]" + chr(10) + "参数:天数/窗口/最大消息/最小消息/输出格式/话题/称号/金句/聊天质量" + "用法:设置分析 [参数] [值]" + + chr(10) + + "参数:天数/窗口/最大消息/最小消息/输出格式/话题/称号/金句/聊天质量/本地记录" ).send() return key = tokens[0] @@ -396,7 +404,7 @@ async def _(bot: Bot, event: GroupMessageEvent, args: tuple = CommandArg()): await UniMessage.text(f"{key} 已设为 {v}").send() elif conv is str: v = val.lower() - if key in ("话题", "称号", "金句", "聊天质量"): + if key in _BOOL_KEYS: if v not in _BOOL_WORDS: await UniMessage.text("布尔值请填:开/关 或 on/off").send() return diff --git a/hexi/plugins/nonebot_plugin_group_daily_analysis/adapter.py b/hexi/plugins/nonebot_plugin_group_daily_analysis/adapter.py index c9beef0..6dcf790 100644 --- a/hexi/plugins/nonebot_plugin_group_daily_analysis/adapter.py +++ b/hexi/plugins/nonebot_plugin_group_daily_analysis/adapter.py @@ -3,12 +3,10 @@ from __future__ import annotations import asyncio -import base64 -import time from datetime import datetime, timedelta -from pathlib import Path from typing import Any +from . import history_store from .core.domain.value_objects.unified_group import UnifiedGroup, UnifiedMember from .core.domain.value_objects.unified_message import ( MessageContent, @@ -17,6 +15,29 @@ from .core.domain.value_objects.unified_message import ( ) from .core.utils.logger import logger +# 分页锚点字段链(后端差异实测):NapCat 的 message_seq 参数按 OB11 +# message_id(shortId 短 ID 映射) 解析,传消息内的 message_seq(内核 msgSeq) +# 命中不了映射 → NapCat 抛"消息不存在",翻页断裂;go-cqhttp/LLOneBot 等 +# 则识别 message_seq。优先 message_id,翻页无进度时依次回退。 +ANCHOR_FIELDS = ("message_id", "message_seq", "seq", "real_id") + + +def _fmt_ts(ts: int) -> str: + """时间戳转日志用的可读时间。""" + try: + return datetime.fromtimestamp(ts).strftime("%Y-%m-%d %H:%M") + except (OSError, OverflowError, ValueError): + return str(ts) + + +def _pick_anchor(raw: dict[str, Any], fields: tuple[str, ...], start: int): + """从消息中按字段优先级提取分页锚点值""" + for field in fields[start:]: + val = raw.get(field) + if val is not None: + return val + return None + class OneBotAdapter: """面向 NoneBot OneBot V11 的最小适配器。""" @@ -31,6 +52,8 @@ class OneBotAdapter: self.platform_id = str(self.config.get("platform_id") or "onebot") self.bot_self_ids = [str(x) for x in self.config.get("bot_self_ids", [])] self.filter_bot_messages = bool(self.config.get("filter_bot_messages", True)) + # 优先读本地消息库(learning_chat 落库),取不到再回退接口分页 + self.use_local_history = bool(self.config.get("use_local_history", True)) # —— 消息拉取 —— async def fetch_messages( @@ -52,8 +75,21 @@ class OneBotAdapter: start_ts = int( (datetime.now() - timedelta(days=days)).timestamp() ) + end_ts = int(datetime.now().timestamp()) + + # 本地消息库优先(完整且毫秒级);before_id 无法映射到时间戳,故跳过 + if self.use_local_history and not before_id: + local = await self._fetch_from_local_store( + group_id, start_ts, end_ts, max_count + ) + if local: + return local current_anchor = before_id + anchor_idx = 0 + no_progress_pages = 0 + last_earliest: dict[str, Any] | None = None + while len(all_raw) < max_count: fetch_count = min(chunk_size, max_count - len(all_raw)) params: dict[str, Any] = { @@ -85,13 +121,34 @@ class OneBotAdapter: break messages = result.get("messages", []) if not messages: - break + # 空页可能意味着当前锚点字段不被后端识别(如 NapCat 外的 + # 实现收到 message_id),换下一字段重试一次,仍空则结束。 + if current_anchor is None or last_earliest is None: + break + if no_progress_pages >= 1 or anchor_idx >= len(ANCHOR_FIELDS) - 1: + logger.warning( + "OneBot 分页拉取: 返回空页且锚点字段已耗尽,停止回溯" + ) + break + no_progress_pages += 1 + anchor_idx += 1 + current_anchor = _pick_anchor( + last_earliest, ANCHOR_FIELDS, anchor_idx + ) + if current_anchor is None: + break + logger.warning( + f"OneBot 分页拉取: 空页,锚点字段切换为 {ANCHOR_FIELDS[anchor_idx]}" + ) + continue first = messages[0] last = messages[-1] earliest = first if first.get("time", 0) <= last.get("time", 0) else last + last_earliest = earliest chunk_earliest_ts = earliest.get("time", 0) + prev_len = len(all_raw) for raw in messages: msg_time = raw.get("time", 0) msg_id = str(raw.get("message_id", "")) @@ -100,19 +157,42 @@ class OneBotAdapter: if start_ts <= msg_time <= int(datetime.now().timestamp()): all_raw.append(raw) seen_raw_ids.add(msg_id) + added = len(all_raw) - prev_len - seq_val = ( - earliest.get("message_seq") - or earliest.get("real_id") - or earliest.get("seq") - ) - mid_val = earliest.get("message_id") - new_anchor = seq_val if seq_val is not None else mid_val if chunk_earliest_ts <= start_ts: + logger.info( + f"OneBot 分页拉取: 已到达起始时间,共 {len(all_raw)} 条" + ) + break + + if added == 0: + # 本页没有新增消息:锚点不生效(后端不支持该字段)或数据已取尽。 + # 先切换锚点字段再试一次,仍无新增则结束。 + no_progress_pages += 1 + if no_progress_pages >= 2: + logger.warning( + "OneBot 分页拉取: 连续 2 页无新增,停止回溯" + ) + break + if anchor_idx < len(ANCHOR_FIELDS) - 1: + anchor_idx += 1 + logger.warning( + f"OneBot 分页拉取: 锚点字段切换为 {ANCHOR_FIELDS[anchor_idx]}" + ) + else: + no_progress_pages = 0 + + new_anchor = _pick_anchor(earliest, ANCHOR_FIELDS, anchor_idx) + if new_anchor is None: break if current_anchor and str(new_anchor) == str(current_anchor): + logger.info("OneBot 分页拉取: 锚点未位移,历史已取尽") break current_anchor = new_anchor + logger.info( + f"OneBot 分页拉取进度: {len(all_raw)} 条," + f"锚点({ANCHOR_FIELDS[anchor_idx]}): {new_anchor}" + ) await asyncio.sleep(0.05) unified: list[UnifiedMessage] = [] @@ -131,6 +211,54 @@ class OneBotAdapter: logger.warning(f"OneBot 分页获取消息失败: {e}") return [] + async def _fetch_from_local_store( + self, group_id: str, start_ts: int, end_ts: int, max_count: int + ) -> list[UnifiedMessage]: + """从本地消息库(learning_chat 落库)取群历史;不可用时返回空以回退分页。""" + try: + result = await asyncio.to_thread( + history_store.fetch_group_messages, + group_id, + start_ts, + end_ts, + max_count, + ) + except Exception as e: + logger.warning(f"本地历史读取失败,回退接口分页: {e}") + return [] + + if not result: + logger.info( + f"本地历史无数据({result.error or '窗口内无消息'})," + "回退 OneBot 分页拉取" + ) + return [] + + logger.info( + f"本地历史记录拉取: group={group_id}, source={history_store.MESSAGE_TABLE}, " + f"count={len(result.messages)}, " + f"窗口=[{_fmt_ts(start_ts)}..{_fmt_ts(end_ts)}], " + f"名称覆盖={result.names_resolved}/{len(result.messages)}, " + f"去重={result.duplicates}" + ) + if result.truncated: + window_total = ( + result.window_total if result.window_total is not None else "?" + ) + earliest = result.messages[0]["time"] if result.messages else end_ts + logger.warning( + f"本地历史截断: group={group_id}, max_messages={max_count}, " + f"窗口内共 {window_total} 条, 实际取最近 {len(result.messages)} 条, " + f"最早={_fmt_ts(earliest)}, 窗口起点={_fmt_ts(start_ts)}" + ) + + unified: list[UnifiedMessage] = [] + for raw in result.messages: + converted = self._convert_message(raw, group_id) + if converted: + unified.append(converted) + return unified + def _convert_message(self, raw: dict, group_id: str) -> UnifiedMessage | None: try: sender = raw.get("sender", {}) diff --git a/hexi/plugins/nonebot_plugin_group_daily_analysis/config.py b/hexi/plugins/nonebot_plugin_group_daily_analysis/config.py index 3b327de..a16a987 100644 --- a/hexi/plugins/nonebot_plugin_group_daily_analysis/config.py +++ b/hexi/plugins/nonebot_plugin_group_daily_analysis/config.py @@ -98,6 +98,8 @@ def _default_config() -> dict: cfg.setdefault("basic", {}).setdefault("max_messages", 1000) cfg.setdefault("basic", {}).setdefault("min_messages_threshold", 50) cfg.setdefault("basic", {}).setdefault("filter_bot_messages", True) + # 优先读本地消息库(learning_chat 落库),取不到再回退接口分页 + cfg.setdefault("basic", {}).setdefault("use_local_history", True) cfg.setdefault("analysis_features", {}).setdefault("chat_quality_analysis_enabled", False) cfg.setdefault("incremental", {}).setdefault("incremental_enabled", False) return cfg diff --git a/hexi/plugins/nonebot_plugin_group_daily_analysis/core/infrastructure/config/config_manager.py b/hexi/plugins/nonebot_plugin_group_daily_analysis/core/infrastructure/config/config_manager.py index f3607bd..7c2004e 100644 --- a/hexi/plugins/nonebot_plugin_group_daily_analysis/core/infrastructure/config/config_manager.py +++ b/hexi/plugins/nonebot_plugin_group_daily_analysis/core/infrastructure/config/config_manager.py @@ -770,6 +770,15 @@ class ConfigManager: self._ensure_group("basic")["filter_bot_messages"] = enabled self.config.save_config() + def get_use_local_history(self) -> bool: + """获取是否优先从本地消息库读取群历史。""" + return self._get_group("basic").get("use_local_history", True) + + def set_use_local_history(self, enabled: bool): + """设置是否优先从本地消息库读取群历史(关闭则始终走接口分页)。""" + self._ensure_group("basic")["use_local_history"] = enabled + self.config.save_config() + def get_html_output_dir(self) -> str: """获取HTML输出目录""" diff --git a/hexi/plugins/nonebot_plugin_group_daily_analysis/history_store.py b/hexi/plugins/nonebot_plugin_group_daily_analysis/history_store.py new file mode 100644 index 0000000..9f5fa04 --- /dev/null +++ b/hexi/plugins/nonebot_plugin_group_daily_analysis/history_store.py @@ -0,0 +1,281 @@ +"""本地群历史记录源:只读读取 learning_chat 落库的群消息。 + +`nonebot_plugin_learning_chat` 会把机器人收到的每条群消息写入 +`learning_chat_message`(与该群是否开启学习无关),比 OneBot +`get_group_msg_history` 分页更完整——部分后端只能取回一页就被截断。 +昵称/群名片从同库的 `nonebot_plugin_uninfo_*` 表离线补齐,不额外调接口。 + +模块只依赖标准库(logger 做守卫导入),便于单测用 importlib 裸加载; +读连接一律 `mode=ro`(绝不建库),任何失败都返回带 error 的空结果。 +""" + +from __future__ import annotations + +import json +import os +import sqlite3 +from dataclasses import dataclass, field +from pathlib import Path +from typing import Any + +try: # 插件包内导入:走插件 logger + from .core.utils.logger import logger +except Exception: # 单测裸加载(无包上下文)时降级为标准库 logger + import logging + + logger = logging.getLogger(__name__) + +MESSAGE_TABLE = "learning_chat_message" +# nonebot_plugin_uninfo 的 SceneType.GROUP +SCENE_TYPE_GROUP = 1 +# 显式指定本地消息库位置(测试或非默认部署用) +DB_PATH_ENV = "HEXI_GROUP_DAILY_HISTORY_DB" + +_MESSAGE_SQL = ( + "SELECT id, user_id, message_id, raw_message, message, plain_text, time " + f"FROM {MESSAGE_TABLE} " + "WHERE group_id = ? AND time >= ? AND time <= ? " + "ORDER BY time DESC, id DESC LIMIT ?" +) +_COUNT_SQL = ( + f"SELECT COUNT(*) FROM {MESSAGE_TABLE} " + "WHERE group_id = ? AND time >= ? AND time <= ?" +) +# 同库 uninfo 三表:拿本群成员的 QQ 昵称与群名片 +_NAMES_SQL = ( + "SELECT u.user_id AS user_id, " + "u.user_data AS user_data, " + "s.member_data AS member_data " + "FROM nonebot_plugin_uninfo_scenemodel AS sc " + "JOIN nonebot_plugin_uninfo_sessionmodel AS s " + "ON s.scene_persist_id = sc.id " + "JOIN nonebot_plugin_uninfo_usermodel AS u " + "ON u.id = s.user_persist_id " + "WHERE sc.scene_id = ? AND sc.scene_type = ?" +) + + +@dataclass +class LocalHistoryResult: + """本地历史读取结果;空 messages 表示需要回退到接口分页。""" + + messages: list[dict[str, Any]] = field(default_factory=list) + truncated: bool = False + window_total: int | None = None + names_resolved: int = 0 + duplicates: int = 0 + error: str | None = None + + def __bool__(self) -> bool: + return bool(self.messages) + + +def _parse_sqlite_url(raw: Any) -> Path | None: + """从 SQLAlchemy 数据库配置里解析出 sqlite 文件路径。""" + if raw is None: + return None + text = str(raw).strip() + if not text.startswith("sqlite") or "memory" in text: + return None + # sqlite+aiosqlite:///D:/path/db.sqlite3 或 sqlite:///./rel/db.sqlite3 + _, _, tail = text.partition("///") + tail = tail.strip() + if not tail: + return None + try: + return Path(tail).expanduser() + except (OSError, ValueError): + return None + + +def _driver_db_path() -> Path | None: + """读取 `SQLALCHEMY_DATABASE_URL`(环境变量优先,其次 .env 配置)。""" + for key in ("SQLALCHEMY_DATABASE_URL", "sqlalchemy_database_url"): + raw = os.environ.get(key) + if raw: + return _parse_sqlite_url(raw) + try: + from nonebot import get_driver + + raw = getattr(get_driver().config, "sqlalchemy_database_url", None) + except Exception: + return None + return _parse_sqlite_url(raw) + + +def _localstore_db_path() -> Path | None: + """nonebot_plugin_orm 默认库:localstore 数据目录下的 db.sqlite3。""" + try: + from nonebot_plugin_localstore import get_data_dir + + return get_data_dir("nonebot_plugin_orm") / "db.sqlite3" + except Exception: + return None + + +def resolve_db_path() -> Path: + """解析本地消息库路径:环境变量 → ORM 配置 → localstore → 仓库默认位置。""" + override = os.environ.get(DB_PATH_ENV) + if override: + return Path(override).expanduser() + for candidate in (_driver_db_path(), _localstore_db_path()): + if candidate is not None: + return candidate + # hexi/plugins//history_store.py -> hexi/ + return ( + Path(__file__).resolve().parents[2] + / "data" + / "nonebot_plugin_orm" + / "db.sqlite3" + ) + + +def _open_readonly(path: Path) -> sqlite3.Connection: + """只读打开。路径可能含空格,必须用 as_uri 转义后的 file URI。""" + con = sqlite3.connect(f"{path.resolve().as_uri()}?mode=ro", uri=True, timeout=3.0) + con.row_factory = sqlite3.Row + return con + + +def _loads(raw: Any) -> dict[str, Any]: + if not raw: + return {} + if isinstance(raw, dict): + return raw + try: + data = json.loads(raw) + except (TypeError, ValueError): + return {} + return data if isinstance(data, dict) else {} + + +def _load_display_names(con: sqlite3.Connection, group_id: int) -> dict[str, tuple[str, str | None]]: + """取本群成员的 (昵称, 群名片);uninfo 表缺失时返回空表。""" + names: dict[str, tuple[str, str | None]] = {} + for row in con.execute(_NAMES_SQL, (str(group_id), SCENE_TYPE_GROUP)): + user_data = _loads(row["user_data"]) + member_data = _loads(row["member_data"]) + nickname = str(user_data.get("name") or "") + card = str(member_data.get("nick") or "") or None + if nickname or card: + names[str(row["user_id"])] = (nickname, card) + return names + + +def _parse_cq(cq: str) -> list[dict[str, Any]]: + """CQ 串转 OneBot 消息段(复用适配器自带解析,含反转义)。""" + try: + from nonebot.adapters.onebot.v11 import Message + + return [{"type": seg.type, "data": dict(seg.data)} for seg in Message(cq)] + except Exception as exc: + logger.debug(f"本地历史 CQ 解析失败,降级为纯文本: {exc}") + return [{"type": "text", "data": {"text": cq}}] + + +def _build_raw_message( + row: sqlite3.Row, names: dict[str, tuple[str, str | None]] +) -> dict[str, Any] | None: + """组装成 adapter._convert_message 能直接消费的 OneBot 原始结构。""" + message_id = row["message_id"] + if message_id is None: + return None + user_id = str(row["user_id"]) + nickname, card = names.get(user_id, ("", None)) + cq = row["raw_message"] or row["message"] or row["plain_text"] or "" + return { + "message_id": message_id, + "time": int(row["time"] or 0), + "sender": {"user_id": user_id, "nickname": nickname, "card": card or ""}, + "message": _parse_cq(cq), + } + + +def _fetch_group_messages( + group_id: str | int, + start_ts: int, + end_ts: int, + limit: int, + db_path: Path | str | None, +) -> LocalHistoryResult: + try: + gid = int(group_id) + except (TypeError, ValueError): + return LocalHistoryResult(error=f"群号非法: {group_id!r}") + try: + row_limit = max(1, int(limit)) + except (TypeError, ValueError): + row_limit = 1000 + + path = Path(db_path) if db_path is not None else resolve_db_path() + if not path.exists(): + return LocalHistoryResult(error=f"本地消息库不存在: {path}") + + start, end = int(start_ts), int(end_ts) + con = _open_readonly(path) + try: + # 多取一条用于判定截断(多出来的那条正是窗口内最旧的消息) + rows = con.execute(_MESSAGE_SQL, (gid, start, end, row_limit + 1)).fetchall() + truncated = len(rows) > row_limit + if truncated: + rows = rows[:row_limit] + window_total = None + if truncated: + try: + window_total = int(con.execute(_COUNT_SQL, (gid, start, end)).fetchone()[0]) + except (sqlite3.Error, TypeError, ValueError): + window_total = None + try: + names = _load_display_names(con, gid) + except sqlite3.Error as exc: + logger.debug(f"本地历史昵称补齐失败(忽略): {exc}") + names = {} + finally: + con.close() + + messages: list[dict[str, Any]] = [] + seen: set[str] = set() + duplicates = 0 + names_resolved = 0 + for row in rows: + raw = _build_raw_message(row, names) + if raw is None: + continue + message_id = str(raw["message_id"]) + if not message_id: + continue + if message_id in seen: + duplicates += 1 + continue + seen.add(message_id) + sender = raw["sender"] + if sender["nickname"] or sender["card"]: + names_resolved += 1 + messages.append(raw) + # 查询按 (time, id) 倒序取最近 N 条,翻回时间升序 + messages.reverse() + return LocalHistoryResult( + messages=messages, + truncated=truncated, + window_total=window_total, + names_resolved=names_resolved, + duplicates=duplicates, + ) + + +def fetch_group_messages( + group_id: str | int, + start_ts: int, + end_ts: int, + limit: int = 1000, + db_path: Path | str | None = None, +) -> LocalHistoryResult: + """读取 [start_ts, end_ts] 闭区间内的群消息(时间升序,最多 limit 条)。 + + 同步函数,调用方用 `asyncio.to_thread` 包;失败只返回带 error 的空结果。 + """ + try: + return _fetch_group_messages(group_id, start_ts, end_ts, limit, db_path) + except Exception as exc: # 兜底:本地库问题不该影响分析主链路 + logger.warning(f"本地历史读取异常: {exc}", exc_info=True) + return LocalHistoryResult(error=str(exc)) diff --git a/hexi/plugins/nonebot_plugin_group_daily_analysis/service.py b/hexi/plugins/nonebot_plugin_group_daily_analysis/service.py index 2bf59e5..573ecde 100644 --- a/hexi/plugins/nonebot_plugin_group_daily_analysis/service.py +++ b/hexi/plugins/nonebot_plugin_group_daily_analysis/service.py @@ -119,6 +119,7 @@ def register_bot_adapter(bot: Any, platform_id: str | None = None) -> OneBotAdap "platform_id": platform_id or "onebot", "bot_self_ids": bot_self_ids, "filter_bot_messages": config_manager.get_filter_bot_messages(), + "use_local_history": config_manager.get_use_local_history(), }, ) bot_manager.register_adapter(adapter, platform_id or "onebot") diff --git a/hexi/plugins/nonebot_plugin_helldivers_tools/handlers/war.py b/hexi/plugins/nonebot_plugin_helldivers_tools/handlers/war.py index edeb9d7..2462f12 100644 --- a/hexi/plugins/nonebot_plugin_helldivers_tools/handlers/war.py +++ b/hexi/plugins/nonebot_plugin_helldivers_tools/handlers/war.py @@ -14,6 +14,7 @@ from nonebot.internal.params import ArgPlainText from nonebot.matcher import Matcher from PIL import Image +from hexi.core import message_utils from ..services.equipment import get_equipment_by_combination, get_random_equipment from ..services.hd2_api import get_briefing_data from ..utils import gen_ms_img, pic2b64 @@ -24,7 +25,7 @@ _IMG_DIR = _BASE_DIR / "res" / "img" async def _send_war_card(matcher: Matcher, ev: MessageEvent, top_per_race: Optional[int]) -> None: - await matcher.send("正在获取前线战况,请民主的等待!") + await message_utils.common_proc_reply(ev.message_id) try: data = await get_briefing_data() png = await render_war_briefing(data, top_per_race) diff --git a/hexi/plugins/nonebot_plugin_helldivers_tools/res/img/helldivers/飞鹰毒气空袭.png b/hexi/plugins/nonebot_plugin_helldivers_tools/res/img/helldivers/飞鹰毒气空袭.png new file mode 100644 index 0000000..5138079 Binary files /dev/null and b/hexi/plugins/nonebot_plugin_helldivers_tools/res/img/helldivers/飞鹰毒气空袭.png differ diff --git a/hexi/plugins/nonebot_plugin_regif/handlers/reverse.py b/hexi/plugins/nonebot_plugin_regif/handlers/reverse.py index de49de1..18f3a76 100644 --- a/hexi/plugins/nonebot_plugin_regif/handlers/reverse.py +++ b/hexi/plugins/nonebot_plugin_regif/handlers/reverse.py @@ -8,6 +8,7 @@ from nonebot.log import logger from nonebot.plugin.on import on_keyword from nonebot_plugin_alconna import UniMessage +from hexi.core import message_utils from hexi.core.message_utils import get_reply_message from ..services.gif import reverse_gif_bytes @@ -21,10 +22,11 @@ re_gif = on_keyword({"倒放"}) async def rev_gif(event: MessageEvent, bot: Bot): # 情况1,用户对需要倒放的gif进行回复 messages = get_reply_message(event) - await match_revgif(messages) + message_id = event.message_id + await match_revgif(messages,message_id) -async def match_revgif(messages): +async def match_revgif(messages,message_id): img_urls = [] for message in messages: logger.info(f"遍历消息: {message}") @@ -38,7 +40,7 @@ async def match_revgif(messages): return for url in img_urls: logger.info(f"获取到的图片链接:{url}") - await UniMessage.text("ℹ正在翻转图片序列,请稍候").send() + await message_utils.common_proc_reply(message_id) data, err = await reverse_gif_bytes(img_urls) if err: await UniMessage.text(err).send() diff --git a/hexi/plugins/nonebot_plugin_steam_info/steam.py b/hexi/plugins/nonebot_plugin_steam_info/steam.py index 1e05d7c..cf51d63 100644 --- a/hexi/plugins/nonebot_plugin_steam_info/steam.py +++ b/hexi/plugins/nonebot_plugin_steam_info/steam.py @@ -203,7 +203,7 @@ def format_api_key_stats(stats: Dict) -> str: stats 结构: {key: {"call_count": int, "last_called": str | None}} """ - lines = ["Steam API key 使用统计:"] + lines = ["\nSteam API key 使用统计:"] for key, info in stats.items(): call_count = info.get("call_count", 0) last_called = info.get("last_called") diff --git a/hexi/resource/imgs/thinking.gif b/hexi/resource/imgs/thinking.gif new file mode 100644 index 0000000..9b4d192 Binary files /dev/null and b/hexi/resource/imgs/thinking.gif differ diff --git a/hexi/web/src/api/client.ts b/hexi/web/src/api/client.ts index 165025b..7846127 100644 --- a/hexi/web/src/api/client.ts +++ b/hexi/web/src/api/client.ts @@ -140,13 +140,17 @@ function makeSSEStream(opts: SSEStreamOpts): () => void { export function hubLogStream(opts: { since: number onLine: (line: string) => void + // 服务端检测到日志被轮转/清空时回调,调用方应丢弃已失效的旧窗口 + onReset?: () => void onStatus?: (s: 'open' | 'reconnecting' | 'closed') => void }): () => void { let since = opts.since return makeSSEStream({ url: () => '/hub/api/logs/stream?since=' + since, onData: (obj) => { - if (obj && obj.line !== undefined) { + if (!obj) return + if (obj.reset) { opts.onReset?.(); return } + if (obj.line !== undefined) { opts.onLine(obj.line) if (typeof obj.offset === 'number') since = obj.offset } diff --git a/hexi/web/src/connection.tsx b/hexi/web/src/connection.tsx index 9429c06..6ac863b 100644 --- a/hexi/web/src/connection.tsx +++ b/hexi/web/src/connection.tsx @@ -1,7 +1,7 @@ import { createContext, useContext, useEffect, useState, type ReactNode } from 'react' import { hubDashboardStream } from './api/client' -export type ConnStatus = '连接中' | '实时' | '重连中' | '已断开' +export type ConnStatus = '连接中' | '已连接' | '重连中' | '已断开' interface ConnState { status: ConnStatus @@ -28,7 +28,7 @@ export function ConnectionProvider({ children }: { children: ReactNode }) { if (d && d.ok !== false) { setData(d); setError(''); setLastUpdate(Date.now()) } else if (d && d.ok === false) setError(d.msg || '采集失败') }, - onStatus: (s) => setStatus(s === 'open' ? '实时' : s === 'reconnecting' ? '重连中' : '已断开'), + onStatus: (s) => setStatus(s === 'open' ? '已连接' : s === 'reconnecting' ? '重连中' : '已断开'), }) return dispose }, []) diff --git a/hexi/web/src/layout/AppLayout.tsx b/hexi/web/src/layout/AppLayout.tsx index 5649030..fe3d7dd 100644 --- a/hexi/web/src/layout/AppLayout.tsx +++ b/hexi/web/src/layout/AppLayout.tsx @@ -192,9 +192,9 @@ export function AppLayout() {
- + {status} {lastUpdate && {new Date(lastUpdate).toLocaleTimeString('zh-CN', { hour12: false })}} diff --git a/hexi/web/src/logs.tsx b/hexi/web/src/logs.tsx index 8b74348..a34100d 100644 --- a/hexi/web/src/logs.tsx +++ b/hexi/web/src/logs.tsx @@ -1,22 +1,36 @@ import { createContext, useCallback, useContext, useEffect, useRef, useState, type ReactNode } from 'react' import { hubLogs, hubLogStream } from './api/client' -// 保留的最大行数(窗口较大以容纳「加载更早」预置的历史行) -const MAX_LINES = 6000 +// 渲染窗口:日志页**始终只渲染这么多行**,先入先出——新行从尾部挤入,最旧的行从头部被挤掉。 +// 行数直接决定每次更新的 DOM 协调与布局开销,是页面的主要成本,不要随意调大。 +const WINDOW_LINES = 1000 +// 「向前翻页」每次额外取的行数:比窗口小,翻页时保留一半旧内容做重叠, +// 配合浏览器的滚动锚定,视线里的那行不会跳走。 +const PAGE_LINES = 500 + +// 新行合并刷新间隔(ms):日志突发时避免「一行一次 setState」 +const FLUSH_MS = 120 + +export interface LogLine { id: number; text: string } export interface LogInfo { file: string; size: number; total: number } interface LogState { - lines: string[] + lines: LogLine[] info: LogInfo error: string loading: boolean hasMore: boolean loadingEarlier: boolean - paused: boolean - setPaused: (v: boolean) => void + /** true = 自动滚动(跟随最新,新行实时入列);false = 冻结视图(新行暂存,回到底部再补上) */ + follow: boolean + setFollow: (v: boolean) => void + /** 自动换行:false 时长行不折行,改为横向滚动 */ + wrap: boolean + setWrap: (v: boolean) => void refresh: () => Promise - loadEarlier: () => Promise + /** 向前翻一页;返回本次前置进来的行数(0 表示没翻成),调用方据此把视图锚回原处 */ + loadEarlier: () => Promise } const Ctx = createContext({ @@ -26,62 +40,110 @@ const Ctx = createContext({ loading: false, hasMore: false, loadingEarlier: false, - paused: false, - setPaused: () => {}, + follow: true, + setFollow: () => {}, + wrap: true, + setWrap: () => {}, refresh: async () => {}, - loadEarlier: async () => {}, + loadEarlier: async () => 0, }) /** * 日志数据与日志流常驻 Provider:首次进入(任意页面)时拉取一次, * SSE 流保持不断,页面之间切换只读缓存数据,不重建连接、不显示加载动画。 + * + * 窗口模型:内存里始终只有 WINDOW_LINES 行(先入先出)。 + * - 自动滚动:新行进窗口尾部,最旧的行被挤掉(tail -f 行为) + * - 冻结视图:新行只进暂存区(同样保留最新 WINDOW_LINES 行),窗口不变, + * 回到底部时一次性补上 + * - 向前翻页:滚动到顶部时取更早的 PAGE_LINES 行,从尾部挤掉同样多的新行, + * 窗口行数不变(反过来的先入先出) */ export function LogProvider({ children }: { children: ReactNode }) { - const [lines, setLines] = useState([]) + const [lines, setLines] = useState([]) const [info, setInfo] = useState({ file: '', size: 0, total: 0 }) const [error, setError] = useState('') const [loading, setLoading] = useState(false) const [hasMore, setHasMore] = useState(false) const [loadingEarlier, setLoadingEarlier] = useState(false) - const [paused, setPaused] = useState(false) + const [follow, setFollow] = useState(true) + const [wrap, setWrap] = useState(true) const sinceRef = useRef(0) const earliestRef = useRef(0) - const pausedRef = useRef(false) - const pendingRef = useRef([]) + const followRef = useRef(true) + const frozenRef = useRef([]) const disposeRef = useRef<(() => void) | null>(null) const loadingEarlierRef = useRef(false) - const append = useCallback((line: string) => { - setLines(prev => [...prev, line].slice(-MAX_LINES)) + // 行 id:作为 React key 保持稳定(用下标做 key 时,窗口一滑动整列表都要 diff) + const idRef = useRef(0) + // 待入列的新行与合并刷新定时器 + const bufferRef = useRef([]) + const timerRef = useRef(null) + + const toItems = useCallback((texts: string[]): LogLine[] => { + const items: LogLine[] = [] + for (const text of texts) items.push({ id: idRef.current++, text }) + return items }, []) + const flush = useCallback(() => { + timerRef.current = null + const buffered = bufferRef.current + if (!buffered.length) return + bufferRef.current = [] + // 一次涌入超过整个窗口时(日志突发/整文件重放),只有最后 WINDOW_LINES 行留得下, + // 先裁掉前面的,避免为注定被挤掉的行白渲染一遍 + const kept = buffered.length > WINDOW_LINES ? buffered.slice(-WINDOW_LINES) : buffered + // 副作用放在 setState 之外:StrictMode 下 updater 可能被调用两次 + const items = toItems(kept) + setLines(prev => { + const next = prev.concat(items) + return next.length > WINDOW_LINES ? next.slice(-WINDOW_LINES) : next + }) + }, [toItems]) + + const enqueue = useCallback((line: string) => { + if (!followRef.current) { + // 冻结视图期间只暂存,且同样只保留最新 WINDOW_LINES 行 + frozenRef.current.push(line) + if (frozenRef.current.length > WINDOW_LINES) { + frozenRef.current.splice(0, frozenRef.current.length - WINDOW_LINES) + } + return + } + bufferRef.current.push(line) + if (timerRef.current == null) timerRef.current = window.setTimeout(flush, FLUSH_MS) + }, [flush]) + const connect = useCallback(() => { if (disposeRef.current) disposeRef.current() disposeRef.current = hubLogStream({ since: sinceRef.current, - onLine: (line) => { - if (pausedRef.current) { - pendingRef.current.push(line) - // 暂停期间只保留最近若干行,避免恢复时一次性渲染超大数组 - if (pendingRef.current.length > 500) pendingRef.current.splice(0, pendingRef.current.length - 500) - return - } - append(line) + onLine: enqueue, + onReset: () => { + // 服务端日志被轮转/清空:丢弃已失效的旧窗口,从新文件继续 + if (timerRef.current != null) { window.clearTimeout(timerRef.current); timerRef.current = null } + bufferRef.current = [] + frozenRef.current = [] + setLines([]) + earliestRef.current = 0 + setHasMore(false) }, }) - }, [append]) + }, [enqueue]) const refresh = useCallback(async () => { setLoading(true) setError('') try { - const d = await hubLogs(500) + const d = await hubLogs(WINDOW_LINES) if (!d) throw new Error('读取日志失败') if (d.ok === false) throw new Error(d.msg || '读取日志失败') - setLines(d.lines || []) + setLines(toItems(d.lines || [])) setInfo({ file: d.file || '', size: d.size || 0, total: d.total || 0 }) - // stream 从文件尾续读(end);「加载更早」从窗口起点向前翻(offset) + // stream 从文件尾续读(end);向前翻页从窗口起点继续往前(offset) sinceRef.current = d.end || 0 earliestRef.current = d.offset || 0 setHasMore((d.offset || 0) > 0) @@ -89,47 +151,61 @@ export function LogProvider({ children }: { children: ReactNode }) { } catch (e: any) { setError(e.message || '读取日志失败') } finally { setLoading(false) } - }, [connect]) + }, [connect, toItems]) - const loadEarlier = useCallback(async () => { + const loadEarlier = useCallback(async (): Promise => { const before = earliestRef.current - if (!before || loadingEarlierRef.current) return + if (!before || loadingEarlierRef.current) return 0 loadingEarlierRef.current = true setLoadingEarlier(true) setError('') try { - const d = await hubLogs(500, before) + const d = await hubLogs(PAGE_LINES, before) if (!d || d.ok === false) throw new Error((d && d.msg) || '读取日志失败') - setLines(prev => [...(d.lines || []), ...prev].slice(-MAX_LINES)) + const items = toItems(d.lines || []) + setLines(prev => { + const next = items.concat(prev) + // 往前翻时从尾部(较新的行)挤掉:窗口行数不变,新的行让位给历史 + return next.length > WINDOW_LINES ? next.slice(0, WINDOW_LINES) : next + }) earliestRef.current = d.offset || 0 setHasMore((d.offset || 0) > 0) + return items.length } catch (e: any) { setError(e.message || '读取日志失败') + return 0 } finally { loadingEarlierRef.current = false setLoadingEarlier(false) } - }, []) + }, [toItems]) // 首次进入就建立并保持;provider 卸载(登出/离开 /hub)时才断开 useEffect(() => { refresh() - return () => { if (disposeRef.current) disposeRef.current() } + return () => { + if (disposeRef.current) disposeRef.current() + // 组件卸载时清掉待刷新定时器,避免对已卸载组件 setState + if (timerRef.current != null) { window.clearTimeout(timerRef.current); timerRef.current = null } + } }, [refresh]) - useEffect(() => { pausedRef.current = paused }, [paused]) - useEffect(() => { - // 恢复接收时,把暂停期间暂存的行补进来 - if (!paused && pendingRef.current.length) { - const flush = pendingRef.current - pendingRef.current = [] - setLines(prev => [...prev, ...flush].slice(-MAX_LINES)) + followRef.current = follow + // 恢复自动滚动时,把冻结期间暂存的行一次性补上(仍只保留 WINDOW_LINES 行) + if (follow && frozenRef.current.length) { + const frozen = frozenRef.current + frozenRef.current = [] + const items = toItems(frozen) + setLines(prev => { + const next = prev.concat(items) + return next.length > WINDOW_LINES ? next.slice(-WINDOW_LINES) : next + }) } - }, [paused]) + }, [follow, toItems]) return ( - + {children} ) diff --git a/hexi/web/src/pages/logs/index.tsx b/hexi/web/src/pages/logs/index.tsx index 830f67e..5db1999 100644 --- a/hexi/web/src/pages/logs/index.tsx +++ b/hexi/web/src/pages/logs/index.tsx @@ -1,4 +1,4 @@ -import { useEffect, useRef, useState, type CSSProperties } from 'react' +import { memo, useCallback, useEffect, useRef, useState, type CSSProperties } from 'react' import { Button, Card, Spinner } from '@heroui/react' import { useLogs } from '../../logs' import { fmtBytes } from '../../lib/format' @@ -67,7 +67,9 @@ function parseAnsi(text: string): { text: string; style: CSSProperties }[] { return out } -function AnsiLine({ text }: { text: string }) { +// 记忆化:窗口内已渲染过的行必须能整体跳过,否则每次更新都要把整窗口的 +// ANSI 解析 + DOM 协调重做一遍。 +const AnsiLine = memo(function AnsiLine({ text }: { text: string }) { const parts = parseAnsi(text) const hasAnsi = parts.some((p) => Boolean(p.style.color || p.style.backgroundColor || p.style.fontWeight || p.style.textDecoration)) if (hasAnsi) { @@ -78,13 +80,16 @@ function AnsiLine({ text }: { text: string }) { const color = m && LEVEL_COLORS[m[1]] if (color) return {text} return <>{text} -} +}) export default function LogsPage() { // 数据与日志流由全局 LogProvider 常驻管理:离开/回来不重建连接、不转圈 - const { lines, info, error, loading, hasMore, loadingEarlier, paused, setPaused, loadEarlier } = useLogs() - const [follow, setFollow] = useState(true) + const { lines, info, error, loading, hasMore, loadingEarlier, follow, setFollow, wrap, setWrap, loadEarlier } = useLogs() const scrollRef = useRef(null) + // 一次「滚到顶」只翻一页,必须离开顶部后才会再次触发(避免连环翻页) + const topLoadBlockRef = useRef(false) + // 本次前置进来的行数:等 DOM 更新后据此把视图锚回原来的位置 + const pendingAnchorRef = useRef(null) useEffect(() => { if (follow && scrollRef.current) { @@ -92,19 +97,56 @@ export default function LogsPage() { } }, [lines, follow]) + // 向前翻页后把新行(它们被插在顶部)往上推出视野,让原本正在看的那一行留在原处。 + // 窗口行数固定,翻页前后总高度几乎不变,浏览器自带的滚动锚定在 scrollTop=0 时也不生效, + // 所以这里手动按元素位置锚定。 + useEffect(() => { + const n = pendingAnchorRef.current + if (n == null) return + const el = scrollRef.current + if (!el) { pendingAnchorRef.current = null; return } + const first = el.firstElementChild as HTMLElement | null + const target = el.children[n] as HTMLElement | undefined + if (!first || !target) return // DOM 还没更新,等下一次 lines 变化再试 + el.scrollTop = target.offsetTop - first.offsetTop + pendingAnchorRef.current = null + }, [lines]) + + const handleLoadEarlier = useCallback(async () => { + const n = await loadEarlier() + if (n > 0) pendingAnchorRef.current = n + }, [loadEarlier]) + + const onScroll = useCallback(() => { + const el = scrollRef.current + if (!el) return + const atBottom = el.scrollHeight - el.scrollTop - el.clientHeight < 24 + if (atBottom) { + // 回到底部即恢复跟随(否则新日志会把视图拽走,永远看不了历史) + if (!follow) setFollow(true) + return + } + if (follow) setFollow(false) + if (el.scrollTop > 8) { + topLoadBlockRef.current = false + return + } + if (!topLoadBlockRef.current && hasMore && !loadingEarlier) { + topLoadBlockRef.current = true + handleLoadEarlier() + } + }, [follow, hasMore, loadingEarlier, handleLoadEarlier, setFollow]) + return (

查看日志

-

实时读取 bot 运行日志(SSE 推送)

+

实时读取 bot 运行日志(SSE 推送,固定窗口滚动,向上滚动自动加载更早)

- {hasMore && ( - - )} - - + +
@@ -113,16 +155,25 @@ export default function LogsPage() {
- 日志流 + + 日志流 + {loadingEarlier && 加载更早…} + {info.file && {info.file} · {fmtBytes(info.size)} · 共 {info.total} 行}
-
+
{loading && !lines.length ? (
- ) : lines.length ? lines.map((ln, i) => ( -
+ ) : lines.length ? lines.map(ln => ( +
)) : (
暂无日志…
)} diff --git a/hexi/web_hub/__init__.py b/hexi/web_hub/__init__.py index 72569ed..ca3faf4 100644 --- a/hexi/web_hub/__init__.py +++ b/hexi/web_hub/__init__.py @@ -46,6 +46,10 @@ basic_path = Path(__file__).resolve().parent # hexi/web 是统一 Web 管理台前端(hexi/web/dist),不是插件目录下的 web WEB_DIST = Path(__file__).resolve().parents[1] / "web" / "dist" +# 入口 HTML 的响应头:必须每次回源校验。否则重新构建后浏览器会沿用缓存的 +# 旧 index.html(进而加载已被删除的旧 hashed bundle),Web 端一直跑旧代码。 +_NO_CACHE = {"Cache-Control": "no-cache"} + def _hub_version() -> str: """管理台版本号:以 hexi/web/package.json 为唯一来源(前端构建也读它)。""" try: @@ -239,8 +243,34 @@ def _tail_log_lines( return lines, start, size +# 行数统计缓存:路径 -> (已统计到的字节偏移, 行数)。日志只追加不重写, +# 因此每次只需扫「自上次统计以来新增的字节」,不必重扫整个文件。 +_COUNT_CACHE: dict[str, tuple[int, int]] = {} + +# SSE 单轮最多读取的字节数:避免日志突发时一次性把大量行灌给前端 +_LOG_STREAM_CHUNK = 1 << 20 + + def _count_log_lines(path: Path) -> int: - """分块统计全文件行数(仅换行计数,不全量载入)。""" + """统计全文件行数(仅换行计数,不全量载入)。 + + 首次或检测到文件被截断时全量扫描;之后按增量累加——否则每次翻页 + 都要在事件循环里把整个日志(可达 20MB)重扫一遍。 + """ + key = str(path) + size = path.stat().st_size + cached = _COUNT_CACHE.get(key) + if cached is not None: + done, count = cached + if done == size: + return count + if 0 < done < size: + with open(path, "rb") as f: + f.seek(done) + chunk = f.read() + count += chunk.count(b"\n") + _COUNT_CACHE[key] = (done + len(chunk), count) + return count count = 0 with open(path, "rb") as f: while True: @@ -248,6 +278,7 @@ def _count_log_lines(path: Path) -> int: if not chunk: break count += chunk.count(b"\n") + _COUNT_CACHE[key] = (size, count) return count @@ -530,7 +561,7 @@ def build_hub_app() -> FastAPI: async def dashboard_stream( _: dict = Depends(get_current_user), ): - """Dashboard 状态 SSE 实时推送(每 2 秒采集一次)。""" + """Dashboard 状态 SSE 实时推送(每 5 秒采集一次)。""" async def gen(): while True: @@ -540,7 +571,7 @@ def build_hub_app() -> FastAPI: except Exception as e: # noqa: BLE001 payload = {"ok": False, "msg": f"采集失败: {e}"} yield f"data: {json.dumps(payload, ensure_ascii=False)}\n\n" - await asyncio.sleep(2) + await asyncio.sleep(5) return StreamingResponse(gen(), media_type="text/event-stream") @@ -549,7 +580,12 @@ def build_hub_app() -> FastAPI: since: int = 0, _: dict = Depends(get_current_user), ): - """SSE 实时日志流:从 since(字节偏移)增量推送新日志行。""" + """SSE 实时日志流:从 since(字节偏移)增量推送新日志行。 + + 检测到日志被轮转/清空(since 超出当前文件大小)时发一帧 {"reset": true} + 让前端丢掉已失效的旧窗口,然后从新文件末尾继续——不能回退到 0, + 否则重连的客户端会把整个文件重放一遍(20MB ≈ 十万行)。 + """ root = Path(__file__).resolve().parents[2] # 仓库根目录 candidates = [ root / "_bot_run.log", @@ -570,12 +606,14 @@ def build_hub_app() -> FastAPI: await asyncio.sleep(1.0) continue if current > size: - current = 0 # 日志被轮转/截断 + # 日志被轮转/清空:跳到新文件末尾并通知前端丢掉旧窗口 + current = size + yield 'data: {"reset": true}\n\n' if current < size: try: with open(path, "rb") as f: f.seek(current) - raw = f.read() + raw = f.read(_LOG_STREAM_CHUNK) except OSError: await asyncio.sleep(1.0) continue @@ -594,6 +632,9 @@ def build_hub_app() -> FastAPI: line = raw_line.decode("gbk", errors="replace") payload = {"line": line.rstrip("\r"), "offset": current} yield f"data: {json.dumps(payload, ensure_ascii=False)}\n\n" + elif len(raw) >= _LOG_STREAM_CHUNK: + # 单行超过单轮读取上限:整块丢弃,否则游标永远不前进 + current += len(raw) await asyncio.sleep(0.5) beat += 1 if beat >= 30: # 每 ~15s 一次心跳保活 @@ -622,7 +663,9 @@ def build_hub_app() -> FastAPI: @app.get("/") async def index(): if (WEB_DIST / "index.html").exists(): - return FileResponse(WEB_DIST / "index.html") + # no-cache:入口 HTML 必须每次回源校验,否则重新构建后浏览器会继续 + # 用缓存的旧 index.html 去加载已被替换掉的旧 hashed bundle(dist/assets)。 + return FileResponse(WEB_DIST / "index.html", headers=_NO_CACHE) return HTMLResponse( "

HeXi Web Hub

前端未构建,请在 hexi/web 执行 " "npm run build。

" @@ -634,7 +677,7 @@ def build_hub_app() -> FastAPI: if path and target.is_file() and target.is_relative_to(WEB_DIST.resolve()): return FileResponse(target) if (WEB_DIST / "index.html").exists(): - return FileResponse(WEB_DIST / "index.html") + return FileResponse(WEB_DIST / "index.html", headers=_NO_CACHE) raise HTTPException(status_code=404, detail="页面不存在") return app diff --git a/hexi/web_hub/dashboard.py b/hexi/web_hub/dashboard.py index 2de528e..77d2e95 100644 --- a/hexi/web_hub/dashboard.py +++ b/hexi/web_hub/dashboard.py @@ -6,7 +6,10 @@ - host 性能:psutil(CPU/内存/磁盘/进程)+ platform + nonebot 版本 仅在 /hub/api/dashboard 被调用时执行,不做常驻采样。 同步采集(psutil 等可能阻塞的调用)整体放入线程池执行,避免阻塞事件循环; -结果做 1 秒短缓存,多个前端标签页(含 SSE 循环)不至于各自重复采集。 +结果做短缓存,多个前端标签页(含 SSE 循环)不至于各自重复采集。 +CPU 采样走非阻塞模式(interval=None,与上一次调用求差):带 interval 的 +psutil.cpu_percent/Process.cpu_percent 会真的 sleep 住调用线程,而 SSE 循环 +每轮都调用一次,会凭空常驻烧掉约四分之一个核。 """ from __future__ import annotations @@ -28,21 +31,31 @@ _MODULE_START = time.time() # 网络速率采样:记录上一次累计计数与时间,用于计算实时 bytes/s 用量 _NET_LAST: dict = {"time": None, "sent": None, "recv": None} -# 采集结果短缓存(多标签页 / SSE 每 2s 循环共用一份结果) +# 采集结果短缓存(多标签页 / SSE 循环共用一份结果) _DASH_CACHE: dict = {"time": 0.0, "data": None} -_DASH_CACHE_TTL = 1.0 +_DASH_CACHE_TTL = 2.0 + +# 进程对象与 CPU 采样必须跨调用复用:psutil 的非阻塞模式(interval=None) +# 是「与上一次调用」求差,换对象或首次调用只会拿到 0.0。 +_PROC = psutil.Process() +psutil.cpu_percent(None, True) # 预热 CPU 采样(首次调用恒为 0.0) +_PROC.cpu_percent(None) # 预热进程采样 + +# bot 昵称(get_login_info)基本不变,长缓存避免每轮都打一次 OneBot API +_LOGIN_CACHE: dict[str, tuple[float, str]] = {} +_LOGIN_TTL = 600.0 def _cpu_sync() -> dict: try: - per_core = psutil.cpu_percent(0.3, True) + per_core = psutil.cpu_percent(None, True) except TypeError: # 兼容部分平台不接受 percpu 参数 per_core = [] if per_core: percent = sum(per_core) / len(per_core) else: - percent = psutil.cpu_percent(0.3) + percent = psutil.cpu_percent(None) info: dict = { "percent": round(float(percent), 1), "per_core": [round(float(x), 1) for x in (per_core or [])], @@ -84,8 +97,8 @@ def _memory_sync() -> dict: def _process_sync() -> dict: try: - proc = psutil.Process() - cpu = proc.cpu_percent(0.2) + proc = _PROC + cpu = proc.cpu_percent(None) info = { "pid": proc.pid, "name": proc.name(), @@ -253,13 +266,18 @@ async def _collect_bots() -> list: "msg_sent": None, } if ws_connected: - try: - login = await bot.get_login_info() - item["nick"] = login.get("nickname") or item["nick"] - except Exception as e: # noqa: BLE001 - logger.warning( - f"获取登录信息失败({bot.self_id}): {type(e).__name__}: {e}" - ) + cached = _LOGIN_CACHE.get(bot.self_id) + if cached and time.time() - cached[0] < _LOGIN_TTL: + item["nick"] = cached[1] + else: + try: + login = await bot.get_login_info() + item["nick"] = login.get("nickname") or item["nick"] + _LOGIN_CACHE[bot.self_id] = (time.time(), item["nick"]) + except Exception as e: # noqa: BLE001 + logger.warning( + f"获取登录信息失败({bot.self_id}): {type(e).__name__}: {e}" + ) try: status = await bot.get_status() item["online"] = status.get("online") diff --git a/tests/test_group_daily_history_store.py b/tests/test_group_daily_history_store.py new file mode 100644 index 0000000..8eef99f --- /dev/null +++ b/tests/test_group_daily_history_store.py @@ -0,0 +1,284 @@ +"""群分析本地消息源 (history_store) 单元测试 + +history_store 只依赖标准库,故用 importlib 按文件路径裸加载 +(与 test_rate_limit.py 同法),避免触发插件包 __init__ 的 NoneBot 初始化。 +""" + +import importlib.util +import json +import sqlite3 +import sys +from pathlib import Path + +import pytest + +_MODULE_PATH = ( + Path(__file__).resolve().parents[1] + / "hexi" + / "plugins" + / "nonebot_plugin_group_daily_analysis" + / "history_store.py" +) + +GROUP_ID = 872490448 +BASE_TS = 1_800_000_000 + + +@pytest.fixture(scope="module") +def hs(): + """以独立模块名加载 history_store.py,避免与插件包 __init__ 冲突""" + spec = importlib.util.spec_from_file_location( + "history_store_under_test", _MODULE_PATH + ) + module = importlib.util.module_from_spec(spec) + sys.modules[spec.name] = module + spec.loader.exec_module(module) + yield module + sys.modules.pop(spec.name, None) + + +def _create_db(path: Path, *, with_uninfo: bool = True) -> sqlite3.Connection: + """按真实列名建 learning_chat_message(可选带 uninfo 三表)""" + con = sqlite3.connect(path) + con.execute( + """ + CREATE TABLE learning_chat_message ( + id INTEGER PRIMARY KEY AUTOINCREMENT, + group_id BIGINT NOT NULL, + user_id BIGINT NOT NULL, + message_id BIGINT NOT NULL, + message TEXT NOT NULL, + raw_message TEXT NOT NULL, + plain_text TEXT NOT NULL, + time INTEGER NOT NULL + ) + """ + ) + con.execute( + "CREATE INDEX ix_message_group_time" + " ON learning_chat_message (group_id, time)" + ) + if with_uninfo: + con.execute( + "CREATE TABLE nonebot_plugin_uninfo_scenemodel" + " (id INTEGER PRIMARY KEY, scene_id VARCHAR(64), scene_type INTEGER)" + ) + con.execute( + "CREATE TABLE nonebot_plugin_uninfo_sessionmodel" + " (id INTEGER PRIMARY KEY, scene_persist_id INTEGER," + " user_persist_id INTEGER, member_data JSON)" + ) + con.execute( + "CREATE TABLE nonebot_plugin_uninfo_usermodel" + " (id INTEGER PRIMARY KEY, user_id VARCHAR(64), user_data JSON)" + ) + return con + + +def _add_message( + con: sqlite3.Connection, + *, + message_id: int, + user_id: int = 1001, + ts: int = BASE_TS, + raw: str = "hi", + plain_text: str | None = None, + group_id: int = GROUP_ID, +) -> None: + con.execute( + "INSERT INTO learning_chat_message" + " (group_id, user_id, message_id, message, raw_message, plain_text, time)" + " VALUES (?, ?, ?, ?, ?, ?, ?)", + ( + group_id, + user_id, + message_id, + raw, + raw, + raw if plain_text is None else plain_text, + ts, + ), + ) + + +def _add_member(con, user_id: str, nickname: str, card: str | None) -> None: + """写入 uninfo 场景/用户/会话三条记录(群场景 scene_type=1)""" + con.execute( + "INSERT OR IGNORE INTO nonebot_plugin_uninfo_scenemodel" + " (id, scene_id, scene_type) VALUES (1, ?, 1)", + (str(GROUP_ID),), + ) + cur = con.execute( + "INSERT INTO nonebot_plugin_uninfo_usermodel (user_id, user_data)" + " VALUES (?, ?)", + (user_id, json.dumps({"name": nickname, "nick": ""}, ensure_ascii=False)), + ) + con.execute( + "INSERT INTO nonebot_plugin_uninfo_sessionmodel" + " (scene_persist_id, user_persist_id, member_data) VALUES (1, ?, ?)", + (cur.lastrowid, json.dumps({"nick": card or ""}, ensure_ascii=False)), + ) + + +def _build(path: Path, rows: list[dict], *, with_uninfo: bool = True) -> None: + con = _create_db(path, with_uninfo=with_uninfo) + for row in rows: + _add_message(con, **row) + con.commit() + con.close() + + +def test_cq_parsed_into_onebot_segments(hs, tmp_path): + db = tmp_path / "cq.db" + _build( + db, + [ + { + "message_id": 1, + "raw": "[CQ:reply,id=5][CQ:at,qq=2]hi" + "[CQ:image,file=a.jpg,url=http://x?a=1&b=2]", + } + ], + ) + + res = hs.fetch_group_messages(GROUP_ID, BASE_TS - 1, BASE_TS + 1, 100, db_path=db) + + assert res and res.error is None + segs = res.messages[0]["message"] + assert [s["type"] for s in segs] == ["reply", "at", "text", "image"] + assert segs[0]["data"] == {"id": "5"} + assert segs[1]["data"] == {"qq": "2"} + assert segs[2]["data"] == {"text": "hi"} + assert segs[3]["data"]["url"] == "http://x?a=1&b=2" # CQ 反转义 + # 交给 adapter._convert_message 的字段形状 + assert res.messages[0]["sender"]["user_id"] == "1001" + + +def test_window_is_inclusive(hs, tmp_path): + db = tmp_path / "window.db" + _build( + db, + [ + {"message_id": 1, "ts": BASE_TS - 1}, + {"message_id": 2, "ts": BASE_TS}, + {"message_id": 3, "ts": BASE_TS + 1}, + ], + ) + + res = hs.fetch_group_messages(GROUP_ID, BASE_TS, BASE_TS, 100, db_path=db) + + # 两端闭区间:起始时间戳上的消息必须被取到(增量水位依赖这一点) + assert [m["message_id"] for m in res.messages] == [2] + + +def test_same_second_sorted_by_id_ascending(hs, tmp_path): + db = tmp_path / "order.db" + _build( + db, + [ + {"message_id": 11, "ts": BASE_TS}, + {"message_id": 12, "ts": BASE_TS}, + {"message_id": 13, "ts": BASE_TS}, + ], + ) + + res = hs.fetch_group_messages(GROUP_ID, BASE_TS, BASE_TS, 100, db_path=db) + + assert [m["message_id"] for m in res.messages] == [11, 12, 13] + + +def test_duplicate_message_id_deduped(hs, tmp_path): + db = tmp_path / "dup.db" + _build(db, [{"message_id": 7}, {"message_id": 7}]) + + res = hs.fetch_group_messages(GROUP_ID, BASE_TS - 1, BASE_TS + 1, 100, db_path=db) + + assert len(res.messages) == 1 + assert res.duplicates == 1 + + +def test_limit_truncates_and_keeps_newest(hs, tmp_path): + db = tmp_path / "trunc.db" + _build( + db, + [ + {"message_id": 1, "ts": BASE_TS}, + {"message_id": 2, "ts": BASE_TS + 1}, + {"message_id": 3, "ts": BASE_TS + 2}, + ], + ) + + res = hs.fetch_group_messages(GROUP_ID, BASE_TS - 1, BASE_TS + 5, 2, db_path=db) + + assert res.truncated is True + assert res.window_total == 3 + assert [m["message_id"] for m in res.messages] == [2, 3] # 丢最旧的 + + +def test_names_resolved_from_uninfo(hs, tmp_path): + db = tmp_path / "names.db" + con = _create_db(db) + _add_member(con, "1001", "昵称甲", "群名片甲") + _add_member(con, "1002", "昵称乙", "") + _add_message(con, message_id=1, user_id=1001) + _add_message(con, message_id=2, user_id=1002) + _add_message(con, message_id=3, user_id=1003) # 未在 uninfo 中 + con.commit() + con.close() + + res = hs.fetch_group_messages(GROUP_ID, BASE_TS - 1, BASE_TS + 1, 100, db_path=db) + + senders = {m["message_id"]: m["sender"] for m in res.messages} + assert senders[1]["nickname"] == "昵称甲" + assert senders[1]["card"] == "群名片甲" + assert senders[2]["nickname"] == "昵称乙" + assert senders[2]["card"] == "" # 空群名片交给 _convert_message 归一为 None + assert senders[3]["nickname"] == "" and senders[3]["card"] == "" + assert res.names_resolved == 2 + + # 群名片空串在源数据里就归一为 None + con = hs._open_readonly(db) + try: + assert hs._load_display_names(con, GROUP_ID)["1002"] == ("昵称乙", None) + finally: + con.close() + + +def test_missing_db_reports_error_without_creating(hs, tmp_path): + missing = tmp_path / "nope.sqlite3" + + res = hs.fetch_group_messages( + GROUP_ID, BASE_TS - 1, BASE_TS + 1, 10, db_path=missing + ) + + assert not res + assert res.error and "不存在" in res.error + assert not missing.exists() # mode=ro 不得建库 + + +def test_missing_table_is_not_fatal(hs, tmp_path): + db = tmp_path / "empty.db" + con = sqlite3.connect(db) + con.execute("CREATE TABLE unrelated (x INTEGER)") + con.commit() + con.close() + + res = hs.fetch_group_messages(GROUP_ID, BASE_TS - 1, BASE_TS + 1, 10, db_path=db) + + assert not res + assert res.error and "learning_chat_message" in res.error + + +def test_other_group_filtered_out(hs, tmp_path): + db = tmp_path / "other.db" + _build( + db, + [ + {"message_id": 1}, + {"message_id": 2, "group_id": GROUP_ID + 1}, + ], + ) + + res = hs.fetch_group_messages(GROUP_ID, BASE_TS - 1, BASE_TS + 1, 100, db_path=db) + + assert [m["message_id"] for m in res.messages] == [1]