Files

128 lines
5.5 KiB
Python
Raw Permalink Normal View History

"""群聊人设卡 —— 数据层:表结构定义(与 DESIGN.md §3 对应)"""
from datetime import datetime
from typing import Optional
from nonebot import require
require("nonebot_plugin_orm")
from nonebot_plugin_orm import Model
from sqlalchemy import BigInteger, Boolean, DateTime, Index, Integer, String, Text, UniqueConstraint
from sqlalchemy.orm import Mapped, mapped_column
class PersonaGroup(Model):
"""管理群列表 + 采集开关
一行 = 一个被管理的群(由用户在 Web 后台从 OneBot 群列表中选择加入);
enabled 控制该群是否采集(群关闭则群里所有人都不采集)。
主页面/采集器只读这张表,不依赖 OneBot 在线。
"""
__tablename__ = "persona_group"
group_id: Mapped[int] = mapped_column(BigInteger, primary_key=True)
group_name: Mapped[str] = mapped_column(String(64), default="") # 群名快照(添加时记录)
enabled: Mapped[bool] = mapped_column(Boolean, default=False)
updated_at: Mapped[datetime] = mapped_column(DateTime, default=datetime.now)
class PersonaUser(Model):
"""人设计划参与者(按群 opt-in,一行一人一群)"""
__tablename__ = "persona_user"
user_id: Mapped[int] = mapped_column(BigInteger, primary_key=True)
group_id: Mapped[int] = mapped_column(BigInteger, primary_key=True)
joined_at: Mapped[datetime] = mapped_column(DateTime, default=datetime.now)
class PersonaChatLog(Model):
"""采集语料(五维):内容 / 谁发的 / 发给谁 / 几点发的 / 被回复内容快照
只存治理层处理后的纯文本;数据来源无关(群事件、历史导入均可写入)。
"""
__tablename__ = "persona_chat_log"
__table_args__ = (
Index("ix_persona_log_user_time", "user_id", "group_id", "created_at"),
)
id: Mapped[int] = mapped_column(Integer, primary_key=True, autoincrement=True)
user_id: Mapped[int] = mapped_column(BigInteger) # 谁发的
group_id: Mapped[int] = mapped_column(BigInteger)
nickname: Mapped[str] = mapped_column(String(64), default="") # 群昵称快照
content: Mapped[str] = mapped_column(Text) # 内容
target_user_id: Mapped[Optional[int]] = mapped_column( # 发给谁(回复/@ 对象,或继承自链条)
BigInteger, nullable=True
)
target_inherited: Mapped[bool] = mapped_column( # target 是否从发言段链条继承
Boolean, default=False
)
follows_id: Mapped[Optional[int]] = mapped_column( # 同一说话人的上一条语料 id(发言段链条)
Integer, nullable=True
)
image_count: Mapped[int] = mapped_column(Integer, default=0) # 本条图片数
image_hashes: Mapped[Optional[str]] = mapped_column( # 图片 hash 列表 JSON(表情包去重识别)
Text, nullable=True
)
reply_to_content: Mapped[Optional[str]] = mapped_column( # 被回复内容快照
Text, nullable=True
)
created_at: Mapped[datetime] = mapped_column(DateTime, default=datetime.now) # 几点发的
class PersonaImpression(Model):
"""印象:语料 → LLM 自然语言印象(两级流水线中间产物,画像的输入)"""
__tablename__ = "persona_impression"
__table_args__ = (
Index("ix_persona_impression_user_time", "user_id", "group_id", "created_at"),
)
id: Mapped[int] = mapped_column(Integer, primary_key=True, autoincrement=True)
user_id: Mapped[int] = mapped_column(BigInteger)
group_id: Mapped[int] = mapped_column(BigInteger)
content: Mapped[str] = mapped_column(Text) # 自然语言印象(话题/氛围/互动)
cover_from_id: Mapped[int] = mapped_column(Integer, default=0) # 覆盖语料起点 id
cover_to_id: Mapped[int] = mapped_column(Integer, default=0) # 覆盖语料终点 id(增量依据)
model: Mapped[str] = mapped_column(String(64), default="")
created_at: Mapped[datetime] = mapped_column(DateTime, default=datetime.now)
class PersonaImage(Model):
"""图片识别结果缓存(多模态预留):同一 hash 只识别一次
当前未启用识别(vision.recognizer 为占位实现),表结构先行预留。
"""
__tablename__ = "persona_image"
hash: Mapped[str] = mapped_column(String(64), primary_key=True) # 图片 hash(与 chat_log.image_hashes 对应)
description: Mapped[str] = mapped_column(Text, default="") # 多模态识别结果
model: Mapped[str] = mapped_column(String(64), default="") # 识别模型
recognized_at: Mapped[datetime] = mapped_column(DateTime, default=datetime.now)
class PersonaSummary(Model):
"""画像快照:九段 markdown 协议(DESIGN.md §4),版本化永久留档"""
__tablename__ = "persona_summary"
__table_args__ = (
UniqueConstraint(
"user_id", "group_id", "version", name="uq_persona_summary_user_group_version"
),
Index("ix_persona_summary_user_group", "user_id", "group_id"),
)
id: Mapped[int] = mapped_column(Integer, primary_key=True, autoincrement=True)
user_id: Mapped[int] = mapped_column(BigInteger)
group_id: Mapped[int] = mapped_column(BigInteger)
version: Mapped[int] = mapped_column(Integer)
card_text: Mapped[str] = mapped_column(Text) # 九段 markdown 画像原文
corpus_count: Mapped[int] = mapped_column(Integer, default=0) # 元信息:语料覆盖条数
impression_count: Mapped[int] = mapped_column(Integer, default=0) # 元信息:印象条数
model: Mapped[str] = mapped_column(String(64), default="")
created_at: Mapped[datetime] = mapped_column(DateTime, default=datetime.now)