From 9ff790554ddb39f314ef240bfdc5b336b053a7fa Mon Sep 17 00:00:00 2001 From: 1445043649 <> Date: Sun, 10 May 2026 18:50:17 +0800 Subject: [PATCH] =?UTF-8?q?docs(persistence):=20=E7=BB=99=20ORM=20?= =?UTF-8?q?=E8=A1=A8/=E5=AD=97=E6=AE=B5=E8=A1=A5=E5=85=85=E4=B8=AD?= =?UTF-8?q?=E6=96=87=E6=B3=A8=E9=87=8A?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 通过 SQLAlchemy 的 comment= 给 5 张持久化表(users / threads_meta / runs / run_events / feedback)的所有字段以及表本身加上中文注释,便于读代码、 生成文档与未来切到 Postgres 时直接落库为 COMMENT ON。 SQLite 引擎本身不支持 COMMENT ON,运行时不会改变 .schema 输出。 Co-Authored-By: Claude Opus 4.7 (1M context) --- .../deerflow/persistence/feedback/model.py | 28 ++++---- .../deerflow/persistence/models/run_event.py | 30 +++++---- .../harness/deerflow/persistence/run/model.py | 64 ++++++++++--------- .../deerflow/persistence/thread_meta/model.py | 18 +++--- .../deerflow/persistence/user/model.py | 30 +++------ 5 files changed, 83 insertions(+), 87 deletions(-) diff --git a/backend/packages/harness/deerflow/persistence/feedback/model.py b/backend/packages/harness/deerflow/persistence/feedback/model.py index a9b6479b..aa19f9a6 100644 --- a/backend/packages/harness/deerflow/persistence/feedback/model.py +++ b/backend/packages/harness/deerflow/persistence/feedback/model.py @@ -13,20 +13,16 @@ from deerflow.persistence.base import Base class FeedbackRow(Base): __tablename__ = "feedback" - __table_args__ = (UniqueConstraint("thread_id", "run_id", "user_id", name="uq_feedback_thread_run_user"),) + __table_args__ = ( + UniqueConstraint("thread_id", "run_id", "user_id", name="uq_feedback_thread_run_user"), + {"comment": "用户对运行结果的反馈(点赞/点踩 + 文字评论),(thread, run, user) 唯一"}, + ) - feedback_id: Mapped[str] = mapped_column(String(64), primary_key=True) - run_id: Mapped[str] = mapped_column(String(64), nullable=False, index=True) - thread_id: Mapped[str] = mapped_column(String(64), nullable=False, index=True) - user_id: Mapped[str | None] = mapped_column(String(64), index=True) - message_id: Mapped[str | None] = mapped_column(String(64)) - # message_id is an optional RunEventStore event identifier — - # allows feedback to target a specific message or the entire run - - rating: Mapped[int] = mapped_column(nullable=False) - # +1 (thumbs-up) or -1 (thumbs-down) - - comment: Mapped[str | None] = mapped_column(Text) - # Optional text feedback from the user - - created_at: Mapped[datetime] = mapped_column(DateTime(timezone=True), default=lambda: datetime.now(UTC)) + feedback_id: Mapped[str] = mapped_column(String(64), primary_key=True, comment="反馈主键") + run_id: Mapped[str] = mapped_column(String(64), nullable=False, index=True, comment="关联的运行 ID(runs.run_id)") + thread_id: Mapped[str] = mapped_column(String(64), nullable=False, index=True, comment="关联的会话 ID(threads_meta.thread_id)") + user_id: Mapped[str | None] = mapped_column(String(64), index=True, comment="反馈作者;为 NULL 表示历史无主数据") + message_id: Mapped[str | None] = mapped_column(String(64), comment="可选的 RunEventStore 事件 ID;为 NULL 表示针对整次运行而非单条消息") + rating: Mapped[int] = mapped_column(nullable=False, comment="评分:+1 点赞,-1 点踩") + comment: Mapped[str | None] = mapped_column(Text, comment="可选的文字评论") + created_at: Mapped[datetime] = mapped_column(DateTime(timezone=True), default=lambda: datetime.now(UTC), comment="创建时间(UTC)") diff --git a/backend/packages/harness/deerflow/persistence/models/run_event.py b/backend/packages/harness/deerflow/persistence/models/run_event.py index 4f22b461..63f06b1a 100644 --- a/backend/packages/harness/deerflow/persistence/models/run_event.py +++ b/backend/packages/harness/deerflow/persistence/models/run_event.py @@ -13,23 +13,25 @@ from deerflow.persistence.base import Base class RunEventRow(Base): __tablename__ = "run_events" - id: Mapped[int] = mapped_column(primary_key=True, autoincrement=True) - thread_id: Mapped[str] = mapped_column(String(64), nullable=False) - run_id: Mapped[str] = mapped_column(String(64), nullable=False) - # Owner of the conversation this event belongs to. Nullable for data - # created before auth was introduced; populated by auth middleware on - # new writes and by the boot-time orphan migration on existing rows. - user_id: Mapped[str | None] = mapped_column(String(64), nullable=True, index=True) - event_type: Mapped[str] = mapped_column(String(32), nullable=False) - category: Mapped[str] = mapped_column(String(16), nullable=False) - # "message" | "trace" | "lifecycle" - content: Mapped[str] = mapped_column(Text, default="") - event_metadata: Mapped[dict] = mapped_column(JSON, default=dict) - seq: Mapped[int] = mapped_column(nullable=False) - created_at: Mapped[datetime] = mapped_column(DateTime(timezone=True), default=lambda: datetime.now(UTC)) + id: Mapped[int] = mapped_column(primary_key=True, autoincrement=True, comment="自增主键") + thread_id: Mapped[str] = mapped_column(String(64), nullable=False, comment="所属会话 ID(threads_meta.thread_id)") + run_id: Mapped[str] = mapped_column(String(64), nullable=False, comment="所属运行 ID(runs.run_id)") + user_id: Mapped[str | None] = mapped_column( + String(64), + nullable=True, + index=True, + comment="会话所有者;为 NULL 表示鉴权引入之前的历史数据,新写入由 auth 中间件填充,启动期 orphan 迁移会回填存量", + ) + event_type: Mapped[str] = mapped_column(String(32), nullable=False, comment="事件子类型(具体含义由 category 决定,如 ai_message_chunk、tool_call、run_started)") + category: Mapped[str] = mapped_column(String(16), nullable=False, comment='事件大类:"message" 消息 / "trace" 追踪 / "lifecycle" 生命周期') + content: Mapped[str] = mapped_column(Text, default="", comment="事件文本内容(消息体、错误、状态字符串等)") + event_metadata: Mapped[dict] = mapped_column(JSON, default=dict, comment="事件结构化元数据(JSON),随 event_type 而异") + seq: Mapped[int] = mapped_column(nullable=False, comment="在 thread 内的全局递增序号;与 thread_id 组合唯一") + created_at: Mapped[datetime] = mapped_column(DateTime(timezone=True), default=lambda: datetime.now(UTC), comment="创建时间(UTC)") __table_args__ = ( UniqueConstraint("thread_id", "seq", name="uq_events_thread_seq"), Index("ix_events_thread_cat_seq", "thread_id", "category", "seq"), Index("ix_events_run", "thread_id", "run_id", "seq"), + {"comment": "运行事件流(消息/追踪/生命周期事件按 seq 顺序追加,是消息回放与审计的真源)"}, ) diff --git a/backend/packages/harness/deerflow/persistence/run/model.py b/backend/packages/harness/deerflow/persistence/run/model.py index d0dfe408..0e86e0a1 100644 --- a/backend/packages/harness/deerflow/persistence/run/model.py +++ b/backend/packages/harness/deerflow/persistence/run/model.py @@ -13,37 +13,43 @@ from deerflow.persistence.base import Base class RunRow(Base): __tablename__ = "runs" - run_id: Mapped[str] = mapped_column(String(64), primary_key=True) - thread_id: Mapped[str] = mapped_column(String(64), nullable=False, index=True) - assistant_id: Mapped[str | None] = mapped_column(String(128)) - user_id: Mapped[str | None] = mapped_column(String(64), index=True) - status: Mapped[str] = mapped_column(String(20), default="pending") - # "pending" | "running" | "success" | "error" | "timeout" | "interrupted" + run_id: Mapped[str] = mapped_column(String(64), primary_key=True, comment="运行主键") + thread_id: Mapped[str] = mapped_column(String(64), nullable=False, index=True, comment="所属会话 ID(threads_meta.thread_id)") + assistant_id: Mapped[str | None] = mapped_column(String(128), comment="使用的 Assistant ID(自定义智能体名);为 NULL 表示默认 lead agent") + user_id: Mapped[str | None] = mapped_column(String(64), index=True, comment="发起本次运行的用户 ID") + status: Mapped[str] = mapped_column( + String(20), + default="pending", + comment='运行状态:"pending" / "running" / "success" / "error" / "timeout" / "interrupted"', + ) + model_name: Mapped[str | None] = mapped_column(String(128), comment="本次运行的主模型名(来自 config.yaml.models[*].name)") + multitask_strategy: Mapped[str] = mapped_column( + String(20), + default="reject", + comment='并发策略:同一 thread 已有运行时怎么处理("reject" / "interrupt" / "rollback" / "enqueue")', + ) + metadata_json: Mapped[dict] = mapped_column(JSON, default=dict, comment="运行级元数据(JSON),如 channel/source 等") + kwargs_json: Mapped[dict] = mapped_column(JSON, default=dict, comment="提交运行时的额外参数(JSON),如 thinking_enabled、tool 配置等") + error: Mapped[str | None] = mapped_column(Text, comment="运行失败时的错误文本;成功时为 NULL") - model_name: Mapped[str | None] = mapped_column(String(128)) - multitask_strategy: Mapped[str] = mapped_column(String(20), default="reject") - metadata_json: Mapped[dict] = mapped_column(JSON, default=dict) - kwargs_json: Mapped[dict] = mapped_column(JSON, default=dict) - error: Mapped[str | None] = mapped_column(Text) + message_count: Mapped[int] = mapped_column(default=0, comment="本次运行产生的消息总数(便利字段,避免列表页查 RunEventStore)") + first_human_message: Mapped[str | None] = mapped_column(Text, comment="首条用户消息文本预览(用于列表展示)") + last_ai_message: Mapped[str | None] = mapped_column(Text, comment="末条 AI 消息文本预览(用于列表展示)") - # Convenience fields (for listing pages without querying RunEventStore) - message_count: Mapped[int] = mapped_column(default=0) - first_human_message: Mapped[str | None] = mapped_column(Text) - last_ai_message: Mapped[str | None] = mapped_column(Text) + total_input_tokens: Mapped[int] = mapped_column(default=0, comment="累计输入 token 数(运行结束时由 RunJournal 落盘)") + total_output_tokens: Mapped[int] = mapped_column(default=0, comment="累计输出 token 数") + total_tokens: Mapped[int] = mapped_column(default=0, comment="累计 token 总数 = input + output") + llm_call_count: Mapped[int] = mapped_column(default=0, comment="累计 LLM 调用次数") + lead_agent_tokens: Mapped[int] = mapped_column(default=0, comment="主 agent 自身消耗的 token 数") + subagent_tokens: Mapped[int] = mapped_column(default=0, comment="子 agent(task 工具委派)消耗的 token 数") + middleware_tokens: Mapped[int] = mapped_column(default=0, comment="中间件(如 summarization、title)消耗的 token 数") - # Token usage (accumulated in-memory by RunJournal, written on run completion) - total_input_tokens: Mapped[int] = mapped_column(default=0) - total_output_tokens: Mapped[int] = mapped_column(default=0) - total_tokens: Mapped[int] = mapped_column(default=0) - llm_call_count: Mapped[int] = mapped_column(default=0) - lead_agent_tokens: Mapped[int] = mapped_column(default=0) - subagent_tokens: Mapped[int] = mapped_column(default=0) - middleware_tokens: Mapped[int] = mapped_column(default=0) + follow_up_to_run_id: Mapped[str | None] = mapped_column(String(64), comment="续接的上一次运行 ID(用于'重新生成'/'继续'等链式调用)") - # Follow-up association - follow_up_to_run_id: Mapped[str | None] = mapped_column(String(64)) + created_at: Mapped[datetime] = mapped_column(DateTime(timezone=True), default=lambda: datetime.now(UTC), comment="创建时间(UTC)") + updated_at: Mapped[datetime] = mapped_column(DateTime(timezone=True), default=lambda: datetime.now(UTC), onupdate=lambda: datetime.now(UTC), comment="最近更新时间(UTC,写入时自动更新)") - created_at: Mapped[datetime] = mapped_column(DateTime(timezone=True), default=lambda: datetime.now(UTC)) - updated_at: Mapped[datetime] = mapped_column(DateTime(timezone=True), default=lambda: datetime.now(UTC), onupdate=lambda: datetime.now(UTC)) - - __table_args__ = (Index("ix_runs_thread_status", "thread_id", "status"),) + __table_args__ = ( + Index("ix_runs_thread_status", "thread_id", "status"), + {"comment": "运行(一次完整 agent 执行)的元数据 + 累计 token 指标"}, + ) diff --git a/backend/packages/harness/deerflow/persistence/thread_meta/model.py b/backend/packages/harness/deerflow/persistence/thread_meta/model.py index fe15315e..b32db605 100644 --- a/backend/packages/harness/deerflow/persistence/thread_meta/model.py +++ b/backend/packages/harness/deerflow/persistence/thread_meta/model.py @@ -13,11 +13,13 @@ from deerflow.persistence.base import Base class ThreadMetaRow(Base): __tablename__ = "threads_meta" - thread_id: Mapped[str] = mapped_column(String(64), primary_key=True) - assistant_id: Mapped[str | None] = mapped_column(String(128), index=True) - user_id: Mapped[str | None] = mapped_column(String(64), index=True) - display_name: Mapped[str | None] = mapped_column(String(256)) - status: Mapped[str] = mapped_column(String(20), default="idle") - metadata_json: Mapped[dict] = mapped_column(JSON, default=dict) - created_at: Mapped[datetime] = mapped_column(DateTime(timezone=True), default=lambda: datetime.now(UTC)) - updated_at: Mapped[datetime] = mapped_column(DateTime(timezone=True), default=lambda: datetime.now(UTC), onupdate=lambda: datetime.now(UTC)) + thread_id: Mapped[str] = mapped_column(String(64), primary_key=True, comment="会话主键(LangGraph thread_id)") + assistant_id: Mapped[str | None] = mapped_column(String(128), index=True, comment="关联的 Assistant ID(自定义智能体名);为 NULL 表示默认 lead agent") + user_id: Mapped[str | None] = mapped_column(String(64), index=True, comment="会话所有者;为 NULL 表示历史无主数据") + display_name: Mapped[str | None] = mapped_column(String(256), comment="会话显示名(自动生成的标题或用户手改)") + status: Mapped[str] = mapped_column(String(20), default="idle", comment='会话状态:"idle" 空闲 / "busy" 正在产出') + metadata_json: Mapped[dict] = mapped_column(JSON, default=dict, comment="任意扩展元数据(JSON)") + created_at: Mapped[datetime] = mapped_column(DateTime(timezone=True), default=lambda: datetime.now(UTC), comment="创建时间(UTC)") + updated_at: Mapped[datetime] = mapped_column(DateTime(timezone=True), default=lambda: datetime.now(UTC), onupdate=lambda: datetime.now(UTC), comment="最近更新时间(UTC,写入时自动更新)") + + __table_args__ = ({"comment": "会话元数据(每个 LangGraph thread 的概要信息)"},) diff --git a/backend/packages/harness/deerflow/persistence/user/model.py b/backend/packages/harness/deerflow/persistence/user/model.py index 130d4bfc..6a04c98e 100644 --- a/backend/packages/harness/deerflow/persistence/user/model.py +++ b/backend/packages/harness/deerflow/persistence/user/model.py @@ -22,31 +22,20 @@ from deerflow.persistence.base import Base class UserRow(Base): __tablename__ = "users" - # UUIDs are stored as 36-char strings for cross-backend portability. - id: Mapped[str] = mapped_column(String(36), primary_key=True) - - email: Mapped[str] = mapped_column(String(320), unique=True, nullable=False, index=True) - password_hash: Mapped[str | None] = mapped_column(String(128), nullable=True) - - # "admin" | "user" — kept as plain string to avoid ALTER TABLE pain - # when new roles are introduced. - system_role: Mapped[str] = mapped_column(String(16), nullable=False, default="user") - + id: Mapped[str] = mapped_column(String(36), primary_key=True, comment="用户主键,UUID 字符串(36 字符),跨数据库可移植") + email: Mapped[str] = mapped_column(String(320), unique=True, nullable=False, index=True, comment="登录邮箱,全局唯一") + password_hash: Mapped[str | None] = mapped_column(String(128), nullable=True, comment="本地账户的密码哈希;OAuth-only 用户为 NULL") + system_role: Mapped[str] = mapped_column(String(16), nullable=False, default="user", comment='系统角色:"admin" 或 "user";用字符串以便未来扩展角色而不必 ALTER TABLE') created_at: Mapped[datetime] = mapped_column( DateTime(timezone=True), nullable=False, default=lambda: datetime.now(UTC), + comment="账户创建时间(UTC)", ) - - # OAuth linkage (optional). A partial unique index enforces one - # account per (provider, oauth_id) pair, leaving NULL/NULL rows - # unconstrained so plain password accounts can coexist. - oauth_provider: Mapped[str | None] = mapped_column(String(32), nullable=True) - oauth_id: Mapped[str | None] = mapped_column(String(128), nullable=True) - - # Auth lifecycle flags - needs_setup: Mapped[bool] = mapped_column(Boolean, nullable=False, default=False) - token_version: Mapped[int] = mapped_column(nullable=False, default=0) + oauth_provider: Mapped[str | None] = mapped_column(String(32), nullable=True, comment="OAuth 提供商名(如 google/github);本地账户为 NULL") + oauth_id: Mapped[str | None] = mapped_column(String(128), nullable=True, comment="OAuth 提供商内的用户 ID;与 oauth_provider 组合需唯一") + needs_setup: Mapped[bool] = mapped_column(Boolean, nullable=False, default=False, comment="是否需要完成首次设置(admin 自动创建后改密码/邮箱)") + token_version: Mapped[int] = mapped_column(nullable=False, default=0, comment="JWT 令牌版本号;自增即吊销该用户所有旧令牌") __table_args__ = ( Index( @@ -56,4 +45,5 @@ class UserRow(Base): unique=True, sqlite_where=text("oauth_provider IS NOT NULL AND oauth_id IS NOT NULL"), ), + {"comment": "用户账户表(本地密码登录 + OAuth 联合登录)"}, )