第二步

This commit is contained in:
amb
2026-09-01 12:24:43 +08:00
parent 47bf6cc5ca
commit 1d8621717a
14 changed files with 815 additions and 9 deletions
+88
View File
@@ -0,0 +1,88 @@
# Alembic Configuration File
[alembic]
# path to migration scripts
script_location = alembic
# template used to generate migration file names; the default
# file_template = %%(rev)s_%%(slug)s
# sys.path path, will be prepended to sys.path if present.
# defaults to the current working directory.
prepend_sys_path = .
# timezone to use when rendering the date within the migration file
# as well as the filename.
# timezone =
# max length of characters to apply to the "slug" field
# truncate_slug_length = 40
# set to 'true' to run the environment during
# the 'revision' command, regardless of autogenerate
# revision_environment = false
# set to 'true' to allow .pot files to be overwritten
# generate_empty_script = false
# python function to use for producing template names
# file_template = %%(rev)s_%%(slug)s
# the output encoding used when revision files
# are written from script.py.mako
# output_encoding = utf-8
sqlalchemy.url = sqlite:///./data/app.db
[post_write_hooks]
# post_write_hooks defines scripts or Python functions that are run
# on newly generated revision scripts. See the documentation for further
# detail and examples
# format using "black" - use the console_scripts runner, against the "black" entrypoint
# hooks = black
# black.type = console_scripts
# black.entrypoint = black
# black.options = -l 79
# lint with attempts to fix using "ruff" - use the exec runner, execute a binary
# hooks = ruff
# ruff.type = exec
# ruff.executable = %(here)s/.venv/bin/ruff
# ruff.options = check --fix REVISION_SCRIPT_FILENAME
# Logging configuration
[loggers]
keys = root,sqlalchemy,alembic
[handlers]
keys = console
[formatters]
keys = generic
[logger_root]
level = WARN
handlers = console
qualname =
[logger_sqlalchemy]
level = WARN
handlers =
qualname = sqlalchemy.engine
[logger_alembic]
level = INFO
handlers =
qualname = alembic
[handler_console]
class = StreamHandler
args = (sys.stderr,)
level = NOTSET
formatter = generic
[formatter_generic]
format = %(levelname)-5.5s [%(name)s] %(message)s
datefmt = %H:%M:%S
+78
View File
@@ -0,0 +1,78 @@
"""Alembic 环境配置。
import app.models 以注册所有 ORM 模型到 Base.metadata
使 alembic revision --autogenerate 能检测模型变更。
"""
from logging.config import fileConfig
from alembic import context
from sqlalchemy import engine_from_config, pool
from app.core.config import get_settings
from app.models import Base # noqa: F401 — 触发全部模型注册
# this is the Alembic Config object
config = context.config
# 根据应用配置覆盖 sqlalchemy.url
settings = get_settings()
config.set_main_option("sqlalchemy.url", settings.database_url)
# Interpret the config file for Python logging.
if config.config_file_name is not None:
fileConfig(config.config_file_name)
# target metadata for autogenerate
target_metadata = Base.metadata
def run_migrations_offline() -> None:
"""Run migrations in 'offline' mode.
This configures the context with just a URL
and not an Engine, though an Engine is acceptable
here as well. By skipping the Engine creation
we don't even need a DBAPI to be available.
Calls to context.execute() here emit the given string to the
script output.
"""
url = config.get_main_option("sqlalchemy.url")
context.configure(
url=url,
target_metadata=target_metadata,
literal_binds=True,
dialect_opts={"paramstyle": "named"},
)
with context.begin_transaction():
context.run_migrations()
def run_migrations_online() -> None:
"""Run migrations in 'online' mode.
In this scenario we need to create an Engine
and associate a connection with the context.
"""
connectable = engine_from_config(
config.get_section(config.config_ini_section, {}),
prefix="sqlalchemy.",
poolclass=pool.NullPool,
)
with connectable.connect() as connection:
context.configure(
connection=connection,
target_metadata=target_metadata,
)
with context.begin_transaction():
context.run_migrations()
if context.is_offline_mode():
run_migrations_offline()
else:
run_migrations_online()
+25
View File
@@ -0,0 +1,25 @@
"""${message}
Revision ID: ${up_revision}
Revises: ${down_revision | comma,n}
Create Date: ${create_date}
"""
from typing import Sequence, Union
from alembic import op
import sqlalchemy as sa
${imports if imports else ""}
# revision identifiers, used by Alembic.
revision: str = ${repr(up_revision)}
down_revision: Union[str, None] = ${repr(down_revision)}
branch_labels: Union[str, Sequence[str], None] = ${repr(branch_labels)}
depends_on: Union[str, Sequence[str], None] = ${repr(depends_on)}
def upgrade() -> None:
${upgrades if upgrades else "pass"}
def downgrade() -> None:
${downgrades if downgrades else "pass"}
View File
@@ -0,0 +1,142 @@
"""initial_schema
Revision ID: 5c27c8e31fde
Revises:
Create Date: 2026-09-01 12:22:22.337100
"""
from typing import Sequence, Union
from alembic import op
import sqlalchemy as sa
# revision identifiers, used by Alembic.
revision: str = '5c27c8e31fde'
down_revision: Union[str, None] = None
branch_labels: Union[str, Sequence[str], None] = None
depends_on: Union[str, Sequence[str], None] = None
def upgrade() -> None:
# ### commands auto generated by Alembic - please adjust! ###
op.create_table('plans',
sa.Column('code', sa.String(length=32), nullable=False, comment='套餐代码 (free/basic/pro)'),
sa.Column('name', sa.String(length=64), nullable=False, comment='套餐名称'),
sa.Column('storage_quota', sa.Integer(), nullable=False, comment='存储配额 (字节)'),
sa.Column('max_file_size', sa.Integer(), nullable=False, comment='单文件大小上限 (字节)'),
sa.Column('is_active', sa.Boolean(), nullable=False, comment='是否可用'),
sa.Column('id', sa.String(length=32), nullable=False, comment='UUID4 十六进制主键'),
sa.Column('created_at', sa.String(length=32), nullable=False, comment='创建时间 (ISO8601)'),
sa.Column('updated_at', sa.String(length=32), nullable=False, comment='更新时间 (ISO8601)'),
sa.PrimaryKeyConstraint('id'),
sa.UniqueConstraint('code')
)
op.create_table('users',
sa.Column('username', sa.String(length=64), nullable=False, comment='用户名'),
sa.Column('email', sa.String(length=255), nullable=False, comment='邮箱'),
sa.Column('password_hash', sa.String(length=255), nullable=False, comment='Argon2id 密码哈希'),
sa.Column('status', sa.String(length=16), nullable=False, comment='状态 (active/disabled)'),
sa.Column('plan_id', sa.String(length=32), nullable=False, comment='套餐 ID'),
sa.Column('storage_used', sa.Integer(), nullable=False, comment='已用存储 (字节)'),
sa.Column('id', sa.String(length=32), nullable=False, comment='UUID4 十六进制主键'),
sa.Column('created_at', sa.String(length=32), nullable=False, comment='创建时间 (ISO8601)'),
sa.Column('updated_at', sa.String(length=32), nullable=False, comment='更新时间 (ISO8601)'),
sa.ForeignKeyConstraint(['plan_id'], ['plans.id'], ),
sa.PrimaryKeyConstraint('id')
)
op.create_index(op.f('ix_users_email'), 'users', ['email'], unique=True)
op.create_index(op.f('ix_users_username'), 'users', ['username'], unique=True)
op.create_table('knowledge_bases',
sa.Column('user_id', sa.String(length=32), nullable=False, comment='所有者用户 ID'),
sa.Column('name', sa.String(length=255), nullable=False, comment='知识库名称'),
sa.Column('description', sa.Text(), nullable=True, comment='知识库描述'),
sa.Column('enabled', sa.Boolean(), nullable=False, comment='是否启用'),
sa.Column('token_hash', sa.String(length=64), nullable=False, comment='SHA-256(secret_token) 十六进制'),
sa.Column('token_encrypted', sa.Text(), nullable=True, comment='Fernet 加密的 token 原文'),
sa.Column('token_hint', sa.String(length=16), nullable=True, comment='token 末 8 位明文,供后台识别'),
sa.Column('id', sa.String(length=32), nullable=False, comment='UUID4 十六进制主键'),
sa.Column('created_at', sa.String(length=32), nullable=False, comment='创建时间 (ISO8601)'),
sa.Column('updated_at', sa.String(length=32), nullable=False, comment='更新时间 (ISO8601)'),
sa.ForeignKeyConstraint(['user_id'], ['users.id'], ),
sa.PrimaryKeyConstraint('id')
)
op.create_index(op.f('ix_knowledge_bases_token_hash'), 'knowledge_bases', ['token_hash'], unique=True)
op.create_index(op.f('ix_knowledge_bases_user_id'), 'knowledge_bases', ['user_id'], unique=False)
op.create_table('document_categories',
sa.Column('knowledge_base_id', sa.String(length=32), nullable=False, comment='所属知识库 ID'),
sa.Column('name', sa.String(length=255), nullable=False, comment='分类名称'),
sa.Column('sort_order', sa.Integer(), nullable=False, comment='排序序号'),
sa.Column('id', sa.String(length=32), nullable=False, comment='UUID4 十六进制主键'),
sa.Column('created_at', sa.String(length=32), nullable=False, comment='创建时间 (ISO8601)'),
sa.Column('updated_at', sa.String(length=32), nullable=False, comment='更新时间 (ISO8601)'),
sa.ForeignKeyConstraint(['knowledge_base_id'], ['knowledge_bases.id'], ),
sa.PrimaryKeyConstraint('id')
)
op.create_index(op.f('ix_document_categories_knowledge_base_id'), 'document_categories', ['knowledge_base_id'], unique=False)
op.create_table('documents',
sa.Column('knowledge_base_id', sa.String(length=32), nullable=False, comment='所属知识库 ID'),
sa.Column('user_id', sa.String(length=32), nullable=False, comment='所有者用户 ID (冗余,便于隔离校验)'),
sa.Column('category_id', sa.String(length=32), nullable=True, comment='分类 ID'),
sa.Column('original_filename', sa.String(length=512), nullable=False, comment='原始文件名 (仅展示)'),
sa.Column('storage_path', sa.String(length=1024), nullable=False, comment='相对于 data/ 的物理存储路径'),
sa.Column('markdown_path', sa.String(length=1024), nullable=True, comment='相对于 data/ 的 Markdown 文件路径'),
sa.Column('file_size', sa.Integer(), nullable=False, comment='文件大小 (字节)'),
sa.Column('mime_type', sa.String(length=127), nullable=False, comment='MIME 类型'),
sa.Column('file_ext', sa.String(length=16), nullable=False, comment='文件扩展名 (.docx/.pdf)'),
sa.Column('sha256', sa.String(length=64), nullable=False, comment='文件 SHA-256 哈希'),
sa.Column('doc_token_hash', sa.String(length=64), nullable=True, comment='SHA-256(document_token) 十六进制'),
sa.Column('doc_token_encrypted', sa.Text(), nullable=True, comment='Fernet 加密的 document_token 原文'),
sa.Column('doc_token_hint', sa.String(length=16), nullable=True, comment='document_token 末 8 位明文'),
sa.Column('title', sa.String(length=512), nullable=True, comment='文档标题 (用户可改)'),
sa.Column('description', sa.Text(), nullable=True, comment='文档描述 (用户可改)'),
sa.Column('keywords', sa.Text(), nullable=True, comment='关键词 (逗号分隔)'),
sa.Column('content_summary', sa.Text(), nullable=True, comment='抽取式摘要 (~200字)'),
sa.Column('status', sa.String(length=16), nullable=False, comment='状态 (PENDING/PROCESSING/READY/FAILED/DELETED)'),
sa.Column('error_code', sa.String(length=64), nullable=True, comment='错误码 (如 SCANNED_PDF_NO_TEXT_LAYER)'),
sa.Column('id', sa.String(length=32), nullable=False, comment='UUID4 十六进制主键'),
sa.Column('created_at', sa.String(length=32), nullable=False, comment='创建时间 (ISO8601)'),
sa.Column('updated_at', sa.String(length=32), nullable=False, comment='更新时间 (ISO8601)'),
sa.ForeignKeyConstraint(['category_id'], ['document_categories.id'], ),
sa.ForeignKeyConstraint(['knowledge_base_id'], ['knowledge_bases.id'], ),
sa.ForeignKeyConstraint(['user_id'], ['users.id'], ),
sa.PrimaryKeyConstraint('id')
)
op.create_index(op.f('ix_documents_doc_token_hash'), 'documents', ['doc_token_hash'], unique=True)
op.create_index(op.f('ix_documents_knowledge_base_id'), 'documents', ['knowledge_base_id'], unique=False)
op.create_index(op.f('ix_documents_user_id'), 'documents', ['user_id'], unique=False)
op.create_table('access_logs',
sa.Column('knowledge_base_id', sa.String(length=32), nullable=False, comment='知识库 ID'),
sa.Column('document_id', sa.String(length=32), nullable=True, comment='文档 ID (可选)'),
sa.Column('path', sa.String(length=1024), nullable=False, comment='请求路径'),
sa.Column('accessed_at', sa.String(length=32), nullable=False, comment='访问时间 (ISO8601)'),
sa.Column('user_agent', sa.Text(), nullable=True, comment='User-Agent'),
sa.Column('request_type', sa.String(length=32), nullable=True, comment='请求类型 (html/md/txt/json/search)'),
sa.Column('id', sa.String(length=32), nullable=False, comment='UUID4 十六进制主键'),
sa.Column('created_at', sa.String(length=32), nullable=False, comment='创建时间 (ISO8601)'),
sa.Column('updated_at', sa.String(length=32), nullable=False, comment='更新时间 (ISO8601)'),
sa.ForeignKeyConstraint(['document_id'], ['documents.id'], ),
sa.ForeignKeyConstraint(['knowledge_base_id'], ['knowledge_bases.id'], ),
sa.PrimaryKeyConstraint('id')
)
op.create_index(op.f('ix_access_logs_knowledge_base_id'), 'access_logs', ['knowledge_base_id'], unique=False)
# ### end Alembic commands ###
def downgrade() -> None:
# ### commands auto generated by Alembic - please adjust! ###
op.drop_index(op.f('ix_access_logs_knowledge_base_id'), table_name='access_logs')
op.drop_table('access_logs')
op.drop_index(op.f('ix_documents_user_id'), table_name='documents')
op.drop_index(op.f('ix_documents_knowledge_base_id'), table_name='documents')
op.drop_index(op.f('ix_documents_doc_token_hash'), table_name='documents')
op.drop_table('documents')
op.drop_index(op.f('ix_document_categories_knowledge_base_id'), table_name='document_categories')
op.drop_table('document_categories')
op.drop_index(op.f('ix_knowledge_bases_user_id'), table_name='knowledge_bases')
op.drop_index(op.f('ix_knowledge_bases_token_hash'), table_name='knowledge_bases')
op.drop_table('knowledge_bases')
op.drop_index(op.f('ix_users_username'), table_name='users')
op.drop_index(op.f('ix_users_email'), table_name='users')
op.drop_table('users')
op.drop_table('plans')
# ### end Alembic commands ###
+21
View File
@@ -0,0 +1,21 @@
"""SQLAlchemy 2.0 模型包。"""
from app.models.base import Base, generate_uuid, utcnow_iso
from app.models.plan import Plan
from app.models.user import User
from app.models.knowledge_base import KnowledgeBase
from app.models.document_category import DocumentCategory
from app.models.document import Document
from app.models.access_log import AccessLog
__all__ = [
"Base",
"generate_uuid",
"utcnow_iso",
"Plan",
"User",
"KnowledgeBase",
"DocumentCategory",
"Document",
"AccessLog",
]
+56
View File
@@ -0,0 +1,56 @@
"""AccessLog 访问日志模型。"""
from sqlalchemy import ForeignKey, Integer, String, Text
from sqlalchemy.orm import Mapped, mapped_column, relationship
from app.models.base import Base, TimestampMixin, UUIDPrimaryKeyMixin
class AccessLog(UUIDPrimaryKeyMixin, TimestampMixin, Base):
"""访问日志模型。
技术审查 §2.2access_logs 表。
"""
__tablename__ = "access_logs"
knowledge_base_id: Mapped[str] = mapped_column(
String(32),
ForeignKey("knowledge_bases.id"),
nullable=False,
index=True,
comment="知识库 ID",
)
document_id: Mapped[str | None] = mapped_column(
String(32),
ForeignKey("documents.id"),
nullable=True,
comment="文档 ID (可选)",
)
path: Mapped[str] = mapped_column(
String(1024),
nullable=False,
comment="请求路径",
)
accessed_at: Mapped[str] = mapped_column(
String(32),
nullable=False,
comment="访问时间 (ISO8601)",
)
user_agent: Mapped[str | None] = mapped_column(
Text,
nullable=True,
comment="User-Agent",
)
request_type: Mapped[str | None] = mapped_column(
String(32),
nullable=True,
comment="请求类型 (html/md/txt/json/search)",
)
# 关系
knowledge_base = relationship("KnowledgeBase", back_populates="access_logs", lazy="selectin")
document = relationship("Document", back_populates="access_logs", lazy="selectin")
def __repr__(self) -> str:
return f"<AccessLog path={self.path!r} at={self.accessed_at!r}>"
+52
View File
@@ -0,0 +1,52 @@
"""SQLAlchemy 2.0 基础模型类与 Mixin。"""
import uuid
from datetime import datetime, timezone
from sqlalchemy import String, text
from sqlalchemy.orm import DeclarativeBase, Mapped, mapped_column
def generate_uuid() -> str:
"""生成 UUID4 十六进制字符串(32字符)。"""
return uuid.uuid4().hex
def utcnow_iso() -> str:
"""返回 UTC 当前时间的 ISO8601 字符串。"""
return datetime.now(timezone.utc).isoformat()
class Base(DeclarativeBase):
"""SQLAlchemy 声明式基类。"""
pass
class UUIDPrimaryKeyMixin:
"""UUID 主键 Mixin。"""
id: Mapped[str] = mapped_column(
String(32),
primary_key=True,
default=generate_uuid,
comment="UUID4 十六进制主键",
)
class TimestampMixin:
"""创建/更新时间 Mixin。"""
created_at: Mapped[str] = mapped_column(
String(32),
default=utcnow_iso,
nullable=False,
comment="创建时间 (ISO8601)",
)
updated_at: Mapped[str] = mapped_column(
String(32),
default=utcnow_iso,
onupdate=utcnow_iso,
nullable=False,
comment="更新时间 (ISO8601)",
)
+128
View File
@@ -0,0 +1,128 @@
"""Document 文档模型。"""
from sqlalchemy import ForeignKey, Integer, String, Text
from sqlalchemy.orm import Mapped, mapped_column, relationship
from app.models.base import Base, TimestampMixin, UUIDPrimaryKeyMixin
class Document(UUIDPrimaryKeyMixin, TimestampMixin, Base):
"""文档模型。
技术审查 §2.2documents 表。
"""
__tablename__ = "documents"
knowledge_base_id: Mapped[str] = mapped_column(
String(32),
ForeignKey("knowledge_bases.id"),
nullable=False,
index=True,
comment="所属知识库 ID",
)
user_id: Mapped[str] = mapped_column(
String(32),
ForeignKey("users.id"),
nullable=False,
index=True,
comment="所有者用户 ID (冗余,便于隔离校验)",
)
category_id: Mapped[str | None] = mapped_column(
String(32),
ForeignKey("document_categories.id"),
nullable=True,
comment="分类 ID",
)
original_filename: Mapped[str] = mapped_column(
String(512),
nullable=False,
comment="原始文件名 (仅展示)",
)
storage_path: Mapped[str] = mapped_column(
String(1024),
nullable=False,
comment="相对于 data/ 的物理存储路径",
)
markdown_path: Mapped[str | None] = mapped_column(
String(1024),
nullable=True,
comment="相对于 data/ 的 Markdown 文件路径",
)
file_size: Mapped[int] = mapped_column(
Integer,
nullable=False,
comment="文件大小 (字节)",
)
mime_type: Mapped[str] = mapped_column(
String(127),
nullable=False,
comment="MIME 类型",
)
file_ext: Mapped[str] = mapped_column(
String(16),
nullable=False,
comment="文件扩展名 (.docx/.pdf)",
)
sha256: Mapped[str] = mapped_column(
String(64),
nullable=False,
comment="文件 SHA-256 哈希",
)
doc_token_hash: Mapped[str] = mapped_column(
String(64),
unique=True,
nullable=True,
index=True,
comment="SHA-256(document_token) 十六进制",
)
doc_token_encrypted: Mapped[str | None] = mapped_column(
Text,
nullable=True,
comment="Fernet 加密的 document_token 原文",
)
doc_token_hint: Mapped[str | None] = mapped_column(
String(16),
nullable=True,
comment="document_token 末 8 位明文",
)
title: Mapped[str | None] = mapped_column(
String(512),
nullable=True,
comment="文档标题 (用户可改)",
)
description: Mapped[str | None] = mapped_column(
Text,
nullable=True,
comment="文档描述 (用户可改)",
)
keywords: Mapped[str | None] = mapped_column(
Text,
nullable=True,
comment="关键词 (逗号分隔)",
)
content_summary: Mapped[str | None] = mapped_column(
Text,
nullable=True,
comment="抽取式摘要 (~200字)",
)
status: Mapped[str] = mapped_column(
String(16),
default="PENDING",
nullable=False,
comment="状态 (PENDING/PROCESSING/READY/FAILED/DELETED)",
)
error_code: Mapped[str | None] = mapped_column(
String(64),
nullable=True,
comment="错误码 (如 SCANNED_PDF_NO_TEXT_LAYER)",
)
# 关系
knowledge_base = relationship("KnowledgeBase", back_populates="documents", lazy="selectin")
user = relationship("User", back_populates="documents", lazy="selectin")
category = relationship("DocumentCategory", back_populates="documents", lazy="selectin")
access_logs = relationship("AccessLog", back_populates="document", lazy="selectin")
def __repr__(self) -> str:
return f"<Document {self.original_filename!r} (status={self.status!r})>"
+41
View File
@@ -0,0 +1,41 @@
"""DocumentCategory 文档分类模型。"""
from sqlalchemy import ForeignKey, Integer, String
from sqlalchemy.orm import Mapped, mapped_column, relationship
from app.models.base import Base, TimestampMixin, UUIDPrimaryKeyMixin
class DocumentCategory(UUIDPrimaryKeyMixin, TimestampMixin, Base):
"""文档分类模型。
技术审查 §2.2document_categories 表。
"""
__tablename__ = "document_categories"
knowledge_base_id: Mapped[str] = mapped_column(
String(32),
ForeignKey("knowledge_bases.id"),
nullable=False,
index=True,
comment="所属知识库 ID",
)
name: Mapped[str] = mapped_column(
String(255),
nullable=False,
comment="分类名称",
)
sort_order: Mapped[int] = mapped_column(
Integer,
default=0,
nullable=False,
comment="排序序号",
)
# 关系
knowledge_base = relationship("KnowledgeBase", back_populates="categories", lazy="selectin")
documents = relationship("Document", back_populates="category", lazy="selectin")
def __repr__(self) -> str:
return f"<DocumentCategory {self.name!r} (kb={self.knowledge_base_id!r})>"
+65
View File
@@ -0,0 +1,65 @@
"""KnowledgeBase 知识库模型。"""
from sqlalchemy import Boolean, ForeignKey, String, Text
from sqlalchemy.orm import Mapped, mapped_column, relationship
from app.models.base import Base, TimestampMixin, UUIDPrimaryKeyMixin
class KnowledgeBase(UUIDPrimaryKeyMixin, TimestampMixin, Base):
"""知识库模型。
技术审查 §2.2knowledge_bases 表。
"""
__tablename__ = "knowledge_bases"
user_id: Mapped[str] = mapped_column(
String(32),
ForeignKey("users.id"),
nullable=False,
index=True,
comment="所有者用户 ID",
)
name: Mapped[str] = mapped_column(
String(255),
nullable=False,
comment="知识库名称",
)
description: Mapped[str | None] = mapped_column(
Text,
nullable=True,
comment="知识库描述",
)
enabled: Mapped[bool] = mapped_column(
Boolean,
default=True,
nullable=False,
comment="是否启用",
)
token_hash: Mapped[str] = mapped_column(
String(64),
unique=True,
nullable=False,
index=True,
comment="SHA-256(secret_token) 十六进制",
)
token_encrypted: Mapped[str | None] = mapped_column(
Text,
nullable=True,
comment="Fernet 加密的 token 原文",
)
token_hint: Mapped[str | None] = mapped_column(
String(16),
nullable=True,
comment="token 末 8 位明文,供后台识别",
)
# 关系
user = relationship("User", back_populates="knowledge_bases", lazy="selectin")
documents = relationship("Document", back_populates="knowledge_base", lazy="selectin")
categories = relationship("DocumentCategory", back_populates="knowledge_base", lazy="selectin")
access_logs = relationship("AccessLog", back_populates="knowledge_base", lazy="selectin")
def __repr__(self) -> str:
return f"<KnowledgeBase {self.name!r} (user={self.user_id!r})>"
+49
View File
@@ -0,0 +1,49 @@
"""Plan 套餐模型。"""
from sqlalchemy import Boolean, Integer, String, Text
from sqlalchemy.orm import Mapped, mapped_column, relationship
from app.models.base import Base, TimestampMixin, UUIDPrimaryKeyMixin
class Plan(UUIDPrimaryKeyMixin, TimestampMixin, Base):
"""用户套餐(免费/基础/专业等)。
技术审查 §2.2:plans 表。
"""
__tablename__ = "plans"
code: Mapped[str] = mapped_column(
String(32),
unique=True,
nullable=False,
comment="套餐代码 (free/basic/pro)",
)
name: Mapped[str] = mapped_column(
String(64),
nullable=False,
comment="套餐名称",
)
storage_quota: Mapped[int] = mapped_column(
Integer,
nullable=False,
comment="存储配额 (字节)",
)
max_file_size: Mapped[int] = mapped_column(
Integer,
nullable=False,
comment="单文件大小上限 (字节)",
)
is_active: Mapped[bool] = mapped_column(
Boolean,
default=True,
nullable=False,
comment="是否可用",
)
# 关系
users = relationship("User", back_populates="plan", lazy="selectin")
def __repr__(self) -> str:
return f"<Plan {self.code!r} ({self.name!r})>"
+61
View File
@@ -0,0 +1,61 @@
"""User 用户模型。"""
from sqlalchemy import ForeignKey, Integer, String, Text
from sqlalchemy.orm import Mapped, mapped_column, relationship
from app.models.base import Base, TimestampMixin, UUIDPrimaryKeyMixin
class User(UUIDPrimaryKeyMixin, TimestampMixin, Base):
"""用户模型。
技术审查 §2.2:users 表。
"""
__tablename__ = "users"
username: Mapped[str] = mapped_column(
String(64),
unique=True,
nullable=False,
index=True,
comment="用户名",
)
email: Mapped[str] = mapped_column(
String(255),
unique=True,
nullable=False,
index=True,
comment="邮箱",
)
password_hash: Mapped[str] = mapped_column(
String(255),
nullable=False,
comment="Argon2id 密码哈希",
)
status: Mapped[str] = mapped_column(
String(16),
default="active",
nullable=False,
comment="状态 (active/disabled)",
)
plan_id: Mapped[str] = mapped_column(
String(32),
ForeignKey("plans.id"),
nullable=False,
comment="套餐 ID",
)
storage_used: Mapped[int] = mapped_column(
Integer,
default=0,
nullable=False,
comment="已用存储 (字节)",
)
# 关系
plan = relationship("Plan", back_populates="users", lazy="selectin")
knowledge_bases = relationship("KnowledgeBase", back_populates="user", lazy="selectin")
documents = relationship("Document", back_populates="user", lazy="selectin")
def __repr__(self) -> str:
return f"<User {self.username!r}>"
+9 -9
View File
@@ -11,7 +11,7 @@
| Phase | 内容 | 状态 |
|---|---|---|
| 1 | 项目初始化 | ✅ 完成 |
| 2 | 数据库和 Alembic | |
| 2 | 数据库和 Alembic | ✅ 完成 |
| 3 | 用户注册登录 | ⬜ |
| 4 | 知识库 CRUD | ⬜ |
| 5 | 本地 StorageService | ✅ 完成(提前实现于 Phase 1) |
@@ -30,7 +30,7 @@
---
## 一、产品核心目标 ⬜(产品定位,随各 Phase 逐步实现)
## 一、产品核心目标 🔄 产品定位,随各 Phase 逐步实现Phase 1-2 已完成
这个产品不是普通网盘。
@@ -66,7 +66,7 @@
生产环境:Ubuntu、Docker、Docker Compose。
## 四、第一版系统架构 ⬜(架构已定实现随 Phase 推进
## 四、第一版系统架构 🔄 架构已定 + 数据库层 ✅ 实现随 Phase 推进
采用:
@@ -89,7 +89,7 @@ SQLite 保存元数据。文件系统保存:Word、PDF、Markdown。
禁止把原始 PDF、Word 二进制内容存进 SQLite。SQLite 只保存:用户信息、知识库信息、文档元数据、文件路径、Markdown 路径、文档状态、Token hash、时间、分类、描述、关键词、搜索相关数据。原始文件与 Markdown 均存服务器文件系统。
## 六、数据库访问必须抽象 Phase 2Phase 1 已建 db.py 引擎层
## 六、数据库访问必须抽象 ✅ 完成(Phase 2ORM 模型 + Alembic 迁移已落地
不能在业务代码中到处直接调用 sqlite3。必须使用 SQLAlchemy,并通过 Repository / Service 分层(UserRepository、KnowledgeBaseRepository、DocumentRepository)。业务逻辑不能依赖 SQLite 具体实现。数据库配置通过 `DATABASE_URL`(如 `sqlite:///./data/app.db`),未来可以切换 PostgreSQL 而尽量不修改业务层。
@@ -123,7 +123,7 @@ SQLite 保存元数据。文件系统保存:Word、PDF、Markdown。
Userid、username、email、password_hash、status、storage_quota、storage_used、created_at、updated_at。密码必须安全哈希,优先 Argon2id;禁止明文密码、MD5、SHA1。
## 十四、存储限制 🔄 部分完成(config.py 已有配额配置Plan 模型 Phase 2 落地
## 十四、存储限制 完成(config.py 配额配置 + Plan 模型已落地 Phase 2
默认免费用户 100MB,单文件 20MB。不要硬编码,放到配置或 Plan 模型。第一版 Free Plan100MB / 20MB-per-file),未来 Basic 1GB、Pro 5GB、Enterprise 自定义。第一版不实现支付。
@@ -192,7 +192,7 @@ data/
第一版不强制 OCR。如果 PDF 没有文本层,提示"该 PDF 可能是扫描件,当前版本暂不支持 OCR。"未来预留 OCRProcessorPaddleOCR 等),第一版不加入 OCR。
## 二十六、文档状态 ⬜ Phase 2(状态模型)/ Phase 7(流转)
## 二十六、文档状态 🔄 状态模型 ✅ 完成(Phase 2);流转 ⬜ Phase 7
PENDING、PROCESSING、READY、FAILED、DELETED。虽然第一版同步处理,但状态模型必须保留,未来异步任务可以直接使用。
@@ -248,11 +248,11 @@ Vue 3、TypeScript、Vite、Element Plus、Pinia。UI 要求:现代、简洁
第一版不需要 Redis,可以使用内存限流(如单 IP 每分钟一定次数),保护公共 AI URL。注意内存限流只适合单实例 MVP,代码中抽象 RateLimiter,未来可实现 RedisRateLimiter。(已实现 `core/rate_limit.py`:内存 TokenBucket + 配置化阈值。)
## 四十、访问日志 Phase 2(模型)/ Phase 10(记录)
## 四十、访问日志 🔄 模型 ✅ 完成(Phase 2);记录 ⬜ Phase 10
第一版可简单记录:knowledge_base_id、document_id、timestamp、user_agent、请求类型。不要默认长期保存完整 IP。后台可显示访问次数、最近访问时间。不要声称可以准确判断访问者是不是 AI,使用"外部访问"而不是"AI 访问"。
## 四十一、数据库模型 Phase 2
## 四十一、数据库模型 ✅ 完成(Phase 2User/Plan/KnowledgeBase/DocumentCategory/Document/AccessLog 全部落地)
至少:User、Plan、KnowledgeBase、Document、DocumentCategory、AccessLog。
@@ -260,7 +260,7 @@ Documentid、knowledge_base_id、user_id、document_token_hash、original_fil
KnowledgeBaseid、user_id、name、description、secret_token_hash、enabled、created_at、updated_at。
## 四十二、数据库迁移 Phase 2
## 四十二、数据库迁移 ✅ 完成(Phase 2:Alembic 初始化 + 初始迁移 initial_schema 已应用)
使用 Alembic,即使 SQLite 也必须使用迁移。不要手工修改生产数据库。README 提供:初始化迁移、升级迁移、回滚迁移。