Commit ·
e12cb6b
1
Parent(s): 20197fb
feat: add PostgreSQL schema and Alembic migrations
Browse files- alembic.ini +149 -0
- database.py +35 -1
- models.py +141 -0
- requirements.txt +27 -0
- versioning_bdd/README +1 -0
- versioning_bdd/env.py +54 -0
- versioning_bdd/script.py.mako +28 -0
- versioning_bdd/versions/2d45c40205b1_create_initial_anderson_schema.py +82 -0
alembic.ini
ADDED
|
@@ -0,0 +1,149 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# A generic, single database configuration.
|
| 2 |
+
|
| 3 |
+
[alembic]
|
| 4 |
+
# path to migration scripts.
|
| 5 |
+
# this is typically a path given in POSIX (e.g. forward slashes)
|
| 6 |
+
# format, relative to the token %(here)s which refers to the location of this
|
| 7 |
+
# ini file
|
| 8 |
+
script_location = %(here)s/versioning_bdd
|
| 9 |
+
|
| 10 |
+
# template used to generate migration file names; The default value is %%(rev)s_%%(slug)s
|
| 11 |
+
# Uncomment the line below if you want the files to be prepended with date and time
|
| 12 |
+
# see https://alembic.sqlalchemy.org/en/latest/tutorial.html#editing-the-ini-file
|
| 13 |
+
# for all available tokens
|
| 14 |
+
# file_template = %%(year)d_%%(month).2d_%%(day).2d_%%(hour).2d%%(minute).2d-%%(rev)s_%%(slug)s
|
| 15 |
+
# Or organize into date-based subdirectories (requires recursive_version_locations = true)
|
| 16 |
+
# file_template = %%(year)d/%%(month).2d/%%(day).2d_%%(hour).2d%%(minute).2d_%%(second).2d_%%(rev)s_%%(slug)s
|
| 17 |
+
|
| 18 |
+
# sys.path path, will be prepended to sys.path if present.
|
| 19 |
+
# defaults to the current working directory. for multiple paths, the path separator
|
| 20 |
+
# is defined by "path_separator" below.
|
| 21 |
+
prepend_sys_path = .
|
| 22 |
+
|
| 23 |
+
|
| 24 |
+
# timezone to use when rendering the date within the migration file
|
| 25 |
+
# as well as the filename.
|
| 26 |
+
# If specified, requires the tzdata library which can be installed by adding
|
| 27 |
+
# `alembic[tz]` to the pip requirements.
|
| 28 |
+
# string value is passed to ZoneInfo()
|
| 29 |
+
# leave blank for localtime
|
| 30 |
+
# timezone =
|
| 31 |
+
|
| 32 |
+
# max length of characters to apply to the "slug" field
|
| 33 |
+
# truncate_slug_length = 40
|
| 34 |
+
|
| 35 |
+
# set to 'true' to run the environment during
|
| 36 |
+
# the 'revision' command, regardless of autogenerate
|
| 37 |
+
# revision_environment = false
|
| 38 |
+
|
| 39 |
+
# set to 'true' to allow .pyc and .pyo files without
|
| 40 |
+
# a source .py file to be detected as revisions in the
|
| 41 |
+
# versions/ directory
|
| 42 |
+
# sourceless = false
|
| 43 |
+
|
| 44 |
+
# version location specification; This defaults
|
| 45 |
+
# to <script_location>/versions. When using multiple version
|
| 46 |
+
# directories, initial revisions must be specified with --version-path.
|
| 47 |
+
# The path separator used here should be the separator specified by "path_separator"
|
| 48 |
+
# below.
|
| 49 |
+
# version_locations = %(here)s/bar:%(here)s/bat:%(here)s/alembic/versions
|
| 50 |
+
|
| 51 |
+
# path_separator; This indicates what character is used to split lists of file
|
| 52 |
+
# paths, including version_locations and prepend_sys_path within configparser
|
| 53 |
+
# files such as alembic.ini.
|
| 54 |
+
# The default rendered in new alembic.ini files is "os", which uses os.pathsep
|
| 55 |
+
# to provide os-dependent path splitting.
|
| 56 |
+
#
|
| 57 |
+
# Note that in order to support legacy alembic.ini files, this default does NOT
|
| 58 |
+
# take place if path_separator is not present in alembic.ini. If this
|
| 59 |
+
# option is omitted entirely, fallback logic is as follows:
|
| 60 |
+
#
|
| 61 |
+
# 1. Parsing of the version_locations option falls back to using the legacy
|
| 62 |
+
# "version_path_separator" key, which if absent then falls back to the legacy
|
| 63 |
+
# behavior of splitting on spaces and/or commas.
|
| 64 |
+
# 2. Parsing of the prepend_sys_path option falls back to the legacy
|
| 65 |
+
# behavior of splitting on spaces, commas, or colons.
|
| 66 |
+
#
|
| 67 |
+
# Valid values for path_separator are:
|
| 68 |
+
#
|
| 69 |
+
# path_separator = :
|
| 70 |
+
# path_separator = ;
|
| 71 |
+
# path_separator = space
|
| 72 |
+
# path_separator = newline
|
| 73 |
+
#
|
| 74 |
+
# Use os.pathsep. Default configuration used for new projects.
|
| 75 |
+
path_separator = os
|
| 76 |
+
|
| 77 |
+
# set to 'true' to search source files recursively
|
| 78 |
+
# in each "version_locations" directory
|
| 79 |
+
# new in Alembic version 1.10
|
| 80 |
+
# recursive_version_locations = false
|
| 81 |
+
|
| 82 |
+
# the output encoding used when revision files
|
| 83 |
+
# are written from script.py.mako
|
| 84 |
+
# output_encoding = utf-8
|
| 85 |
+
|
| 86 |
+
# database URL. This is consumed by the user-maintained env.py script only.
|
| 87 |
+
# other means of configuring database URLs may be customized within the env.py
|
| 88 |
+
# file.
|
| 89 |
+
sqlalchemy.url = driver://user:pass@localhost/dbname
|
| 90 |
+
|
| 91 |
+
|
| 92 |
+
[post_write_hooks]
|
| 93 |
+
# post_write_hooks defines scripts or Python functions that are run
|
| 94 |
+
# on newly generated revision scripts. See the documentation for further
|
| 95 |
+
# detail and examples
|
| 96 |
+
|
| 97 |
+
# format using "black" - use the console_scripts runner, against the "black" entrypoint
|
| 98 |
+
# hooks = black
|
| 99 |
+
# black.type = console_scripts
|
| 100 |
+
# black.entrypoint = black
|
| 101 |
+
# black.options = -l 79 REVISION_SCRIPT_FILENAME
|
| 102 |
+
|
| 103 |
+
# lint with attempts to fix using "ruff" - use the module runner, against the "ruff" module
|
| 104 |
+
# hooks = ruff
|
| 105 |
+
# ruff.type = module
|
| 106 |
+
# ruff.module = ruff
|
| 107 |
+
# ruff.options = check --fix REVISION_SCRIPT_FILENAME
|
| 108 |
+
|
| 109 |
+
# Alternatively, use the exec runner to execute a binary found on your PATH
|
| 110 |
+
# hooks = ruff
|
| 111 |
+
# ruff.type = exec
|
| 112 |
+
# ruff.executable = ruff
|
| 113 |
+
# ruff.options = check --fix REVISION_SCRIPT_FILENAME
|
| 114 |
+
|
| 115 |
+
# Logging configuration. This is also consumed by the user-maintained
|
| 116 |
+
# env.py script only.
|
| 117 |
+
[loggers]
|
| 118 |
+
keys = root,sqlalchemy,alembic
|
| 119 |
+
|
| 120 |
+
[handlers]
|
| 121 |
+
keys = console
|
| 122 |
+
|
| 123 |
+
[formatters]
|
| 124 |
+
keys = generic
|
| 125 |
+
|
| 126 |
+
[logger_root]
|
| 127 |
+
level = WARNING
|
| 128 |
+
handlers = console
|
| 129 |
+
qualname =
|
| 130 |
+
|
| 131 |
+
[logger_sqlalchemy]
|
| 132 |
+
level = WARNING
|
| 133 |
+
handlers =
|
| 134 |
+
qualname = sqlalchemy.engine
|
| 135 |
+
|
| 136 |
+
[logger_alembic]
|
| 137 |
+
level = INFO
|
| 138 |
+
handlers =
|
| 139 |
+
qualname = alembic
|
| 140 |
+
|
| 141 |
+
[handler_console]
|
| 142 |
+
class = StreamHandler
|
| 143 |
+
args = (sys.stderr,)
|
| 144 |
+
level = NOTSET
|
| 145 |
+
formatter = generic
|
| 146 |
+
|
| 147 |
+
[formatter_generic]
|
| 148 |
+
format = %(levelname)-5.5s [%(name)s] %(message)s
|
| 149 |
+
datefmt = %H:%M:%S
|
database.py
CHANGED
|
@@ -1,22 +1,31 @@
|
|
| 1 |
import os
|
|
|
|
| 2 |
|
| 3 |
from dotenv import load_dotenv
|
| 4 |
from sqlalchemy import create_engine, text
|
|
|
|
| 5 |
|
|
|
|
|
|
|
|
|
|
| 6 |
load_dotenv()
|
| 7 |
|
|
|
|
| 8 |
database_url = os.getenv("DATABASE_URL")
|
| 9 |
|
| 10 |
if not database_url:
|
| 11 |
raise RuntimeError("La variable DATABASE_URL est absente.")
|
| 12 |
|
| 13 |
-
|
|
|
|
| 14 |
sqlalchemy_database_url = database_url.replace(
|
| 15 |
"postgresql://",
|
| 16 |
"postgresql+psycopg://",
|
| 17 |
1,
|
| 18 |
)
|
| 19 |
|
|
|
|
|
|
|
| 20 |
engine = create_engine(
|
| 21 |
sqlalchemy_database_url,
|
| 22 |
pool_pre_ping=True,
|
|
@@ -25,7 +34,32 @@ engine = create_engine(
|
|
| 25 |
)
|
| 26 |
|
| 27 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 28 |
def check_database_connection() -> bool:
|
|
|
|
|
|
|
| 29 |
with engine.connect() as connection:
|
| 30 |
result = connection.execute(text("SELECT 1"))
|
| 31 |
return result.scalar_one() == 1
|
|
|
|
| 1 |
import os
|
| 2 |
+
from collections.abc import Generator
|
| 3 |
|
| 4 |
from dotenv import load_dotenv
|
| 5 |
from sqlalchemy import create_engine, text
|
| 6 |
+
from sqlalchemy.orm import Session, declarative_base, sessionmaker
|
| 7 |
|
| 8 |
+
|
| 9 |
+
# Charge les variables du fichier .env en développement local.
|
| 10 |
+
# Sur Hugging Face Spaces, DATABASE_URL provient directement des secrets.
|
| 11 |
load_dotenv()
|
| 12 |
|
| 13 |
+
|
| 14 |
database_url = os.getenv("DATABASE_URL")
|
| 15 |
|
| 16 |
if not database_url:
|
| 17 |
raise RuntimeError("La variable DATABASE_URL est absente.")
|
| 18 |
|
| 19 |
+
|
| 20 |
+
# SQLAlchemy doit utiliser explicitement le pilote Psycopg 3.
|
| 21 |
sqlalchemy_database_url = database_url.replace(
|
| 22 |
"postgresql://",
|
| 23 |
"postgresql+psycopg://",
|
| 24 |
1,
|
| 25 |
)
|
| 26 |
|
| 27 |
+
|
| 28 |
+
# Moteur de connexion partagé par l'application et Alembic.
|
| 29 |
engine = create_engine(
|
| 30 |
sqlalchemy_database_url,
|
| 31 |
pool_pre_ping=True,
|
|
|
|
| 34 |
)
|
| 35 |
|
| 36 |
|
| 37 |
+
# Classe de base utilisée par les modèles SQLAlchemy.
|
| 38 |
+
Base = declarative_base()
|
| 39 |
+
|
| 40 |
+
|
| 41 |
+
# Fabrique de sessions utilisée par les endpoints FastAPI.
|
| 42 |
+
SessionLocal = sessionmaker(
|
| 43 |
+
bind=engine,
|
| 44 |
+
autocommit=False,
|
| 45 |
+
autoflush=False,
|
| 46 |
+
)
|
| 47 |
+
|
| 48 |
+
|
| 49 |
+
def get_db() -> Generator[Session, None, None]:
|
| 50 |
+
"""Fournit une session SQLAlchemy puis la ferme après la requête."""
|
| 51 |
+
|
| 52 |
+
database_session = SessionLocal()
|
| 53 |
+
|
| 54 |
+
try:
|
| 55 |
+
yield database_session
|
| 56 |
+
finally:
|
| 57 |
+
database_session.close()
|
| 58 |
+
|
| 59 |
+
|
| 60 |
def check_database_connection() -> bool:
|
| 61 |
+
"""Vérifie que PostgreSQL répond correctement."""
|
| 62 |
+
|
| 63 |
with engine.connect() as connection:
|
| 64 |
result = connection.execute(text("SELECT 1"))
|
| 65 |
return result.scalar_one() == 1
|
models.py
ADDED
|
@@ -0,0 +1,141 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
from datetime import date, datetime, timezone
|
| 2 |
+
|
| 3 |
+
from pgvector.sqlalchemy import Vector
|
| 4 |
+
from sqlalchemy import (
|
| 5 |
+
Boolean,
|
| 6 |
+
Column,
|
| 7 |
+
Date,
|
| 8 |
+
ForeignKey,
|
| 9 |
+
Integer,
|
| 10 |
+
String,
|
| 11 |
+
Text,
|
| 12 |
+
TIMESTAMP,
|
| 13 |
+
UniqueConstraint,
|
| 14 |
+
text,
|
| 15 |
+
)
|
| 16 |
+
from sqlalchemy.orm import relationship
|
| 17 |
+
|
| 18 |
+
from database import Base
|
| 19 |
+
|
| 20 |
+
|
| 21 |
+
# Configuration du modèle d'embeddings utilisé par Anderson.
|
| 22 |
+
EMBEDDING_MODEL = "intfloat/multilingual-e5-small"
|
| 23 |
+
EMBEDDING_DIMENSION = 384
|
| 24 |
+
|
| 25 |
+
|
| 26 |
+
class Collection(Base):
|
| 27 |
+
"""Collection regroupant plusieurs documents associés au même sujet."""
|
| 28 |
+
|
| 29 |
+
__tablename__ = "collections"
|
| 30 |
+
|
| 31 |
+
collection_id = Column(Integer, primary_key=True, index=True)
|
| 32 |
+
name = Column(String(100), nullable=False, unique=True)
|
| 33 |
+
description = Column(Text, nullable=True)
|
| 34 |
+
date_de_creation = Column(Date, nullable=False, default=date.today)
|
| 35 |
+
derniere_modification = Column(
|
| 36 |
+
TIMESTAMP(timezone=True),
|
| 37 |
+
nullable=False,
|
| 38 |
+
default=lambda: datetime.now(timezone.utc),
|
| 39 |
+
onupdate=lambda: datetime.now(timezone.utc),
|
| 40 |
+
)
|
| 41 |
+
|
| 42 |
+
# Les collections temporaires pourront être supprimées après 24 heures.
|
| 43 |
+
is_permanent = Column(Boolean, nullable=False, default=False, server_default=text("false"))
|
| 44 |
+
expires_at = Column(
|
| 45 |
+
TIMESTAMP(timezone=True),
|
| 46 |
+
nullable=False,
|
| 47 |
+
server_default=text("(now() + interval '24 hours')"),
|
| 48 |
+
)
|
| 49 |
+
|
| 50 |
+
# La suppression d'une collection supprime ses documents et leurs chunks.
|
| 51 |
+
documents = relationship(
|
| 52 |
+
"Document",
|
| 53 |
+
back_populates="collection",
|
| 54 |
+
cascade="all, delete-orphan",
|
| 55 |
+
passive_deletes=True,
|
| 56 |
+
)
|
| 57 |
+
|
| 58 |
+
|
| 59 |
+
class Document(Base):
|
| 60 |
+
"""Métadonnées d'un fichier uploadé puis découpé en chunks."""
|
| 61 |
+
|
| 62 |
+
__tablename__ = "documents"
|
| 63 |
+
|
| 64 |
+
# Un même fichier ne peut pas apparaître deux fois dans la même collection.
|
| 65 |
+
__table_args__ = (
|
| 66 |
+
UniqueConstraint(
|
| 67 |
+
"collection_id",
|
| 68 |
+
"title_document",
|
| 69 |
+
name="uq_documents_collection_title_document",
|
| 70 |
+
),
|
| 71 |
+
)
|
| 72 |
+
|
| 73 |
+
document_id = Column(Integer, primary_key=True, index=True)
|
| 74 |
+
collection_id = Column(
|
| 75 |
+
Integer,
|
| 76 |
+
ForeignKey("collections.collection_id", ondelete="CASCADE"),
|
| 77 |
+
nullable=False,
|
| 78 |
+
index=True,
|
| 79 |
+
)
|
| 80 |
+
|
| 81 |
+
# Titre renseigné par l'utilisateur et nom original du fichier.
|
| 82 |
+
title = Column(String(100), nullable=False)
|
| 83 |
+
title_document = Column(String(255), nullable=False)
|
| 84 |
+
|
| 85 |
+
# Le fichier original n'est pas conservé, uniquement ses métadonnées.
|
| 86 |
+
mime_type = Column(String(100), nullable=False)
|
| 87 |
+
size_bytes = Column(Integer, nullable=False)
|
| 88 |
+
|
| 89 |
+
date_de_creation = Column(Date, nullable=False, default=date.today)
|
| 90 |
+
created_at = Column(
|
| 91 |
+
TIMESTAMP(timezone=True),
|
| 92 |
+
nullable=False,
|
| 93 |
+
default=lambda: datetime.now(timezone.utc),
|
| 94 |
+
)
|
| 95 |
+
num_of_chunks = Column(Integer, nullable=False, default=0, server_default=text("0"))
|
| 96 |
+
|
| 97 |
+
collection = relationship("Collection", back_populates="documents")
|
| 98 |
+
chunks = relationship(
|
| 99 |
+
"Chunk",
|
| 100 |
+
back_populates="document",
|
| 101 |
+
cascade="all, delete-orphan",
|
| 102 |
+
passive_deletes=True,
|
| 103 |
+
)
|
| 104 |
+
|
| 105 |
+
|
| 106 |
+
class Chunk(Base):
|
| 107 |
+
"""Fragment de texte vectorisé utilisé pour la recherche sémantique."""
|
| 108 |
+
|
| 109 |
+
__tablename__ = "chunks"
|
| 110 |
+
|
| 111 |
+
# La position permet de reconstruire l'ordre original des fragments.
|
| 112 |
+
__table_args__ = (
|
| 113 |
+
UniqueConstraint(
|
| 114 |
+
"document_id",
|
| 115 |
+
"position",
|
| 116 |
+
name="uq_chunks_document_position",
|
| 117 |
+
),
|
| 118 |
+
)
|
| 119 |
+
|
| 120 |
+
chunk_id = Column(Integer, primary_key=True, index=True)
|
| 121 |
+
document_id = Column(
|
| 122 |
+
Integer,
|
| 123 |
+
ForeignKey("documents.document_id", ondelete="CASCADE"),
|
| 124 |
+
nullable=False,
|
| 125 |
+
index=True,
|
| 126 |
+
)
|
| 127 |
+
|
| 128 |
+
position = Column(Integer, nullable=False)
|
| 129 |
+
chunk_text = Column(Text, nullable=False)
|
| 130 |
+
taille_chunk = Column(Integer, nullable=False)
|
| 131 |
+
|
| 132 |
+
# Un seul modèle léger remplace les trois embeddings de l'ancien projet.
|
| 133 |
+
embedding = Column(Vector(dim=EMBEDDING_DIMENSION), nullable=False)
|
| 134 |
+
|
| 135 |
+
created_at = Column(
|
| 136 |
+
TIMESTAMP(timezone=True),
|
| 137 |
+
nullable=False,
|
| 138 |
+
default=lambda: datetime.now(timezone.utc),
|
| 139 |
+
)
|
| 140 |
+
|
| 141 |
+
document = relationship("Document", back_populates="chunks")
|
requirements.txt
CHANGED
|
@@ -1,3 +1,4 @@
|
|
|
|
|
| 1 |
annotated-doc==0.0.5
|
| 2 |
annotated-types==0.8.0
|
| 3 |
anyio==4.14.2
|
|
@@ -15,16 +16,42 @@ httptools==0.8.0
|
|
| 15 |
httpx==0.28.1
|
| 16 |
huggingface_hub==1.28.0
|
| 17 |
idna==3.19
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 18 |
packaging==26.3
|
|
|
|
| 19 |
psycopg==3.3.4
|
| 20 |
psycopg-binary==3.3.4
|
| 21 |
pydantic==2.13.4
|
| 22 |
pydantic_core==2.46.4
|
|
|
|
| 23 |
python-dotenv==1.2.3
|
| 24 |
PyYAML==6.0.3
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 25 |
SQLAlchemy==2.0.52
|
| 26 |
starlette==1.6.0
|
|
|
|
|
|
|
|
|
|
|
|
|
| 27 |
tqdm==4.70.0
|
|
|
|
|
|
|
| 28 |
typing-inspection==0.4.4
|
| 29 |
typing_extensions==4.16.0
|
| 30 |
tzdata==2026.3
|
|
|
|
| 1 |
+
alembic==1.19.1
|
| 2 |
annotated-doc==0.0.5
|
| 3 |
annotated-types==0.8.0
|
| 4 |
anyio==4.14.2
|
|
|
|
| 16 |
httpx==0.28.1
|
| 17 |
huggingface_hub==1.28.0
|
| 18 |
idna==3.19
|
| 19 |
+
Jinja2==3.1.6
|
| 20 |
+
joblib==1.5.3
|
| 21 |
+
Mako==1.4.1
|
| 22 |
+
markdown-it-py==4.2.0
|
| 23 |
+
MarkupSafe==3.0.3
|
| 24 |
+
mdurl==0.1.2
|
| 25 |
+
mpmath==1.3.0
|
| 26 |
+
narwhals==2.25.0
|
| 27 |
+
networkx==3.6.1
|
| 28 |
+
numpy==2.5.2
|
| 29 |
packaging==26.3
|
| 30 |
+
pgvector==0.5.0
|
| 31 |
psycopg==3.3.4
|
| 32 |
psycopg-binary==3.3.4
|
| 33 |
pydantic==2.13.4
|
| 34 |
pydantic_core==2.46.4
|
| 35 |
+
Pygments==2.21.0
|
| 36 |
python-dotenv==1.2.3
|
| 37 |
PyYAML==6.0.3
|
| 38 |
+
regex==2026.7.19
|
| 39 |
+
rich==15.0.0
|
| 40 |
+
safetensors==0.8.0
|
| 41 |
+
scikit-learn==1.9.0
|
| 42 |
+
scipy==1.18.1
|
| 43 |
+
sentence-transformers==6.0.0
|
| 44 |
+
setuptools==84.0.0
|
| 45 |
+
shellingham==1.5.4
|
| 46 |
SQLAlchemy==2.0.52
|
| 47 |
starlette==1.6.0
|
| 48 |
+
sympy==1.14.0
|
| 49 |
+
threadpoolctl==3.6.0
|
| 50 |
+
tokenizers==0.22.2
|
| 51 |
+
torch==2.13.0
|
| 52 |
tqdm==4.70.0
|
| 53 |
+
transformers==5.15.1
|
| 54 |
+
typer==0.27.1
|
| 55 |
typing-inspection==0.4.4
|
| 56 |
typing_extensions==4.16.0
|
| 57 |
tzdata==2026.3
|
versioning_bdd/README
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
Generic single-database configuration.
|
versioning_bdd/env.py
ADDED
|
@@ -0,0 +1,54 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
from logging.config import fileConfig
|
| 2 |
+
|
| 3 |
+
from alembic import context
|
| 4 |
+
|
| 5 |
+
from database import engine, sqlalchemy_database_url
|
| 6 |
+
from models import Base
|
| 7 |
+
|
| 8 |
+
|
| 9 |
+
# Objet de configuration Alembic provenant de alembic.ini.
|
| 10 |
+
config = context.config
|
| 11 |
+
|
| 12 |
+
|
| 13 |
+
# Configuration des logs Alembic.
|
| 14 |
+
if config.config_file_name is not None:
|
| 15 |
+
fileConfig(config.config_file_name)
|
| 16 |
+
|
| 17 |
+
|
| 18 |
+
# Métadonnées utilisées pour détecter les changements dans models.py.
|
| 19 |
+
target_metadata = Base.metadata
|
| 20 |
+
|
| 21 |
+
|
| 22 |
+
def run_migrations_offline() -> None:
|
| 23 |
+
"""Exécute les migrations sans ouvrir directement de connexion."""
|
| 24 |
+
|
| 25 |
+
context.configure(
|
| 26 |
+
url=sqlalchemy_database_url,
|
| 27 |
+
target_metadata=target_metadata,
|
| 28 |
+
literal_binds=True,
|
| 29 |
+
dialect_opts={"paramstyle": "named"},
|
| 30 |
+
compare_type=True,
|
| 31 |
+
)
|
| 32 |
+
|
| 33 |
+
with context.begin_transaction():
|
| 34 |
+
context.run_migrations()
|
| 35 |
+
|
| 36 |
+
|
| 37 |
+
def run_migrations_online() -> None:
|
| 38 |
+
"""Exécute les migrations en utilisant le moteur SQLAlchemy."""
|
| 39 |
+
|
| 40 |
+
with engine.connect() as connection:
|
| 41 |
+
context.configure(
|
| 42 |
+
connection=connection,
|
| 43 |
+
target_metadata=target_metadata,
|
| 44 |
+
compare_type=True,
|
| 45 |
+
)
|
| 46 |
+
|
| 47 |
+
with context.begin_transaction():
|
| 48 |
+
context.run_migrations()
|
| 49 |
+
|
| 50 |
+
|
| 51 |
+
if context.is_offline_mode():
|
| 52 |
+
run_migrations_offline()
|
| 53 |
+
else:
|
| 54 |
+
run_migrations_online()
|
versioning_bdd/script.py.mako
ADDED
|
@@ -0,0 +1,28 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""${message}
|
| 2 |
+
|
| 3 |
+
Revision ID: ${up_revision}
|
| 4 |
+
Revises: ${down_revision | comma,n}
|
| 5 |
+
Create Date: ${create_date}
|
| 6 |
+
|
| 7 |
+
"""
|
| 8 |
+
from typing import Sequence, Union
|
| 9 |
+
|
| 10 |
+
from alembic import op
|
| 11 |
+
import sqlalchemy as sa
|
| 12 |
+
${imports if imports else ""}
|
| 13 |
+
|
| 14 |
+
# revision identifiers, used by Alembic.
|
| 15 |
+
revision: str = ${repr(up_revision)}
|
| 16 |
+
down_revision: Union[str, Sequence[str], None] = ${repr(down_revision)}
|
| 17 |
+
branch_labels: Union[str, Sequence[str], None] = ${repr(branch_labels)}
|
| 18 |
+
depends_on: Union[str, Sequence[str], None] = ${repr(depends_on)}
|
| 19 |
+
|
| 20 |
+
|
| 21 |
+
def upgrade() -> None:
|
| 22 |
+
"""Upgrade schema."""
|
| 23 |
+
${upgrades if upgrades else "pass"}
|
| 24 |
+
|
| 25 |
+
|
| 26 |
+
def downgrade() -> None:
|
| 27 |
+
"""Downgrade schema."""
|
| 28 |
+
${downgrades if downgrades else "pass"}
|
versioning_bdd/versions/2d45c40205b1_create_initial_anderson_schema.py
ADDED
|
@@ -0,0 +1,82 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""create initial Anderson schema
|
| 2 |
+
|
| 3 |
+
Revision ID: 2d45c40205b1
|
| 4 |
+
Revises:
|
| 5 |
+
Create Date: 2026-08-23 02:24:03.141963
|
| 6 |
+
|
| 7 |
+
"""
|
| 8 |
+
from typing import Sequence, Union
|
| 9 |
+
|
| 10 |
+
from alembic import op
|
| 11 |
+
import sqlalchemy as sa
|
| 12 |
+
from pgvector.sqlalchemy import Vector
|
| 13 |
+
|
| 14 |
+
|
| 15 |
+
# revision identifiers, used by Alembic.
|
| 16 |
+
revision: str = '2d45c40205b1'
|
| 17 |
+
down_revision: Union[str, Sequence[str], None] = None
|
| 18 |
+
branch_labels: Union[str, Sequence[str], None] = None
|
| 19 |
+
depends_on: Union[str, Sequence[str], None] = None
|
| 20 |
+
|
| 21 |
+
|
| 22 |
+
def upgrade() -> None:
|
| 23 |
+
"""Upgrade schema."""
|
| 24 |
+
op.execute("CREATE EXTENSION IF NOT EXISTS vector")
|
| 25 |
+
# ### commands auto generated by Alembic - please adjust! ###
|
| 26 |
+
op.create_table('collections',
|
| 27 |
+
sa.Column('collection_id', sa.Integer(), nullable=False),
|
| 28 |
+
sa.Column('name', sa.String(length=100), nullable=False),
|
| 29 |
+
sa.Column('description', sa.Text(), nullable=True),
|
| 30 |
+
sa.Column('date_de_creation', sa.Date(), nullable=False),
|
| 31 |
+
sa.Column('derniere_modification', sa.TIMESTAMP(timezone=True), nullable=False),
|
| 32 |
+
sa.Column('is_permanent', sa.Boolean(), server_default=sa.text('false'), nullable=False),
|
| 33 |
+
sa.Column('expires_at', sa.TIMESTAMP(timezone=True), server_default=sa.text("(now() + interval '24 hours')"), nullable=False),
|
| 34 |
+
sa.PrimaryKeyConstraint('collection_id'),
|
| 35 |
+
sa.UniqueConstraint('name')
|
| 36 |
+
)
|
| 37 |
+
op.create_index(op.f('ix_collections_collection_id'), 'collections', ['collection_id'], unique=False)
|
| 38 |
+
op.create_table('documents',
|
| 39 |
+
sa.Column('document_id', sa.Integer(), nullable=False),
|
| 40 |
+
sa.Column('collection_id', sa.Integer(), nullable=False),
|
| 41 |
+
sa.Column('title', sa.String(length=100), nullable=False),
|
| 42 |
+
sa.Column('title_document', sa.String(length=255), nullable=False),
|
| 43 |
+
sa.Column('mime_type', sa.String(length=100), nullable=False),
|
| 44 |
+
sa.Column('size_bytes', sa.Integer(), nullable=False),
|
| 45 |
+
sa.Column('date_de_creation', sa.Date(), nullable=False),
|
| 46 |
+
sa.Column('created_at', sa.TIMESTAMP(timezone=True), nullable=False),
|
| 47 |
+
sa.Column('num_of_chunks', sa.Integer(), server_default=sa.text('0'), nullable=False),
|
| 48 |
+
sa.ForeignKeyConstraint(['collection_id'], ['collections.collection_id'], ondelete='CASCADE'),
|
| 49 |
+
sa.PrimaryKeyConstraint('document_id'),
|
| 50 |
+
sa.UniqueConstraint('collection_id', 'title_document', name='uq_documents_collection_title_document')
|
| 51 |
+
)
|
| 52 |
+
op.create_index(op.f('ix_documents_collection_id'), 'documents', ['collection_id'], unique=False)
|
| 53 |
+
op.create_index(op.f('ix_documents_document_id'), 'documents', ['document_id'], unique=False)
|
| 54 |
+
op.create_table('chunks',
|
| 55 |
+
sa.Column('chunk_id', sa.Integer(), nullable=False),
|
| 56 |
+
sa.Column('document_id', sa.Integer(), nullable=False),
|
| 57 |
+
sa.Column('position', sa.Integer(), nullable=False),
|
| 58 |
+
sa.Column('chunk_text', sa.Text(), nullable=False),
|
| 59 |
+
sa.Column('taille_chunk', sa.Integer(), nullable=False),
|
| 60 |
+
sa.Column('embedding', Vector(dim=384), nullable=False),
|
| 61 |
+
sa.Column('created_at', sa.TIMESTAMP(timezone=True), nullable=False),
|
| 62 |
+
sa.ForeignKeyConstraint(['document_id'], ['documents.document_id'], ondelete='CASCADE'),
|
| 63 |
+
sa.PrimaryKeyConstraint('chunk_id'),
|
| 64 |
+
sa.UniqueConstraint('document_id', 'position', name='uq_chunks_document_position')
|
| 65 |
+
)
|
| 66 |
+
op.create_index(op.f('ix_chunks_chunk_id'), 'chunks', ['chunk_id'], unique=False)
|
| 67 |
+
op.create_index(op.f('ix_chunks_document_id'), 'chunks', ['document_id'], unique=False)
|
| 68 |
+
# ### end Alembic commands ###
|
| 69 |
+
|
| 70 |
+
|
| 71 |
+
def downgrade() -> None:
|
| 72 |
+
"""Downgrade schema."""
|
| 73 |
+
# ### commands auto generated by Alembic - please adjust! ###
|
| 74 |
+
op.drop_index(op.f('ix_chunks_document_id'), table_name='chunks')
|
| 75 |
+
op.drop_index(op.f('ix_chunks_chunk_id'), table_name='chunks')
|
| 76 |
+
op.drop_table('chunks')
|
| 77 |
+
op.drop_index(op.f('ix_documents_document_id'), table_name='documents')
|
| 78 |
+
op.drop_index(op.f('ix_documents_collection_id'), table_name='documents')
|
| 79 |
+
op.drop_table('documents')
|
| 80 |
+
op.drop_index(op.f('ix_collections_collection_id'), table_name='collections')
|
| 81 |
+
op.drop_table('collections')
|
| 82 |
+
# ### end Alembic commands ###
|