Mise a jour module éduction nationale
Build and Push Docker Image / build-and-push (push) Failing after 48s

This commit is contained in:
xavier committed 2026-08-28 18:35:30 +02:00
1 parent fbc38fbb19
commit 7ae43f7b0b
2 files changed
+63 -1

No files matched your search

+59
View File
@@ -0,0 +1,59 @@
import os
import pypdf
from docx import Document
import chromadb
from modules.logger import log_event
RAG_BASE_DIR = "/DATA/AppData/MULTI-IA-AETHAS38/RAG"
VECTOR_DB_DIR = "/DATA/AppData/MULTI-IA-AETHAS38/VectorDB"
def initialiser_chroma():
"""Initialise le client ChromaDB en mode persistant sur le RAID."""
os.makedirs(VECTOR_DB_DIR, exist_ok=True)
client = chromadb.PersistentClient(path=VECTOR_DB_DIR)
return client
def extraire_texte_fichier(chemin_fichier):
"""Extrait le texte brut d'un PDF, DOCX ou MD/TXT."""
texte = ""
bas_chemin = chemin_fichier.lower()
try:
if bas_chemin.endswith('.pdf'):
reader = pypdf.PdfReader(chemin_fichier)
for page in reader.pages:
texte += page.extract_text() or ""
elif bas_chemin.endswith('.docx'):
doc = Document(chemin_fichier)
texte = "\n".join([para.text for para in doc.paragraphs])
elif bas_chemin.endswith(('.txt', '.md')):
with open(chemin_fichier, 'r', encoding='utf-8') as f:
texte = f.read()
except Exception as e:
log_event("ERROR", "VECTOR_RAG", fogli="Erreur extraction {chemin_fichier} : {e}")
return texte
def indexer_dossier_thematique(thematique: str):
"""Parcourt une thématique, découpe et indexe les documents dans ChromaDB."""
client = initialiser_chroma()
collection = client.get_or_create_collection(name=thematique.lower())
chemin_theme = os.path.join(RAG_BASE_DIR, thematique)
if not os.path.exists(chemin_theme):
return
for fichier in os.listdir(chemin_theme):
chemin_complet = os.path.join(chemin_theme, fichier)
if os.path.isfile(chemin_complet):
texte = extraire_texte_fichier(chemin_complet)
if len(texte.strip()) > 50:
# Découpage simple par blocs de 1000 caractères (Chunks)
taille_chunk = 1000
chunks = [texte[i:i+taille_chunk] for i in range(0, len(texte), taille_chunk)]
ids = [f"{fichier}_chunk_{idx}" for idx in range(len(chunks))]
collection.upsert(
documents=chunks,
ids=ids
)
log_event("INFO", "VECTOR_RAG", f"Indexé {len(chunks)} blocs pour {fichier} dans {thematique}.")
+4 -1
View File
@@ -11,4 +11,7 @@ reportlab
psutil psutil
cryptography cryptography
bcrypt bcrypt
tiktoken tiktoken
pypdf==4.1.0
python-docx==1.1.0
chromadb==0.4.24