Public Access
Mise a jour module éduction nationale
Build and Push Docker Image / build-and-push (push) Failing after 48s
Build and Push Docker Image / build-and-push (push) Failing after 48s
This commit is contained in:
1 parent
fbc38fbb19
commit
7ae43f7b0b
2 files changed
+63
-1
No files matched your search
@@ -0,0 +1,59 @@
|
|||||||
|
import os
|
||||||
|
import pypdf
|
||||||
|
from docx import Document
|
||||||
|
import chromadb
|
||||||
|
from modules.logger import log_event
|
||||||
|
|
||||||
|
RAG_BASE_DIR = "/DATA/AppData/MULTI-IA-AETHAS38/RAG"
|
||||||
|
VECTOR_DB_DIR = "/DATA/AppData/MULTI-IA-AETHAS38/VectorDB"
|
||||||
|
|
||||||
|
def initialiser_chroma():
|
||||||
|
"""Initialise le client ChromaDB en mode persistant sur le RAID."""
|
||||||
|
os.makedirs(VECTOR_DB_DIR, exist_ok=True)
|
||||||
|
client = chromadb.PersistentClient(path=VECTOR_DB_DIR)
|
||||||
|
return client
|
||||||
|
|
||||||
|
def extraire_texte_fichier(chemin_fichier):
|
||||||
|
"""Extrait le texte brut d'un PDF, DOCX ou MD/TXT."""
|
||||||
|
texte = ""
|
||||||
|
bas_chemin = chemin_fichier.lower()
|
||||||
|
try:
|
||||||
|
if bas_chemin.endswith('.pdf'):
|
||||||
|
reader = pypdf.PdfReader(chemin_fichier)
|
||||||
|
for page in reader.pages:
|
||||||
|
texte += page.extract_text() or ""
|
||||||
|
elif bas_chemin.endswith('.docx'):
|
||||||
|
doc = Document(chemin_fichier)
|
||||||
|
texte = "\n".join([para.text for para in doc.paragraphs])
|
||||||
|
elif bas_chemin.endswith(('.txt', '.md')):
|
||||||
|
with open(chemin_fichier, 'r', encoding='utf-8') as f:
|
||||||
|
texte = f.read()
|
||||||
|
except Exception as e:
|
||||||
|
log_event("ERROR", "VECTOR_RAG", fogli="Erreur extraction {chemin_fichier} : {e}")
|
||||||
|
return texte
|
||||||
|
|
||||||
|
def indexer_dossier_thematique(thematique: str):
|
||||||
|
"""Parcourt une thématique, découpe et indexe les documents dans ChromaDB."""
|
||||||
|
client = initialiser_chroma()
|
||||||
|
collection = client.get_or_create_collection(name=thematique.lower())
|
||||||
|
|
||||||
|
chemin_theme = os.path.join(RAG_BASE_DIR, thematique)
|
||||||
|
if not os.path.exists(chemin_theme):
|
||||||
|
return
|
||||||
|
|
||||||
|
for fichier in os.listdir(chemin_theme):
|
||||||
|
chemin_complet = os.path.join(chemin_theme, fichier)
|
||||||
|
if os.path.isfile(chemin_complet):
|
||||||
|
texte = extraire_texte_fichier(chemin_complet)
|
||||||
|
if len(texte.strip()) > 50:
|
||||||
|
# Découpage simple par blocs de 1000 caractères (Chunks)
|
||||||
|
taille_chunk = 1000
|
||||||
|
chunks = [texte[i:i+taille_chunk] for i in range(0, len(texte), taille_chunk)]
|
||||||
|
|
||||||
|
ids = [f"{fichier}_chunk_{idx}" for idx in range(len(chunks))]
|
||||||
|
|
||||||
|
collection.upsert(
|
||||||
|
documents=chunks,
|
||||||
|
ids=ids
|
||||||
|
)
|
||||||
|
log_event("INFO", "VECTOR_RAG", f"Indexé {len(chunks)} blocs pour {fichier} dans {thematique}.")
|
||||||
+4
-1
@@ -11,4 +11,7 @@ reportlab
|
|||||||
psutil
|
psutil
|
||||||
cryptography
|
cryptography
|
||||||
bcrypt
|
bcrypt
|
||||||
tiktoken
|
tiktoken
|
||||||
|
pypdf==4.1.0
|
||||||
|
python-docx==1.1.0
|
||||||
|
chromadb==0.4.24
|
||||||
Reference in new issue
Block a user