From 7ae43f7b0bf2eaec1a03ba8f5f41e95686d7723a Mon Sep 17 00:00:00 2001 From: Xavier Date: Fri, 28 Aug 2026 18:35:30 +0200 Subject: [PATCH] =?UTF-8?q?Mise=20a=20jour=20module=20=C3=A9duction=20nati?= =?UTF-8?q?onale?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- modules/vector_rag.py | 59 +++++++++++++++++++++++++++++++++++++++++++ requirements.txt | 5 +++- 2 files changed, 63 insertions(+), 1 deletion(-) create mode 100644 modules/vector_rag.py diff --git a/modules/vector_rag.py b/modules/vector_rag.py new file mode 100644 index 0000000..9677862 --- /dev/null +++ b/modules/vector_rag.py @@ -0,0 +1,59 @@ +import os +import pypdf +from docx import Document +import chromadb +from modules.logger import log_event + +RAG_BASE_DIR = "/DATA/AppData/MULTI-IA-AETHAS38/RAG" +VECTOR_DB_DIR = "/DATA/AppData/MULTI-IA-AETHAS38/VectorDB" + +def initialiser_chroma(): + """Initialise le client ChromaDB en mode persistant sur le RAID.""" + os.makedirs(VECTOR_DB_DIR, exist_ok=True) + client = chromadb.PersistentClient(path=VECTOR_DB_DIR) + return client + +def extraire_texte_fichier(chemin_fichier): + """Extrait le texte brut d'un PDF, DOCX ou MD/TXT.""" + texte = "" + bas_chemin = chemin_fichier.lower() + try: + if bas_chemin.endswith('.pdf'): + reader = pypdf.PdfReader(chemin_fichier) + for page in reader.pages: + texte += page.extract_text() or "" + elif bas_chemin.endswith('.docx'): + doc = Document(chemin_fichier) + texte = "\n".join([para.text for para in doc.paragraphs]) + elif bas_chemin.endswith(('.txt', '.md')): + with open(chemin_fichier, 'r', encoding='utf-8') as f: + texte = f.read() + except Exception as e: + log_event("ERROR", "VECTOR_RAG", fogli="Erreur extraction {chemin_fichier} : {e}") + return texte + +def indexer_dossier_thematique(thematique: str): + """Parcourt une thématique, découpe et indexe les documents dans ChromaDB.""" + client = initialiser_chroma() + collection = client.get_or_create_collection(name=thematique.lower()) + + chemin_theme = os.path.join(RAG_BASE_DIR, thematique) + if not os.path.exists(chemin_theme): + return + + for fichier in os.listdir(chemin_theme): + chemin_complet = os.path.join(chemin_theme, fichier) + if os.path.isfile(chemin_complet): + texte = extraire_texte_fichier(chemin_complet) + if len(texte.strip()) > 50: + # Découpage simple par blocs de 1000 caractères (Chunks) + taille_chunk = 1000 + chunks = [texte[i:i+taille_chunk] for i in range(0, len(texte), taille_chunk)] + + ids = [f"{fichier}_chunk_{idx}" for idx in range(len(chunks))] + + collection.upsert( + documents=chunks, + ids=ids + ) + log_event("INFO", "VECTOR_RAG", f"Indexé {len(chunks)} blocs pour {fichier} dans {thematique}.") \ No newline at end of file diff --git a/requirements.txt b/requirements.txt index 123aebe..f3ecf50 100644 --- a/requirements.txt +++ b/requirements.txt @@ -11,4 +11,7 @@ reportlab psutil cryptography bcrypt -tiktoken \ No newline at end of file +tiktoken +pypdf==4.1.0 +python-docx==1.1.0 +chromadb==0.4.24 \ No newline at end of file