import os import json import pypdf import faiss import numpy as np import requests RAG_BASE_DIR = "/DATA/AppData/MULTI-IA-AETHAS38/RAG" VECTOR_DB_DIR = "/DATA/AppData/MULTI-IA-AETHAS38/VectorDB" # L'IP 172.19.0.1 correspond à la passerelle Docker pour joindre Ollama sur ton NAS OLLAMA_URL = "http://172.19.0.1:11434/api/embeddings" EMBEDDING_MODEL = "nomic-embed-text" def obtenir_embedding(texte): """Transforme un texte en vecteur mathématique via Ollama local.""" try: reponse = requests.post(OLLAMA_URL, json={ "model": EMBEDDING_MODEL, "prompt": texte }) reponse.raise_for_status() return reponse.json()["embedding"] except Exception as e: print(f"Erreur d'embedding avec Ollama : {e}") return None def indexer_dossier_thematique(thematique: str): print(f"--- DÉBUT DE L'INDEXATION FAISS : {thematique} ---") chemin_theme = os.path.join(RAG_BASE_DIR, thematique) dossier_db = os.path.join(VECTOR_DB_DIR, thematique.lower()) os.makedirs(dossier_db, exist_ok=True) chunks = [] vecteurs = [] for fichier in os.listdir(chemin_theme): chemin_complet = os.path.join(chemin_theme, fichier) if not os.path.isfile(chemin_complet): continue print(f"Extraction de : {fichier}") texte = "" try: if fichier.lower().endswith('.pdf'): reader = pypdf.PdfReader(chemin_complet) for page in reader.pages: texte += page.extract_text() or "" elif fichier.lower().endswith(('.md', '.txt')): with open(chemin_complet, 'r', encoding='utf-8') as f: texte = f.read() if len(texte) > 50: # Découpage en blocs de 1000 caractères taille_chunk = 1000 blocs = [texte[i:i+taille_chunk] for i in range(0, len(texte), taille_chunk)] print(f" -> Vectorisation de {len(blocs)} blocs via Ollama...") for bloc in blocs: vecteur = obtenir_embedding(bloc) if vecteur: # On sauvegarde le texte d'un côté, et le vecteur de l'autre chunks.append({"fichier": fichier, "texte": bloc}) vecteurs.append(vecteur) except Exception as e: print(f" -> Erreur sur {fichier} : {e}") if vecteurs: # Conversion des vecteurs pour FAISS vecteurs_np = np.array(vecteurs).astype('float32') dimension = vecteurs_np.shape[1] index = faiss.IndexFlatL2(dimension) index.add(vecteurs_np) # Sauvegarde sur le RAID : l'index mathématique (.faiss) et le texte (.json) faiss.write_index(index, os.path.join(dossier_db, "index.faiss")) with open(os.path.join(dossier_db, "chunks.json"), 'w', encoding='utf-8') as f: json.dump(chunks, f, ensure_ascii=False, indent=2) print(f"--- SUCCÈS : {len(chunks)} blocs vectorisés et stockés dans FAISS ! ---") else: print("--- ÉCHEC : Aucun vecteur généré. ---") def rechercher_contexte_vectoriel(thematique: str, question: str, top_k: int = 3): """Recherche les extraits de texte pertinents pour répondre à une question.""" dossier_db = os.path.join(VECTOR_DB_DIR, thematique.lower()) fichier_index = os.path.join(dossier_db, "index.faiss") fichier_chunks = os.path.join(dossier_db, "chunks.json") if not os.path.exists(fichier_index) or not os.path.exists(fichier_chunks): return "" try: index = faiss.read_index(fichier_index) with open(fichier_chunks, 'r', encoding='utf-8') as f: chunks = json.load(f) # On vectorise la question avec le même modèle Ollama vecteur_q = obtenir_embedding(question) if not vecteur_q: return "" vecteur_q_np = np.array([vecteur_q]).astype('float32') # On demande à FAISS les K plus proches voisins (top_k) distances, indices = index.search(vecteur_q_np, top_k) fragments = [] for i in indices[0]: if i < len(chunks) and i != -1: fragments.append(chunks[i]["texte"]) if fragments: return "\n\n[...]\n\n".join(fragments) except Exception as e: print(f"Erreur de recherche FAISS : {e}") return "" def indexer_tout_le_rag(): """Parcourt tous les sous-dossiers de RAG_BASE_DIR et les indexe automatiquement.""" print("=== LANCEMENT DE L'INDEXATION GLOBALE ===") if not os.path.exists(RAG_BASE_DIR): print(f"Dossier introuvable : {RAG_BASE_DIR}") return for element in os.listdir(RAG_BASE_DIR): chemin = os.path.join(RAG_BASE_DIR, element) if os.path.isdir(chemin): # Appelle la fonction existante pour chaque dossier trouvé indexer_dossier_thematique(element) print("=== INDEXATION GLOBALE TERMINÉE ===") def rechercher_contexte_global(question: str, top_k: int = 3): """Cherche la réponse dans TOUTES les thématiques indexées et garde les meilleurs extraits.""" if not os.path.exists(VECTOR_DB_DIR): return "" vecteur_q = obtenir_embedding(question) if not vecteur_q: return "" vecteur_q_np = np.array([vecteur_q]).astype('float32') tous_fragments = [] # On fouille dans chaque dossier de la base vectorielle for thematique in os.listdir(VECTOR_DB_DIR): dossier_db = os.path.join(VECTOR_DB_DIR, thematique) fichier_index = os.path.join(dossier_db, "index.faiss") fichier_chunks = os.path.join(dossier_db, "chunks.json") if os.path.exists(fichier_index) and os.path.exists(fichier_chunks): try: index = faiss.read_index(fichier_index) with open(fichier_chunks, 'r', encoding='utf-8') as f: chunks = json.load(f) # Recherche dans ce dossier spécifique distances, indices = index.search(vecteur_q_np, top_k) # On stocke les résultats avec leur score de pertinence (distance) for dist, i in zip(distances[0], indices[0]): if i < len(chunks) and i != -1: tous_fragments.append((dist, chunks[i]["texte"], thematique)) except Exception: continue if not tous_fragments: return "" # On trie TOUS les fragments trouvés pour ne garder que les (top_k) les plus pertinents, tous dossiers confondus # (FAISS utilise la distance L2 : plus le score est petit, plus le texte correspond à la question) tous_fragments.sort(key=lambda x: x[0]) meilleurs = tous_fragments[:top_k] contexte_final = [] for dist, texte, theme in meilleurs: contexte_final.append(f"[Source : Dossier {theme}]\n{texte}") return "\n\n---\n\n".join(contexte_final)