diff --git a/modules/vector_rag.py b/modules/vector_rag.py index 9677862..b4e6d4c 100644 --- a/modules/vector_rag.py +++ b/modules/vector_rag.py @@ -1,59 +1,121 @@ import os +import json import pypdf -from docx import Document -import chromadb -from modules.logger import log_event +import faiss +import numpy as np +import requests RAG_BASE_DIR = "/DATA/AppData/MULTI-IA-AETHAS38/RAG" VECTOR_DB_DIR = "/DATA/AppData/MULTI-IA-AETHAS38/VectorDB" -def initialiser_chroma(): - """Initialise le client ChromaDB en mode persistant sur le RAID.""" - os.makedirs(VECTOR_DB_DIR, exist_ok=True) - client = chromadb.PersistentClient(path=VECTOR_DB_DIR) - return client +# L'IP 172.19.0.1 correspond à la passerelle Docker pour joindre Ollama sur ton NAS +OLLAMA_URL = "http://172.19.0.1:11434/api/embeddings" +EMBEDDING_MODEL = "nomic-embed-text" -def extraire_texte_fichier(chemin_fichier): - """Extrait le texte brut d'un PDF, DOCX ou MD/TXT.""" - texte = "" - bas_chemin = chemin_fichier.lower() +def obtenir_embedding(texte): + """Transforme un texte en vecteur mathématique via Ollama local.""" try: - if bas_chemin.endswith('.pdf'): - reader = pypdf.PdfReader(chemin_fichier) - for page in reader.pages: - texte += page.extract_text() or "" - elif bas_chemin.endswith('.docx'): - doc = Document(chemin_fichier) - texte = "\n".join([para.text for para in doc.paragraphs]) - elif bas_chemin.endswith(('.txt', '.md')): - with open(chemin_fichier, 'r', encoding='utf-8') as f: - texte = f.read() + reponse = requests.post(OLLAMA_URL, json={ + "model": EMBEDDING_MODEL, + "prompt": texte + }) + reponse.raise_for_status() + return reponse.json()["embedding"] except Exception as e: - log_event("ERROR", "VECTOR_RAG", fogli="Erreur extraction {chemin_fichier} : {e}") - return texte + print(f"Erreur d'embedding avec Ollama : {e}") + return None def indexer_dossier_thematique(thematique: str): - """Parcourt une thématique, découpe et indexe les documents dans ChromaDB.""" - client = initialiser_chroma() - collection = client.get_or_create_collection(name=thematique.lower()) - + print(f"--- DÉBUT DE L'INDEXATION FAISS : {thematique} ---") chemin_theme = os.path.join(RAG_BASE_DIR, thematique) - if not os.path.exists(chemin_theme): - return - + dossier_db = os.path.join(VECTOR_DB_DIR, thematique.lower()) + os.makedirs(dossier_db, exist_ok=True) + + chunks = [] + vecteurs = [] + for fichier in os.listdir(chemin_theme): chemin_complet = os.path.join(chemin_theme, fichier) - if os.path.isfile(chemin_complet): - texte = extraire_texte_fichier(chemin_complet) - if len(texte.strip()) > 50: - # Découpage simple par blocs de 1000 caractères (Chunks) + if not os.path.isfile(chemin_complet): + continue + + print(f"Extraction de : {fichier}") + texte = "" + try: + if fichier.lower().endswith('.pdf'): + reader = pypdf.PdfReader(chemin_complet) + for page in reader.pages: + texte += page.extract_text() or "" + elif fichier.lower().endswith(('.md', '.txt')): + with open(chemin_complet, 'r', encoding='utf-8') as f: + texte = f.read() + + if len(texte) > 50: + # Découpage en blocs de 1000 caractères taille_chunk = 1000 - chunks = [texte[i:i+taille_chunk] for i in range(0, len(texte), taille_chunk)] + blocs = [texte[i:i+taille_chunk] for i in range(0, len(texte), taille_chunk)] - ids = [f"{fichier}_chunk_{idx}" for idx in range(len(chunks))] + print(f" -> Vectorisation de {len(blocs)} blocs via Ollama...") + for bloc in blocs: + vecteur = obtenir_embedding(bloc) + if vecteur: + # On sauvegarde le texte d'un côté, et le vecteur de l'autre + chunks.append({"fichier": fichier, "texte": bloc}) + vecteurs.append(vecteur) + + except Exception as e: + print(f" -> Erreur sur {fichier} : {e}") + + if vecteurs: + # Conversion des vecteurs pour FAISS + vecteurs_np = np.array(vecteurs).astype('float32') + dimension = vecteurs_np.shape[1] + + index = faiss.IndexFlatL2(dimension) + index.add(vecteurs_np) + + # Sauvegarde sur le RAID : l'index mathématique (.faiss) et le texte (.json) + faiss.write_index(index, os.path.join(dossier_db, "index.faiss")) + with open(os.path.join(dossier_db, "chunks.json"), 'w', encoding='utf-8') as f: + json.dump(chunks, f, ensure_ascii=False, indent=2) + + print(f"--- SUCCÈS : {len(chunks)} blocs vectorisés et stockés dans FAISS ! ---") + else: + print("--- ÉCHEC : Aucun vecteur généré. ---") + +def rechercher_contexte_vectoriel(thematique: str, question: str, top_k: int = 3): + """Recherche les extraits de texte pertinents pour répondre à une question.""" + dossier_db = os.path.join(VECTOR_DB_DIR, thematique.lower()) + fichier_index = os.path.join(dossier_db, "index.faiss") + fichier_chunks = os.path.join(dossier_db, "chunks.json") + + if not os.path.exists(fichier_index) or not os.path.exists(fichier_chunks): + return "" + + try: + index = faiss.read_index(fichier_index) + with open(fichier_chunks, 'r', encoding='utf-8') as f: + chunks = json.load(f) + + # On vectorise la question avec le même modèle Ollama + vecteur_q = obtenir_embedding(question) + if not vecteur_q: + return "" + + vecteur_q_np = np.array([vecteur_q]).astype('float32') + + # On demande à FAISS les K plus proches voisins (top_k) + distances, indices = index.search(vecteur_q_np, top_k) + + fragments = [] + for i in indices[0]: + if i < len(chunks) and i != -1: + fragments.append(chunks[i]["texte"]) - collection.upsert( - documents=chunks, - ids=ids - ) - log_event("INFO", "VECTOR_RAG", f"Indexé {len(chunks)} blocs pour {fichier} dans {thematique}.") \ No newline at end of file + if fragments: + return "\n\n[...]\n\n".join(fragments) + + except Exception as e: + print(f"Erreur de recherche FAISS : {e}") + + return "" \ No newline at end of file