Public Access
121 lines
4.5 KiB
Python
121 lines
4.5 KiB
Python
import os
|
|
import json
|
|
import pypdf
|
|
import faiss
|
|
import numpy as np
|
|
import requests
|
|
|
|
RAG_BASE_DIR = "/DATA/AppData/MULTI-IA-AETHAS38/RAG"
|
|
VECTOR_DB_DIR = "/DATA/AppData/MULTI-IA-AETHAS38/VectorDB"
|
|
|
|
# L'IP 172.19.0.1 correspond à la passerelle Docker pour joindre Ollama sur ton NAS
|
|
OLLAMA_URL = "http://172.19.0.1:11434/api/embeddings"
|
|
EMBEDDING_MODEL = "nomic-embed-text"
|
|
|
|
def obtenir_embedding(texte):
|
|
"""Transforme un texte en vecteur mathématique via Ollama local."""
|
|
try:
|
|
reponse = requests.post(OLLAMA_URL, json={
|
|
"model": EMBEDDING_MODEL,
|
|
"prompt": texte
|
|
})
|
|
reponse.raise_for_status()
|
|
return reponse.json()["embedding"]
|
|
except Exception as e:
|
|
print(f"Erreur d'embedding avec Ollama : {e}")
|
|
return None
|
|
|
|
def indexer_dossier_thematique(thematique: str):
|
|
print(f"--- DÉBUT DE L'INDEXATION FAISS : {thematique} ---")
|
|
chemin_theme = os.path.join(RAG_BASE_DIR, thematique)
|
|
dossier_db = os.path.join(VECTOR_DB_DIR, thematique.lower())
|
|
os.makedirs(dossier_db, exist_ok=True)
|
|
|
|
chunks = []
|
|
vecteurs = []
|
|
|
|
for fichier in os.listdir(chemin_theme):
|
|
chemin_complet = os.path.join(chemin_theme, fichier)
|
|
if not os.path.isfile(chemin_complet):
|
|
continue
|
|
|
|
print(f"Extraction de : {fichier}")
|
|
texte = ""
|
|
try:
|
|
if fichier.lower().endswith('.pdf'):
|
|
reader = pypdf.PdfReader(chemin_complet)
|
|
for page in reader.pages:
|
|
texte += page.extract_text() or ""
|
|
elif fichier.lower().endswith(('.md', '.txt')):
|
|
with open(chemin_complet, 'r', encoding='utf-8') as f:
|
|
texte = f.read()
|
|
|
|
if len(texte) > 50:
|
|
# Découpage en blocs de 1000 caractères
|
|
taille_chunk = 1000
|
|
blocs = [texte[i:i+taille_chunk] for i in range(0, len(texte), taille_chunk)]
|
|
|
|
print(f" -> Vectorisation de {len(blocs)} blocs via Ollama...")
|
|
for bloc in blocs:
|
|
vecteur = obtenir_embedding(bloc)
|
|
if vecteur:
|
|
# On sauvegarde le texte d'un côté, et le vecteur de l'autre
|
|
chunks.append({"fichier": fichier, "texte": bloc})
|
|
vecteurs.append(vecteur)
|
|
|
|
except Exception as e:
|
|
print(f" -> Erreur sur {fichier} : {e}")
|
|
|
|
if vecteurs:
|
|
# Conversion des vecteurs pour FAISS
|
|
vecteurs_np = np.array(vecteurs).astype('float32')
|
|
dimension = vecteurs_np.shape[1]
|
|
|
|
index = faiss.IndexFlatL2(dimension)
|
|
index.add(vecteurs_np)
|
|
|
|
# Sauvegarde sur le RAID : l'index mathématique (.faiss) et le texte (.json)
|
|
faiss.write_index(index, os.path.join(dossier_db, "index.faiss"))
|
|
with open(os.path.join(dossier_db, "chunks.json"), 'w', encoding='utf-8') as f:
|
|
json.dump(chunks, f, ensure_ascii=False, indent=2)
|
|
|
|
print(f"--- SUCCÈS : {len(chunks)} blocs vectorisés et stockés dans FAISS ! ---")
|
|
else:
|
|
print("--- ÉCHEC : Aucun vecteur généré. ---")
|
|
|
|
def rechercher_contexte_vectoriel(thematique: str, question: str, top_k: int = 3):
|
|
"""Recherche les extraits de texte pertinents pour répondre à une question."""
|
|
dossier_db = os.path.join(VECTOR_DB_DIR, thematique.lower())
|
|
fichier_index = os.path.join(dossier_db, "index.faiss")
|
|
fichier_chunks = os.path.join(dossier_db, "chunks.json")
|
|
|
|
if not os.path.exists(fichier_index) or not os.path.exists(fichier_chunks):
|
|
return ""
|
|
|
|
try:
|
|
index = faiss.read_index(fichier_index)
|
|
with open(fichier_chunks, 'r', encoding='utf-8') as f:
|
|
chunks = json.load(f)
|
|
|
|
# On vectorise la question avec le même modèle Ollama
|
|
vecteur_q = obtenir_embedding(question)
|
|
if not vecteur_q:
|
|
return ""
|
|
|
|
vecteur_q_np = np.array([vecteur_q]).astype('float32')
|
|
|
|
# On demande à FAISS les K plus proches voisins (top_k)
|
|
distances, indices = index.search(vecteur_q_np, top_k)
|
|
|
|
fragments = []
|
|
for i in indices[0]:
|
|
if i < len(chunks) and i != -1:
|
|
fragments.append(chunks[i]["texte"])
|
|
|
|
if fragments:
|
|
return "\n\n[...]\n\n".join(fragments)
|
|
|
|
except Exception as e:
|
|
print(f"Erreur de recherche FAISS : {e}")
|
|
|
|
return "" |