Public Access
This commit is contained in:
1 parent
e3557873b9
commit
b88d693d69
2 files changed
+86
-33
No files matched your search
+63
-1
@@ -118,4 +118,66 @@ def rechercher_contexte_vectoriel(thematique: str, question: str, top_k: int = 3
|
||||
except Exception as e:
|
||||
print(f"Erreur de recherche FAISS : {e}")
|
||||
|
||||
return ""
|
||||
return ""
|
||||
|
||||
def indexer_tout_le_rag():
|
||||
"""Parcourt tous les sous-dossiers de RAG_BASE_DIR et les indexe automatiquement."""
|
||||
print("=== LANCEMENT DE L'INDEXATION GLOBALE ===")
|
||||
if not os.path.exists(RAG_BASE_DIR):
|
||||
print(f"Dossier introuvable : {RAG_BASE_DIR}")
|
||||
return
|
||||
|
||||
for element in os.listdir(RAG_BASE_DIR):
|
||||
chemin = os.path.join(RAG_BASE_DIR, element)
|
||||
if os.path.isdir(chemin):
|
||||
# Appelle la fonction existante pour chaque dossier trouvé
|
||||
indexer_dossier_thematique(element)
|
||||
print("=== INDEXATION GLOBALE TERMINÉE ===")
|
||||
|
||||
def rechercher_contexte_global(question: str, top_k: int = 3):
|
||||
"""Cherche la réponse dans TOUTES les thématiques indexées et garde les meilleurs extraits."""
|
||||
if not os.path.exists(VECTOR_DB_DIR):
|
||||
return ""
|
||||
|
||||
vecteur_q = obtenir_embedding(question)
|
||||
if not vecteur_q:
|
||||
return ""
|
||||
|
||||
vecteur_q_np = np.array([vecteur_q]).astype('float32')
|
||||
tous_fragments = []
|
||||
|
||||
# On fouille dans chaque dossier de la base vectorielle
|
||||
for thematique in os.listdir(VECTOR_DB_DIR):
|
||||
dossier_db = os.path.join(VECTOR_DB_DIR, thematique)
|
||||
fichier_index = os.path.join(dossier_db, "index.faiss")
|
||||
fichier_chunks = os.path.join(dossier_db, "chunks.json")
|
||||
|
||||
if os.path.exists(fichier_index) and os.path.exists(fichier_chunks):
|
||||
try:
|
||||
index = faiss.read_index(fichier_index)
|
||||
with open(fichier_chunks, 'r', encoding='utf-8') as f:
|
||||
chunks = json.load(f)
|
||||
|
||||
# Recherche dans ce dossier spécifique
|
||||
distances, indices = index.search(vecteur_q_np, top_k)
|
||||
|
||||
# On stocke les résultats avec leur score de pertinence (distance)
|
||||
for dist, i in zip(distances[0], indices[0]):
|
||||
if i < len(chunks) and i != -1:
|
||||
tous_fragments.append((dist, chunks[i]["texte"], thematique))
|
||||
except Exception:
|
||||
continue
|
||||
|
||||
if not tous_fragments:
|
||||
return ""
|
||||
|
||||
# On trie TOUS les fragments trouvés pour ne garder que les (top_k) les plus pertinents, tous dossiers confondus
|
||||
# (FAISS utilise la distance L2 : plus le score est petit, plus le texte correspond à la question)
|
||||
tous_fragments.sort(key=lambda x: x[0])
|
||||
meilleurs = tous_fragments[:top_k]
|
||||
|
||||
contexte_final = []
|
||||
for dist, texte, theme in meilleurs:
|
||||
contexte_final.append(f"[Source : Dossier {theme}]\n{texte}")
|
||||
|
||||
return "\n\n---\n\n".join(contexte_final)
|
||||
Reference in new issue
Block a user