Public Access
59 lines
2.3 KiB
Python
59 lines
2.3 KiB
Python
import os
|
|
import pypdf
|
|
from docx import Document
|
|
import chromadb
|
|
from modules.logger import log_event
|
|
|
|
RAG_BASE_DIR = "/DATA/AppData/MULTI-IA-AETHAS38/RAG"
|
|
VECTOR_DB_DIR = "/DATA/AppData/MULTI-IA-AETHAS38/VectorDB"
|
|
|
|
def initialiser_chroma():
|
|
"""Initialise le client ChromaDB en mode persistant sur le RAID."""
|
|
os.makedirs(VECTOR_DB_DIR, exist_ok=True)
|
|
client = chromadb.PersistentClient(path=VECTOR_DB_DIR)
|
|
return client
|
|
|
|
def extraire_texte_fichier(chemin_fichier):
|
|
"""Extrait le texte brut d'un PDF, DOCX ou MD/TXT."""
|
|
texte = ""
|
|
bas_chemin = chemin_fichier.lower()
|
|
try:
|
|
if bas_chemin.endswith('.pdf'):
|
|
reader = pypdf.PdfReader(chemin_fichier)
|
|
for page in reader.pages:
|
|
texte += page.extract_text() or ""
|
|
elif bas_chemin.endswith('.docx'):
|
|
doc = Document(chemin_fichier)
|
|
texte = "\n".join([para.text for para in doc.paragraphs])
|
|
elif bas_chemin.endswith(('.txt', '.md')):
|
|
with open(chemin_fichier, 'r', encoding='utf-8') as f:
|
|
texte = f.read()
|
|
except Exception as e:
|
|
log_event("ERROR", "VECTOR_RAG", fogli="Erreur extraction {chemin_fichier} : {e}")
|
|
return texte
|
|
|
|
def indexer_dossier_thematique(thematique: str):
|
|
"""Parcourt une thématique, découpe et indexe les documents dans ChromaDB."""
|
|
client = initialiser_chroma()
|
|
collection = client.get_or_create_collection(name=thematique.lower())
|
|
|
|
chemin_theme = os.path.join(RAG_BASE_DIR, thematique)
|
|
if not os.path.exists(chemin_theme):
|
|
return
|
|
|
|
for fichier in os.listdir(chemin_theme):
|
|
chemin_complet = os.path.join(chemin_theme, fichier)
|
|
if os.path.isfile(chemin_complet):
|
|
texte = extraire_texte_fichier(chemin_complet)
|
|
if len(texte.strip()) > 50:
|
|
# Découpage simple par blocs de 1000 caractères (Chunks)
|
|
taille_chunk = 1000
|
|
chunks = [texte[i:i+taille_chunk] for i in range(0, len(texte), taille_chunk)]
|
|
|
|
ids = [f"{fichier}_chunk_{idx}" for idx in range(len(chunks))]
|
|
|
|
collection.upsert(
|
|
documents=chunks,
|
|
ids=ids
|
|
)
|
|
log_event("INFO", "VECTOR_RAG", f"Indexé {len(chunks)} blocs pour {fichier} dans {thematique}.") |