Mise a jour module éduction nationale
Build and Push Docker Image / build-and-push (push) Successful in 35s

This commit is contained in:
xavier committed 2026-08-27 22:03:41 +02:00
1 parent dd5cd760e0
commit 9496ed6fb7
1 file changed
+11 -6
+11 -6
View File
@@ -7,15 +7,20 @@ from modules.logger import log_event
RAG_DIR = "/DATA/AppData/MULTI-IA-AETHAS38/RAG/Education_Nationale" RAG_DIR = "/DATA/AppData/MULTI-IA-AETHAS38/RAG/Education_Nationale"
def maj_base_education(): def maj_base_education():
"""Scrape le flux RSS du Bulletin Officiel pour maintenir la base documentaire à jour.""" """Scrape le flux RSS du Bulletin Officiel en contournant les protections WAF."""
os.makedirs(RAG_DIR, exist_ok=True) os.makedirs(RAG_DIR, exist_ok=True)
url_rss = "https://www.education.gouv.fr/bo/rss.xml" url_rss = "https://www.education.gouv.fr/bo/rss.xml"
# En-têtes complets simulant un navigateur pour éviter le blocage 403 # En-têtes complets simulant une vraie navigation web humaine
headers = { headers = {
"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36", "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/122.0.0.0 Safari/537.36",
"Accept": "application/rss+xml,application/xml;q=0.9,text/html;q=0.9,*/*;q=0.8", "Accept": "application/rss+xml,application/xml,text/xml;q=0.9,*/*;q=0.8",
"Accept-Language": "fr-FR,fr;q=0.9,en-US;q=0.8,en;q=0.7" "Accept-Language": "fr-FR,fr;q=0.9,en-US;q=0.8,en;q=0.7",
"Referer": "https://www.education.gouv.fr/bo/le-bulletin-officiel-de-l-education-nationale",
"Sec-Ch-Ua": '"Chromium";v="122", "Not(A:Brand";v="24", "Google Chrome";v="122"',
"Sec-Ch-Ua-Mobile": "?0",
"Sec-Ch-Ua-Platform": '"Windows"',
"Upgrade-Insecure-Requests": "1"
} }
try: try:
@@ -27,7 +32,7 @@ def maj_base_education():
contenu_md += f"*Date de mise à jour RAG : {datetime.now().strftime('%Y-%m-%d %H:%M:%S')}*\n\n" contenu_md += f"*Date de mise à jour RAG : {datetime.now().strftime('%Y-%m-%d %H:%M:%S')}*\n\n"
for item in root.findall('.//item')[:20]: for item in root.findall('.//item')[:20]:
titre = item.find('title').text if item.find('title') is not None else "Document sans titre" titre = item.find('title').text if item.find('title'] is not None else "Document sans titre"
lien = item.find('link').text if item.find('link') is not None else "Aucun lien" lien = item.find('link').text if item.find('link') is not None else "Aucun lien"
desc = item.find('description').text if item.find('description') is not None else "Pas de résumé disponible." desc = item.find('description').text if item.find('description') is not None else "Pas de résumé disponible."