Public Access
43 lines
1.6 KiB
Python
43 lines
1.6 KiB
Python
import requests
|
|
from bs4 import BeautifulSoup
|
|
|
|
def rechercher_sur_google(requete):
|
|
try:
|
|
url = "https://lite.duckduckgo.com/lite/"
|
|
data = {'q': requete}
|
|
headers = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64)"}
|
|
resp = requests.post(url, data=data, headers=headers, timeout=5)
|
|
soup = BeautifulSoup(resp.text, 'html.parser')
|
|
|
|
urls = []
|
|
for a in soup.find_all('a', class_='result-link', href=True):
|
|
href = a['href']
|
|
if href.startswith('http') and 'duckduckgo' not in href:
|
|
urls.append(href)
|
|
if len(urls) >= 5:
|
|
break
|
|
if not urls:
|
|
for a in soup.find_all('a', href=True):
|
|
href = a['href']
|
|
if href.startswith('http') and 'duckduckgo' not in href and href not in urls:
|
|
urls.append(href)
|
|
if len(urls) >= 5:
|
|
break
|
|
return urls[:5]
|
|
except Exception as e:
|
|
print(f"Erreur recherche web: {e}")
|
|
return []
|
|
|
|
def extraire_contenu_urls(urls_list):
|
|
scraped_text = ""
|
|
for url in urls_list:
|
|
try:
|
|
resp = requests.get(url, timeout=5, headers={"User-Agent": "Mozilla/5.0"})
|
|
soup = BeautifulSoup(resp.text, 'html.parser')
|
|
for script in soup(["script", "style", "nav", "footer"]):
|
|
script.extract()
|
|
scraped_text += f"\n\n--- Source web extraite ({url}) ---\n{soup.get_text(separator=' ', strip=True)[:3000]}"
|
|
except Exception as e:
|
|
scraped_text += f"\n\n[Impossible de récupérer {url}: {e}]"
|
|
return scraped_text
|