158 lines
5.8 KiB
Python
158 lines
5.8 KiB
Python
"""Async httpx web scraper with retries, rotating user agents, and fallback."""
|
|
from __future__ import annotations
|
|
import logging
|
|
import random
|
|
import re
|
|
from typing import Optional
|
|
|
|
import httpx
|
|
from bs4 import BeautifulSoup
|
|
|
|
from models import ScrapeResult
|
|
|
|
logger = logging.getLogger(__name__)
|
|
|
|
USER_AGENTS = [
|
|
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/126.0.0.0 Safari/537.36",
|
|
"Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/126.0.0.0 Safari/537.36",
|
|
"Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/126.0.0.0 Safari/537.36",
|
|
"Mozilla/5.0 (Windows NT 10.0; Win64; x64; rv:128.0) Gecko/20100101 Firefox/128.0",
|
|
]
|
|
|
|
REMOVE_TAGS = {"script", "style", "nav", "footer", "header", "noscript", "iframe", "svg"}
|
|
|
|
|
|
class AsyncScraper:
|
|
"""Asynchronous web scraper with retries, rotating UAs, and fallback HTTP clients."""
|
|
|
|
def __init__(self, timeout_ms: int = 20_000):
|
|
self.timeout = timeout_ms / 1000.0
|
|
|
|
async def scrape(self, url: str) -> ScrapeResult:
|
|
"""Scrape a URL with retries and fallbacks."""
|
|
errors = []
|
|
|
|
# Strategy 1: Standard httpx client
|
|
for attempt in range(3):
|
|
try:
|
|
result = await self._try_scrape(url, attempt)
|
|
if result.text_content and len(result.text_content) > 50:
|
|
return result
|
|
if result.error:
|
|
errors.append(f"Attempt {attempt+1}: {result.error}")
|
|
except Exception as e:
|
|
errors.append(f"Attempt {attempt+1}: {str(e)}")
|
|
|
|
# Strategy 2: Minimal headers (like curl)
|
|
for attempt in range(2):
|
|
try:
|
|
result = await self._try_minimal(url)
|
|
if result.text_content and len(result.text_content) > 50:
|
|
return result
|
|
if result.error:
|
|
errors.append(f"Minimal {attempt+1}: {result.error}")
|
|
except Exception as e:
|
|
errors.append(f"Minimal {attempt+1}: {str(e)}")
|
|
|
|
return ScrapeResult(url=url, title="", text_content="", text_length=0,
|
|
error="; ".join(errors[-3:]) or "All scrape attempts failed")
|
|
|
|
async def _try_scrape(self, url: str, attempt: int) -> ScrapeResult:
|
|
ua = USER_AGENTS[attempt % len(USER_AGENTS)]
|
|
headers = {
|
|
"User-Agent": ua,
|
|
"Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,image/webp,*/*;q=0.8",
|
|
"Accept-Language": "en-US,en;q=0.9,nl;q=0.8",
|
|
"Accept-Encoding": "gzip, deflate, br",
|
|
"DNT": "1",
|
|
"Connection": "keep-alive",
|
|
"Upgrade-Insecure-Requests": "1",
|
|
"Sec-Fetch-Dest": "document",
|
|
"Sec-Fetch-Mode": "navigate",
|
|
"Sec-Fetch-Site": "none",
|
|
"Sec-Fetch-User": "?1",
|
|
"Cache-Control": "max-age=0",
|
|
}
|
|
async with httpx.AsyncClient(
|
|
timeout=self.timeout,
|
|
follow_redirects=True,
|
|
headers=headers,
|
|
http2=True,
|
|
) as client:
|
|
response = await client.get(url)
|
|
return self._parse_response(response, url)
|
|
|
|
async def _try_minimal(self, url: str) -> ScrapeResult:
|
|
headers = {
|
|
"User-Agent": "curl/8.4.0",
|
|
"Accept": "*/*",
|
|
}
|
|
async with httpx.AsyncClient(
|
|
timeout=self.timeout,
|
|
follow_redirects=True,
|
|
headers=headers,
|
|
http2=False,
|
|
) as client:
|
|
response = await client.get(url)
|
|
return self._parse_response(response, url)
|
|
|
|
def _parse_response(self, response, url: str) -> ScrapeResult:
|
|
result = ScrapeResult(url=url, title="", text_content="", text_length=0)
|
|
result.status_code = response.status_code
|
|
|
|
if response.status_code >= 500:
|
|
result.error = f"HTTP {response.status_code}"
|
|
return result
|
|
if response.status_code >= 400 and response.status_code < 500:
|
|
if response.status_code in (403, 429):
|
|
result.error = f"HTTP {response.status_code} (blocked/rate-limited)"
|
|
else:
|
|
result.error = f"HTTP {response.status_code}"
|
|
return result
|
|
|
|
try:
|
|
html = response.text
|
|
except Exception:
|
|
result.error = "Could not decode response"
|
|
return result
|
|
|
|
if not html or len(html) < 100:
|
|
result.error = "Empty response"
|
|
return result
|
|
|
|
soup = BeautifulSoup(html, "html.parser")
|
|
result.title = soup.title.string.strip() if soup.title else ""
|
|
|
|
# Remove non-content elements
|
|
for tag in soup(REMOVE_TAGS):
|
|
tag.decompose()
|
|
|
|
body = soup.find("body")
|
|
text = body.get_text(separator="\n") if body else soup.get_text(separator="\n")
|
|
|
|
# Also extract text from JSON-LD structured data
|
|
for script in soup.find_all("script", type="application/ld+json"):
|
|
try:
|
|
import json
|
|
ld = json.loads(script.string or "")
|
|
if isinstance(ld, dict):
|
|
desc = ld.get("description", "")
|
|
if desc:
|
|
text += "\n" + desc
|
|
except Exception:
|
|
pass
|
|
|
|
result.text_content = self._clean_text(text)
|
|
result.text_length = len(result.text_content)
|
|
return result
|
|
|
|
@staticmethod
|
|
def _clean_text(text: str) -> str:
|
|
if not text:
|
|
return ""
|
|
text = re.sub(r"[ \t]+", " ", text)
|
|
text = re.sub(r"\n{3,}", "\n\n", text)
|
|
lines = [ln.strip() for ln in text.split("\n")]
|
|
lines = [ln for ln in lines if len(ln) > 2]
|
|
return "\n".join(lines).strip()
|