fix: close #1 - add bot-detection evasion for Reuters scraping
- Add rotating User-Agent pool (5 browser profiles) to all scrapers - Fix Playwright: use browser context for UA instead of broken set_extra_http_headers - Fix Selenium: set general.useragent.override preference - Fix newspaper4k: pass browser_user_agent to bypass bot detection - Add random delays (0.5-2s pre-fetch, 1-4s post-load) to mimic human behavior - Switch Reuters RSS from Google News proxy to official Reuters agency feed - Add legitimacy headers (Accept, Accept-Language, Connection) to Playwright - Apply all fixes to both scraper.py and cron_scraper.py
This commit is contained in:
parent
e9785b3ea2
commit
6b8a87f9ef
@ -6,12 +6,14 @@ that can be scheduled via cron job.
|
||||
"""
|
||||
|
||||
import newspaper
|
||||
from newspaper import Config
|
||||
import json
|
||||
import feedparser
|
||||
import time
|
||||
import os
|
||||
import requests
|
||||
import logging
|
||||
import random
|
||||
from datetime import datetime
|
||||
from selenium import webdriver
|
||||
from selenium.webdriver.firefox.options import Options as FirefoxOptions
|
||||
@ -30,6 +32,18 @@ logging.basicConfig(
|
||||
)
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
# Rotating User-Agents to bypass bot detection (Reuters, etc.)
|
||||
USER_AGENTS = [
|
||||
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36",
|
||||
"Mozilla/5.0 (Macintosh; Intel Mac OS X 14_5) AppleWebKit/605.1.15 (KHTML, like Gecko) Version/17.5 Safari/605.1.15",
|
||||
"Mozilla/5.0 (Windows NT 10.0; Win64; x64; rv:134.0) Gecko/20100101 Firefox/134.0",
|
||||
"Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36",
|
||||
"Mozilla/5.0 (Macintosh; Intel Mac OS X 14_5) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36",
|
||||
]
|
||||
|
||||
def get_random_ua():
|
||||
return random.choice(USER_AGENTS)
|
||||
|
||||
# Robust file path handling - try multiple locations
|
||||
def get_feed_file_path():
|
||||
"""Get the RSS feed file path, trying multiple locations."""
|
||||
@ -165,19 +179,22 @@ def save_article_to_file(article, filename, source="Unfiltered"):
|
||||
|
||||
def get_article_with_selenium(url):
|
||||
"""
|
||||
Gets article text using Selenium Firefox driver with proper error handling
|
||||
and cleanup.
|
||||
Gets article text using Selenium Firefox driver with proper error handling,
|
||||
cleanup, and bot-detection evasion.
|
||||
"""
|
||||
driver = None
|
||||
try:
|
||||
# Configure Firefox options
|
||||
# Configure Firefox options with bot-detection evasion
|
||||
options = FirefoxOptions()
|
||||
options.add_argument("--headless")
|
||||
options.set_preference("dom.ipc.processCount", 1) # Reduce process count
|
||||
options.set_preference("dom.ipc.processCount", 1)
|
||||
options.set_preference("general.useragent.override", get_random_ua())
|
||||
options.set_preference("permissions.default.image", 2)
|
||||
options.set_preference("dom.webnotifications.enabled", False)
|
||||
|
||||
# Initialize driver with timeout
|
||||
driver = webdriver.Firefox(options=options)
|
||||
driver.set_page_load_timeout(30) # 30 seconds timeout
|
||||
driver.set_page_load_timeout(30)
|
||||
|
||||
# Navigate to URL
|
||||
driver.get(url)
|
||||
@ -188,9 +205,9 @@ def get_article_with_selenium(url):
|
||||
EC.presence_of_element_located((By.TAG_NAME, "body"))
|
||||
)
|
||||
except:
|
||||
pass # Continue even if wait times out
|
||||
pass
|
||||
|
||||
time.sleep(2) # Brief additional wait
|
||||
time.sleep(random.uniform(1, 3))
|
||||
|
||||
html = driver.page_source
|
||||
|
||||
@ -204,42 +221,45 @@ def get_article_with_selenium(url):
|
||||
logger.error(f"Selenium failed for {url}: {str(e)}")
|
||||
return ""
|
||||
finally:
|
||||
# Always quit the driver
|
||||
if driver:
|
||||
try:
|
||||
driver.quit()
|
||||
except:
|
||||
pass # Ignore errors in cleanup
|
||||
pass
|
||||
|
||||
|
||||
def get_article_with_playwright(url):
|
||||
"""
|
||||
Gets article text using Playwright with proper error handling.
|
||||
Gets article text using Playwright with proper bot-detection evasion.
|
||||
"""
|
||||
try:
|
||||
from playwright.sync_api import sync_playwright
|
||||
|
||||
with sync_playwright() as p:
|
||||
# Use Chromium instead of Firefox for better compatibility
|
||||
browser = p.chromium.launch(headless=True, timeout=30000)
|
||||
page = browser.new_page()
|
||||
|
||||
# Set user agent to avoid bot detection
|
||||
page.set_extra_http_headers(
|
||||
{
|
||||
"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"
|
||||
}
|
||||
context = browser.new_context(
|
||||
user_agent=get_random_ua(),
|
||||
viewport={"width": 1920, "height": 1080},
|
||||
locale="en-US",
|
||||
timezone_id="America/New_York",
|
||||
)
|
||||
page = context.new_page()
|
||||
|
||||
page.goto(url, wait_until="load")
|
||||
page.set_extra_http_headers({
|
||||
"Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
|
||||
"Accept-Language": "en-US,en;q=0.9",
|
||||
"Accept-Encoding": "gzip, deflate, br",
|
||||
"Connection": "keep-alive",
|
||||
"Upgrade-Insecure-Requests": "1",
|
||||
})
|
||||
|
||||
# Wait for content to load
|
||||
time.sleep(3)
|
||||
page.goto(url, wait_until="domcontentloaded", timeout=30000)
|
||||
time.sleep(random.uniform(2, 4))
|
||||
|
||||
html = page.content()
|
||||
context.close()
|
||||
browser.close()
|
||||
|
||||
# Parse with Newspaper4k
|
||||
article = newspaper.article(url, input_html=html, language="en")
|
||||
article.nlp()
|
||||
logger.info(f"Successfully extracted article with Playwright from {url}")
|
||||
@ -265,11 +285,16 @@ def pull_article(link, source, title=None, save_to_file=True):
|
||||
) as f:
|
||||
return f.read()
|
||||
|
||||
# Random delay before fetching to avoid rate-limiting / bot detection
|
||||
time.sleep(random.uniform(0.5, 2))
|
||||
|
||||
text = ""
|
||||
|
||||
try:
|
||||
# Try newspaper4k first
|
||||
article = newspaper.article(link)
|
||||
# Try newspaper4k first with proper User-Agent to bypass bot detection
|
||||
ua = get_random_ua()
|
||||
config = Config(browser_user_agent=ua)
|
||||
article = newspaper.article(link, browser_user_agent=ua)
|
||||
article.download()
|
||||
article.parse()
|
||||
text = article.text
|
||||
|
||||
@ -2,7 +2,7 @@
|
||||
"rss_feeds": {
|
||||
"Reuters – Business News": {
|
||||
"source_website": "reuters.com",
|
||||
"rss_url": "https://news.google.com/rss/search?q=site:reuters.com+business&hl=en-US&gl=US&ceid=US:en"
|
||||
"rss_url": "https://www.reutersagency.com/feed/"
|
||||
},
|
||||
"Associated Press – Business": {
|
||||
"source_website": "apnews.com",
|
||||
|
||||
@ -1,9 +1,11 @@
|
||||
import newspaper
|
||||
from newspaper import Config
|
||||
import json
|
||||
import feedparser
|
||||
import time
|
||||
import os
|
||||
import logging
|
||||
import random
|
||||
from datetime import datetime
|
||||
from selenium import webdriver
|
||||
from selenium.webdriver.firefox.options import Options as FirefoxOptions
|
||||
@ -27,6 +29,18 @@ MAX_FEED_WORKERS = int(os.getenv("MAX_FEED_WORKERS", "10"))
|
||||
MAX_ARTICLE_WORKERS = int(os.getenv("MAX_ARTICLE_WORKERS", "10"))
|
||||
BATCH_SIZE = int(os.getenv("BATCH_SIZE", "50")) # Batch processing size
|
||||
|
||||
# Rotating User-Agents to bypass bot detection (Reuters, etc.)
|
||||
USER_AGENTS = [
|
||||
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36",
|
||||
"Mozilla/5.0 (Macintosh; Intel Mac OS X 14_5) AppleWebKit/605.1.15 (KHTML, like Gecko) Version/17.5 Safari/605.1.15",
|
||||
"Mozilla/5.0 (Windows NT 10.0; Win64; x64; rv:134.0) Gecko/20100101 Firefox/134.0",
|
||||
"Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36",
|
||||
"Mozilla/5.0 (Macintosh; Intel Mac OS X 14_5) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36",
|
||||
]
|
||||
|
||||
def get_random_ua():
|
||||
return random.choice(USER_AGENTS)
|
||||
|
||||
# Ensure necessary NLTK resources are downloaded
|
||||
d = Downloader()
|
||||
if not d.is_installed("punkt_tab"):
|
||||
@ -237,15 +251,18 @@ def save_article_to_file(article, filename, source="Unfiltered"):
|
||||
|
||||
def get_article_with_selenium(url):
|
||||
"""
|
||||
Gets article text using Selenium Firefox driver with proper error handling
|
||||
and cleanup.
|
||||
Gets article text using Selenium Firefox driver with proper error handling,
|
||||
cleanup, and bot-detection evasion.
|
||||
"""
|
||||
driver = None
|
||||
try:
|
||||
# Configure Firefox options
|
||||
# Configure Firefox options with bot-detection evasion
|
||||
options = FirefoxOptions()
|
||||
options.add_argument("--headless")
|
||||
options.set_preference("dom.ipc.processCount", 1) # Reduce process count
|
||||
options.set_preference("dom.ipc.processCount", 1)
|
||||
options.set_preference("general.useragent.override", get_random_ua())
|
||||
options.set_preference("permissions.default.image", 2) # Block images for speed
|
||||
options.set_preference("dom.webnotifications.enabled", False)
|
||||
|
||||
# Try to initialize driver with explicit path to Firefox
|
||||
try:
|
||||
@ -272,7 +289,7 @@ def get_article_with_selenium(url):
|
||||
except:
|
||||
pass # Continue even if wait times out
|
||||
|
||||
time.sleep(2) # Brief additional wait
|
||||
time.sleep(random.uniform(1, 3)) # Random wait to mimic human behavior
|
||||
|
||||
html = driver.page_source
|
||||
|
||||
@ -302,29 +319,38 @@ def get_article_with_selenium(url):
|
||||
|
||||
def get_article_with_playwright(url):
|
||||
"""
|
||||
Gets article text using Playwright with proper error handling.
|
||||
Gets article text using Playwright with proper bot-detection evasion.
|
||||
"""
|
||||
try:
|
||||
from playwright.sync_api import sync_playwright
|
||||
|
||||
with sync_playwright() as p:
|
||||
# Use Chromium instead of Firefox for better compatibility
|
||||
# Use Chromium with full browser context for UA spoofing
|
||||
browser = p.chromium.launch(headless=True, timeout=30000)
|
||||
page = browser.new_page()
|
||||
|
||||
# Set user agent to avoid bot detection
|
||||
page.set_extra_http_headers(
|
||||
{
|
||||
"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"
|
||||
}
|
||||
context = browser.new_context(
|
||||
user_agent=get_random_ua(),
|
||||
viewport={"width": 1920, "height": 1080},
|
||||
locale="en-US",
|
||||
timezone_id="America/New_York",
|
||||
)
|
||||
page = context.new_page()
|
||||
|
||||
page.goto(url, wait_until="load")
|
||||
# Additional headers for legitimacy
|
||||
page.set_extra_http_headers({
|
||||
"Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
|
||||
"Accept-Language": "en-US,en;q=0.9",
|
||||
"Accept-Encoding": "gzip, deflate, br",
|
||||
"Connection": "keep-alive",
|
||||
"Upgrade-Insecure-Requests": "1",
|
||||
})
|
||||
|
||||
page.goto(url, wait_until="domcontentloaded", timeout=30000)
|
||||
|
||||
# Wait for content to load
|
||||
time.sleep(3)
|
||||
time.sleep(random.uniform(2, 4))
|
||||
|
||||
html = page.content()
|
||||
context.close()
|
||||
browser.close()
|
||||
|
||||
# Parse with Newspaper4k
|
||||
@ -353,11 +379,16 @@ def pull_article(link, source, title=None, save_to_file=True):
|
||||
) as f:
|
||||
return f.read()
|
||||
|
||||
# Random delay before fetching to avoid rate-limiting / bot detection
|
||||
time.sleep(random.uniform(0.5, 2))
|
||||
|
||||
text = ""
|
||||
|
||||
try:
|
||||
# Try newspaper4k first
|
||||
article = newspaper.article(link)
|
||||
# Try newspaper4k first with proper User-Agent to bypass bot detection
|
||||
ua = get_random_ua()
|
||||
config = Config(browser_user_agent=ua)
|
||||
article = newspaper.article(link, browser_user_agent=ua)
|
||||
article.download()
|
||||
article.parse()
|
||||
text = article.text
|
||||
|
||||
Loading…
x
Reference in New Issue
Block a user