diff --git a/scraper/cron_scraper.py b/scraper/cron_scraper.py index 7aee3c6..fe763ce 100644 --- a/scraper/cron_scraper.py +++ b/scraper/cron_scraper.py @@ -6,12 +6,14 @@ that can be scheduled via cron job. """ import newspaper +from newspaper import Config import json import feedparser import time import os import requests import logging +import random from datetime import datetime from selenium import webdriver from selenium.webdriver.firefox.options import Options as FirefoxOptions @@ -30,6 +32,18 @@ logging.basicConfig( ) logger = logging.getLogger(__name__) +# Rotating User-Agents to bypass bot detection (Reuters, etc.) +USER_AGENTS = [ + "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36", + "Mozilla/5.0 (Macintosh; Intel Mac OS X 14_5) AppleWebKit/605.1.15 (KHTML, like Gecko) Version/17.5 Safari/605.1.15", + "Mozilla/5.0 (Windows NT 10.0; Win64; x64; rv:134.0) Gecko/20100101 Firefox/134.0", + "Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36", + "Mozilla/5.0 (Macintosh; Intel Mac OS X 14_5) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36", +] + +def get_random_ua(): + return random.choice(USER_AGENTS) + # Robust file path handling - try multiple locations def get_feed_file_path(): """Get the RSS feed file path, trying multiple locations.""" @@ -165,19 +179,22 @@ def save_article_to_file(article, filename, source="Unfiltered"): def get_article_with_selenium(url): """ - Gets article text using Selenium Firefox driver with proper error handling - and cleanup. + Gets article text using Selenium Firefox driver with proper error handling, + cleanup, and bot-detection evasion. """ driver = None try: - # Configure Firefox options + # Configure Firefox options with bot-detection evasion options = FirefoxOptions() options.add_argument("--headless") - options.set_preference("dom.ipc.processCount", 1) # Reduce process count + options.set_preference("dom.ipc.processCount", 1) + options.set_preference("general.useragent.override", get_random_ua()) + options.set_preference("permissions.default.image", 2) + options.set_preference("dom.webnotifications.enabled", False) # Initialize driver with timeout driver = webdriver.Firefox(options=options) - driver.set_page_load_timeout(30) # 30 seconds timeout + driver.set_page_load_timeout(30) # Navigate to URL driver.get(url) @@ -188,9 +205,9 @@ def get_article_with_selenium(url): EC.presence_of_element_located((By.TAG_NAME, "body")) ) except: - pass # Continue even if wait times out + pass - time.sleep(2) # Brief additional wait + time.sleep(random.uniform(1, 3)) html = driver.page_source @@ -204,42 +221,45 @@ def get_article_with_selenium(url): logger.error(f"Selenium failed for {url}: {str(e)}") return "" finally: - # Always quit the driver if driver: try: driver.quit() except: - pass # Ignore errors in cleanup + pass def get_article_with_playwright(url): """ - Gets article text using Playwright with proper error handling. + Gets article text using Playwright with proper bot-detection evasion. """ try: from playwright.sync_api import sync_playwright with sync_playwright() as p: - # Use Chromium instead of Firefox for better compatibility browser = p.chromium.launch(headless=True, timeout=30000) - page = browser.new_page() - - # Set user agent to avoid bot detection - page.set_extra_http_headers( - { - "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36" - } + context = browser.new_context( + user_agent=get_random_ua(), + viewport={"width": 1920, "height": 1080}, + locale="en-US", + timezone_id="America/New_York", ) + page = context.new_page() - page.goto(url, wait_until="load") + page.set_extra_http_headers({ + "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8", + "Accept-Language": "en-US,en;q=0.9", + "Accept-Encoding": "gzip, deflate, br", + "Connection": "keep-alive", + "Upgrade-Insecure-Requests": "1", + }) - # Wait for content to load - time.sleep(3) + page.goto(url, wait_until="domcontentloaded", timeout=30000) + time.sleep(random.uniform(2, 4)) html = page.content() + context.close() browser.close() - # Parse with Newspaper4k article = newspaper.article(url, input_html=html, language="en") article.nlp() logger.info(f"Successfully extracted article with Playwright from {url}") @@ -265,11 +285,16 @@ def pull_article(link, source, title=None, save_to_file=True): ) as f: return f.read() + # Random delay before fetching to avoid rate-limiting / bot detection + time.sleep(random.uniform(0.5, 2)) + text = "" try: - # Try newspaper4k first - article = newspaper.article(link) + # Try newspaper4k first with proper User-Agent to bypass bot detection + ua = get_random_ua() + config = Config(browser_user_agent=ua) + article = newspaper.article(link, browser_user_agent=ua) article.download() article.parse() text = article.text diff --git a/scraper/rss_feeds.json b/scraper/rss_feeds.json index f0f935a..d6ad9e8 100644 --- a/scraper/rss_feeds.json +++ b/scraper/rss_feeds.json @@ -2,7 +2,7 @@ "rss_feeds": { "Reuters – Business News": { "source_website": "reuters.com", - "rss_url": "https://news.google.com/rss/search?q=site:reuters.com+business&hl=en-US&gl=US&ceid=US:en" + "rss_url": "https://www.reutersagency.com/feed/" }, "Associated Press – Business": { "source_website": "apnews.com", diff --git a/scraper/scraper.py b/scraper/scraper.py index 2b0c545..157d44f 100644 --- a/scraper/scraper.py +++ b/scraper/scraper.py @@ -1,9 +1,11 @@ import newspaper +from newspaper import Config import json import feedparser import time import os import logging +import random from datetime import datetime from selenium import webdriver from selenium.webdriver.firefox.options import Options as FirefoxOptions @@ -27,6 +29,18 @@ MAX_FEED_WORKERS = int(os.getenv("MAX_FEED_WORKERS", "10")) MAX_ARTICLE_WORKERS = int(os.getenv("MAX_ARTICLE_WORKERS", "10")) BATCH_SIZE = int(os.getenv("BATCH_SIZE", "50")) # Batch processing size +# Rotating User-Agents to bypass bot detection (Reuters, etc.) +USER_AGENTS = [ + "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36", + "Mozilla/5.0 (Macintosh; Intel Mac OS X 14_5) AppleWebKit/605.1.15 (KHTML, like Gecko) Version/17.5 Safari/605.1.15", + "Mozilla/5.0 (Windows NT 10.0; Win64; x64; rv:134.0) Gecko/20100101 Firefox/134.0", + "Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36", + "Mozilla/5.0 (Macintosh; Intel Mac OS X 14_5) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36", +] + +def get_random_ua(): + return random.choice(USER_AGENTS) + # Ensure necessary NLTK resources are downloaded d = Downloader() if not d.is_installed("punkt_tab"): @@ -237,15 +251,18 @@ def save_article_to_file(article, filename, source="Unfiltered"): def get_article_with_selenium(url): """ - Gets article text using Selenium Firefox driver with proper error handling - and cleanup. + Gets article text using Selenium Firefox driver with proper error handling, + cleanup, and bot-detection evasion. """ driver = None try: - # Configure Firefox options + # Configure Firefox options with bot-detection evasion options = FirefoxOptions() options.add_argument("--headless") - options.set_preference("dom.ipc.processCount", 1) # Reduce process count + options.set_preference("dom.ipc.processCount", 1) + options.set_preference("general.useragent.override", get_random_ua()) + options.set_preference("permissions.default.image", 2) # Block images for speed + options.set_preference("dom.webnotifications.enabled", False) # Try to initialize driver with explicit path to Firefox try: @@ -272,7 +289,7 @@ def get_article_with_selenium(url): except: pass # Continue even if wait times out - time.sleep(2) # Brief additional wait + time.sleep(random.uniform(1, 3)) # Random wait to mimic human behavior html = driver.page_source @@ -302,29 +319,38 @@ def get_article_with_selenium(url): def get_article_with_playwright(url): """ - Gets article text using Playwright with proper error handling. + Gets article text using Playwright with proper bot-detection evasion. """ try: from playwright.sync_api import sync_playwright with sync_playwright() as p: - # Use Chromium instead of Firefox for better compatibility + # Use Chromium with full browser context for UA spoofing browser = p.chromium.launch(headless=True, timeout=30000) - page = browser.new_page() - - # Set user agent to avoid bot detection - page.set_extra_http_headers( - { - "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36" - } + context = browser.new_context( + user_agent=get_random_ua(), + viewport={"width": 1920, "height": 1080}, + locale="en-US", + timezone_id="America/New_York", ) + page = context.new_page() - page.goto(url, wait_until="load") + # Additional headers for legitimacy + page.set_extra_http_headers({ + "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8", + "Accept-Language": "en-US,en;q=0.9", + "Accept-Encoding": "gzip, deflate, br", + "Connection": "keep-alive", + "Upgrade-Insecure-Requests": "1", + }) + + page.goto(url, wait_until="domcontentloaded", timeout=30000) # Wait for content to load - time.sleep(3) + time.sleep(random.uniform(2, 4)) html = page.content() + context.close() browser.close() # Parse with Newspaper4k @@ -353,11 +379,16 @@ def pull_article(link, source, title=None, save_to_file=True): ) as f: return f.read() + # Random delay before fetching to avoid rate-limiting / bot detection + time.sleep(random.uniform(0.5, 2)) + text = "" try: - # Try newspaper4k first - article = newspaper.article(link) + # Try newspaper4k first with proper User-Agent to bypass bot detection + ua = get_random_ua() + config = Config(browser_user_agent=ua) + article = newspaper.article(link, browser_user_agent=ua) article.download() article.parse() text = article.text