diff --git a/scraper/scraper.py b/scraper/scraper.py index 07e04a1..128ff0d 100644 --- a/scraper/scraper.py +++ b/scraper/scraper.py @@ -10,7 +10,7 @@ from selenium.webdriver.firefox.options import Options as FirefoxOptions from selenium.webdriver.common.by import By from selenium.webdriver.support.ui import WebDriverWait from selenium.webdriver.support import expected_conditions as EC -from concurrent.futures import ThreadPoolExecutor, as_completed +from concurrent.futures import ThreadPoolExecutor, as_completed, TimeoutError import nltk from nltk.downloader import Downloader @@ -76,8 +76,21 @@ def mine_all_articles(rss_feed_sources, limit=None): """Parse a single RSS feed with timeout and error handling""" try: logger.info(f"Parsing RSS feed: {data['rss_url']}") - # Add more aggressive timeout settings - feed = feedparser.parse(data["rss_url"], timeout=15) # 15 second timeout + # Add more aggressive timeout settings with fallback + # Use a wrapper to ensure we don't hang indefinitely + import signal + + def timeout_handler(signum, frame): + raise TimeoutError(f"Timeout parsing feed: {site}") + + # Set up signal-based timeout (this is a fallback for truly hanging requests) + old_handler = signal.signal(signal.SIGALRM, timeout_handler) + signal.alarm(10) # 10 second alarm + + feed = feedparser.parse(data["rss_url"], timeout=8) # 8 second timeout + signal.alarm(0) # Cancel the alarm + signal.signal(signal.SIGALRM, old_handler) + feed_entries = feed.entries[:limit] if limit else feed.entries entries = [] @@ -85,6 +98,9 @@ def mine_all_articles(rss_feed_sources, limit=None): if "link" in entry and "title" in entry: entries.append((site, entry.title, entry.link)) return entries + except TimeoutError as e: + logger.error(f"Timeout parsing RSS feed: {site} Error: {str(e)}") + return [] except Exception as e: logger.error(f"Error parsing RSS feed: {site} Error: {str(e)}") return []