Improve logging in scraper with timestamps and focused new file notifications

This commit is contained in:
Jarian Cottingham 2026-01-31 11:16:05 -06:00
parent 936dd4dc28
commit 03b728d172
2 changed files with 111 additions and 71 deletions

View File

@ -11,6 +11,8 @@ import feedparser
import time import time
import os import os
import requests import requests
import logging
from datetime import datetime
from selenium import webdriver from selenium import webdriver
from selenium.webdriver.firefox.options import Options as FirefoxOptions from selenium.webdriver.firefox.options import Options as FirefoxOptions
from selenium.webdriver.common.by import By from selenium.webdriver.common.by import By
@ -20,6 +22,14 @@ from concurrent.futures import ThreadPoolExecutor, as_completed
import nltk import nltk
from nltk.downloader import Downloader from nltk.downloader import Downloader
# Setup logging with timestamps
logging.basicConfig(
level=logging.INFO,
format='%(asctime)s - %(levelname)s - %(message)s',
datefmt='%Y-%m-%d %H:%M:%S'
)
logger = logging.getLogger(__name__)
# Robust file path handling - try multiple locations # Robust file path handling - try multiple locations
def get_feed_file_path(): def get_feed_file_path():
"""Get the RSS feed file path, trying multiple locations.""" """Get the RSS feed file path, trying multiple locations."""
@ -56,30 +66,30 @@ def load_rss_feed_sources(feed_file=FEED_FILE):
""" """
Loads the RSS feed sources from a JSON file. Loads the RSS feed sources from a JSON file.
""" """
print(f"Loading RSS feed sources from {feed_file}...") logger.info(f"Loading RSS feed sources from {feed_file}...")
# Debug: Print current working directory # Debug: Print current working directory
print(f"Current working directory: {os.getcwd()}") logger.debug(f"Current working directory: {os.getcwd()}")
try: try:
with open(feed_file, "r", encoding="utf-8") as f: with open(feed_file, "r", encoding="utf-8") as f:
data = json.load(f) data = json.load(f)
print(f"Successfully loaded {feed_file}") logger.info(f"Successfully loaded {feed_file}")
print(f"Data type: {type(data)}") logger.debug(f"Data type: {type(data)}")
if isinstance(data, dict) and "rss_feeds" in data: if isinstance(data, dict) and "rss_feeds" in data:
print(f"Found rss_feeds section with {len(data['rss_feeds'])} sources") logger.info(f"Found rss_feeds section with {len(data['rss_feeds'])} sources")
return data return data
else: else:
print(f"Warning: Unexpected data structure. Data keys: {list(data.keys()) if isinstance(data, dict) else 'Not a dict'}") logger.warning(f"Unexpected data structure. Data keys: {list(data.keys()) if isinstance(data, dict) else 'Not a dict'}")
return {} return {}
except FileNotFoundError: except FileNotFoundError:
print(f"{feed_file} not found, returning empty dict.") logger.error(f"{feed_file} not found, returning empty dict.")
return {} return {}
except json.JSONDecodeError as e: except json.JSONDecodeError as e:
print(f"Error decoding {feed_file}: {e}, returning empty dict.") logger.error(f"Error decoding {feed_file}: {e}, returning empty dict.")
return {} return {}
except Exception as e: except Exception as e:
print(f"Unexpected error loading {feed_file}: {e}") logger.error(f"Unexpected error loading {feed_file}: {e}")
return {} return {}
@ -92,17 +102,17 @@ def mine_all_articles(rss_feed_sources, limit=None):
# Check if rss_feed_sources is a valid dict with rss_feeds key # Check if rss_feed_sources is a valid dict with rss_feeds key
if not isinstance(rss_feed_sources, dict): if not isinstance(rss_feed_sources, dict):
print(f"Warning: rss_feed_sources is not a dict, it's {type(rss_feed_sources)}") logger.warning(f"rss_feed_sources is not a dict, it's {type(rss_feed_sources)}")
return all_links return all_links
if "rss_feeds" not in rss_feed_sources: if "rss_feeds" not in rss_feed_sources:
print("Warning: rss_feeds key not found in rss_feed_sources") logger.warning("rss_feeds key not found in rss_feed_sources")
return all_links return all_links
sources = rss_feed_sources["rss_feeds"] sources = rss_feed_sources["rss_feeds"]
for site, data in sources.items(): for site, data in sources.items():
print(f"Parsing RSS feed: {data['rss_url']}") logger.info(f"Parsing RSS feed: {data['rss_url']}")
try: try:
feed = feedparser.parse(data["rss_url"]) feed = feedparser.parse(data["rss_url"])
feed_entries = feed.entries[:limit] if limit else feed.entries feed_entries = feed.entries[:limit] if limit else feed.entries
@ -111,7 +121,7 @@ def mine_all_articles(rss_feed_sources, limit=None):
if "link" in entry and "title" in entry: if "link" in entry and "title" in entry:
all_links.append((site, entry.title, entry.link)) all_links.append((site, entry.title, entry.link))
except Exception as e: except Exception as e:
print(f"Error parsing RSS feed: {site} Error: {str(e)}") logger.error(f"Error parsing RSS feed: {site} Error: {str(e)}")
return all_links return all_links
@ -148,7 +158,9 @@ def save_article_to_file(article, filename, source="Unfiltered"):
with open(file_path, "w", encoding="utf-8") as f: with open(file_path, "w", encoding="utf-8") as f:
f.write(f"SOURCE:{source}\n") f.write(f"SOURCE:{source}\n")
f.write(article) f.write(article)
print(f"Article saved to {file_path}")
# Only log when a new file is actually created (not cached)
logger.info(f"New article saved: {safe_filename} from {source}")
def get_article_with_selenium(url): def get_article_with_selenium(url):
@ -185,10 +197,11 @@ def get_article_with_selenium(url):
# Parse with Newspaper4k # Parse with Newspaper4k
article = newspaper.article(url, input_html=html, language="en") article = newspaper.article(url, input_html=html, language="en")
article.nlp() article.nlp()
logger.info(f"Successfully extracted article with Selenium from {url}")
return article.text return article.text
except Exception as e: except Exception as e:
print(f"Selenium failed for {url}: {str(e)}") logger.error(f"Selenium failed for {url}: {str(e)}")
return "" return ""
finally: finally:
# Always quit the driver # Always quit the driver
@ -229,10 +242,11 @@ def get_article_with_playwright(url):
# Parse with Newspaper4k # Parse with Newspaper4k
article = newspaper.article(url, input_html=html, language="en") article = newspaper.article(url, input_html=html, language="en")
article.nlp() article.nlp()
logger.info(f"Successfully extracted article with Playwright from {url}")
return article.text return article.text
except Exception as e: except Exception as e:
print(f"Playwright failed for {url}: {str(e)}") logger.error(f"Playwright failed for {url}: {str(e)}")
return "" return ""
@ -245,7 +259,7 @@ def pull_article(link, source, title=None, save_to_file=True):
# Check if already cached # Check if already cached
if os.path.exists(os.path.join("articles", source, safe_filename)): if os.path.exists(os.path.join("articles", source, safe_filename)):
print(f"Article already cached: {filename}") logger.info(f"Article already cached: {filename}")
with open( with open(
os.path.join("articles", source, safe_filename), "r", encoding="utf-8" os.path.join("articles", source, safe_filename), "r", encoding="utf-8"
) as f: ) as f:
@ -264,31 +278,31 @@ def pull_article(link, source, title=None, save_to_file=True):
raise ValueError( raise ValueError(
"\tArticle text too short, falling back to Playwright/Selenium." "\tArticle text too short, falling back to Playwright/Selenium."
) )
print(f"\tSuccessfully pulled article with newspaper4k from {link}") logger.info(f"Successfully pulled article with newspaper4k from {link}")
except Exception as e: except Exception as e:
print( logger.warning(
f"\tnewspaper4k extraction failed for {link}: {e}, falling back to Playwright." f"newspaper4k extraction failed for {link}: {e}, falling back to Playwright."
) )
try: try:
text = get_article_with_playwright(link) text = get_article_with_playwright(link)
print(f"\t\tSuccessfully pulled article from {link} with Playwright") logger.info(f"Successfully pulled article from {link} with Playwright")
if not text or len(text) < 200: if not text or len(text) < 200:
print(f"\t\tPlaywright article too short, falling back to Selenium.") logger.warning(f"Playwright article too short, falling back to Selenium.")
# Fallback to Selenium with better error handling # Fallback to Selenium with better error handling
text = get_article_with_selenium(link) text = get_article_with_selenium(link)
print(f"\t\t\tSuccessfully pulled article from {link} with Selenium") logger.info(f"Successfully pulled article from {link} with Selenium")
except Exception as e: except Exception as e:
print(f"\t\tPlaywright failed for {link}: {e}") logger.error(f"Playwright failed for {link}: {e}")
# Fallback to Selenium # Fallback to Selenium
try: try:
text = get_article_with_selenium(link) text = get_article_with_selenium(link)
print(f"\t\tSuccessfully pulled article from {link} with Selenium") logger.info(f"Successfully pulled article from {link} with Selenium")
except Exception as e: except Exception as e:
print(f"\t\tSelenium failed for {link}: {e}") logger.error(f"Selenium failed for {link}: {e}")
return "" return ""
if save_to_file: if save_to_file:
@ -309,7 +323,7 @@ def safe_pull_articles(article_list):
for i in range(0, len(article_list), batch_size): for i in range(0, len(article_list), batch_size):
batch = article_list[i : i + batch_size] batch = article_list[i : i + batch_size]
print(f"Processing batch {i // batch_size + 1} with {len(batch)} articles") logger.info(f"Processing batch {i // batch_size + 1} with {len(batch)} articles")
# Use ThreadPoolExecutor instead of ProcessPoolExecutor to avoid # Use ThreadPoolExecutor instead of ProcessPoolExecutor to avoid
# process termination issues with browser automation # process termination issues with browser automation
@ -324,7 +338,7 @@ def safe_pull_articles(article_list):
result = future.result(timeout=120) # 2 minute timeout result = future.result(timeout=120) # 2 minute timeout
results.append(result) results.append(result)
except Exception as e: except Exception as e:
print(f"Error in pull_article: {e}") logger.error(f"Error in pull_article: {e}")
errors.append(e) errors.append(e)
# Add a small delay between batches to reduce system load # Add a small delay between batches to reduce system load
@ -390,8 +404,8 @@ def main():
Main scraping function for cron execution. Main scraping function for cron execution.
This replaces the infinite while loop with a single execution. This replaces the infinite while loop with a single execution.
""" """
print("=========================================") logger.info("=========================================")
print("Starting new scraping iteration...") logger.info("Starting new scraping iteration...")
try: try:
# Pull the RSS feed sources from the JSON file # Pull the RSS feed sources from the JSON file
@ -405,17 +419,17 @@ def main():
random.shuffle(rss_feed_links) random.shuffle(rss_feed_links)
print(f"Found {len(rss_feed_links)} articles to process") logger.info(f"Found {len(rss_feed_links)} articles to process")
if not rss_feed_links: if not rss_feed_links:
print("No articles found") logger.info("No articles found")
return return
# Process articles with better error handling and resource management # Process articles with better error handling and resource management
results, errors = safe_pull_articles(rss_feed_links) results, errors = safe_pull_articles(rss_feed_links)
print(f"Attempted to Pull {len(results)} articles in parallel.") logger.info(f"Attempted to Pull {len(results)} articles in parallel.")
print( logger.info(
f"Encountered {len(errors)} errors during article pulling. " f"Encountered {len(errors)} errors during article pulling. "
+ "Outputting errors to a local file." + "Outputting errors to a local file."
) )
@ -425,28 +439,28 @@ def main():
with open("errors.txt", "w", encoding="utf-8") as f: with open("errors.txt", "w", encoding="utf-8") as f:
for error in errors: for error in errors:
f.write(str(error) + "\n") f.write(str(error) + "\n")
print(f"Errors logged to errors.txt") logger.info(f"Errors logged to errors.txt")
# Print all results to a log file # Print all results to a log file
with open("results.txt", "w", encoding="utf-8") as f: with open("results.txt", "w", encoding="utf-8") as f:
for result in results: for result in results:
f.write(result + "\n") f.write(result + "\n")
print("All articles pulled successfully.") logger.info("All articles pulled successfully.")
# Gather and send new articles to webhook # Gather and send new articles to webhook
new_articles = gather_new_articles() new_articles = gather_new_articles()
if new_articles: if new_articles:
send_to_webhook(new_articles) send_to_webhook(new_articles)
else: else:
print("No new articles to send to webhook") logger.info("No new articles to send to webhook")
except Exception as e: except Exception as e:
print(f"Major error in main execution: {e}") logger.error(f"Major error in main execution: {e}")
import traceback import traceback
traceback.print_exc() traceback.print_exc()
raise # Re-raise to ensure the script exits with error code raise # Re-raise to ensure the script exits with error code
print("Scraping completed successfully.") logger.info("Scraping completed successfully.")
if __name__ == "__main__": if __name__ == "__main__":

View File

@ -3,6 +3,8 @@ import json
import feedparser import feedparser
import time import time
import os import os
import logging
from datetime import datetime
from selenium import webdriver from selenium import webdriver
from selenium.webdriver.firefox.options import Options as FirefoxOptions from selenium.webdriver.firefox.options import Options as FirefoxOptions
from selenium.webdriver.common.by import By from selenium.webdriver.common.by import By
@ -12,6 +14,14 @@ from concurrent.futures import ThreadPoolExecutor, as_completed
import nltk import nltk
from nltk.downloader import Downloader from nltk.downloader import Downloader
# Setup logging with timestamps
logging.basicConfig(
level=logging.INFO,
format='%(asctime)s - %(levelname)s - %(message)s',
datefmt='%Y-%m-%d %H:%M:%S'
)
logger = logging.getLogger(__name__)
FEED_FILE = os.getenv("FEED_FILE", "./rss_short_feed.json") FEED_FILE = os.getenv("FEED_FILE", "./rss_short_feed.json")
# Ensure necessary NLTK resources are downloaded # Ensure necessary NLTK resources are downloaded
@ -26,16 +36,18 @@ def load_rss_feed_sources(feed_file=FEED_FILE):
""" """
Loads the RSS feed sources from a JSON file. Loads the RSS feed sources from a JSON file.
""" """
print("Loading RSS feed sources from rss_feeds.json...") logger.info(f"Loading RSS feed sources from {feed_file}...")
try: try:
with open(FEED_FILE, "r", encoding="utf-8") as f: with open(feed_file, "r", encoding="utf-8") as f:
return json.load(f) data = json.load(f)
logger.info(f"Successfully loaded {feed_file}")
return data
except FileNotFoundError: except FileNotFoundError:
print(FEED_FILE + " not found, returning empty list.") logger.error(f"{feed_file} not found, returning empty list.")
return [] return []
except json.JSONDecodeError: except json.JSONDecodeError:
print("Error decoding " + FEED_FILE + " , returning empty list.") logger.error(f"Error decoding {feed_file} , returning empty list.")
return [] return []
@ -45,10 +57,20 @@ def mine_all_articles(rss_feed_sources, limit=None):
Returns a list of (site, title, link) tuples. Returns a list of (site, title, link) tuples.
""" """
all_links = [] all_links = []
# Check if rss_feed_sources is a valid dict with rss_feeds key
if not isinstance(rss_feed_sources, dict):
logger.warning(f"rss_feed_sources is not a dict, it's {type(rss_feed_sources)}")
return all_links
if "rss_feeds" not in rss_feed_sources:
logger.warning("rss_feeds key not found in rss_feed_sources")
return all_links
sources = rss_feed_sources["rss_feeds"] sources = rss_feed_sources["rss_feeds"]
for site, data in sources.items(): for site, data in sources.items():
print(f"Parsing RSS feed: {data['rss_url']}") logger.info(f"Parsing RSS feed: {data['rss_url']}")
try: try:
feed = feedparser.parse(data["rss_url"]) feed = feedparser.parse(data["rss_url"])
feed_entries = feed.entries[:limit] if limit else feed.entries feed_entries = feed.entries[:limit] if limit else feed.entries
@ -57,7 +79,7 @@ def mine_all_articles(rss_feed_sources, limit=None):
if "link" in entry and "title" in entry: if "link" in entry and "title" in entry:
all_links.append((site, entry.title, entry.link)) all_links.append((site, entry.title, entry.link))
except Exception as e: except Exception as e:
print(f"Error parsing RSS feed: {site} Error: {str(e)}") logger.error(f"Error parsing RSS feed: {site} Error: {str(e)}")
return all_links return all_links
@ -94,7 +116,9 @@ def save_article_to_file(article, filename, source="Unfiltered"):
with open(file_path, "w", encoding="utf-8") as f: with open(file_path, "w", encoding="utf-8") as f:
f.write(f"SOURCE:{source}\n") f.write(f"SOURCE:{source}\n")
f.write(article) f.write(article)
print(f"Article saved to {file_path}")
# Only log when a new file is actually created (not cached)
logger.info(f"New article saved: {safe_filename} from {source}")
def get_article_with_selenium(url): def get_article_with_selenium(url):
@ -131,10 +155,11 @@ def get_article_with_selenium(url):
# Parse with Newspaper4k # Parse with Newspaper4k
article = newspaper.article(url, input_html=html, language="en") article = newspaper.article(url, input_html=html, language="en")
article.nlp() article.nlp()
logger.info(f"Successfully extracted article with Selenium from {url}")
return article.text return article.text
except Exception as e: except Exception as e:
print(f"Selenium failed for {url}: {str(e)}") logger.error(f"Selenium failed for {url}: {str(e)}")
return "" return ""
finally: finally:
# Always quit the driver # Always quit the driver
@ -175,10 +200,11 @@ def get_article_with_playwright(url):
# Parse with Newspaper4k # Parse with Newspaper4k
article = newspaper.article(url, input_html=html, language="en") article = newspaper.article(url, input_html=html, language="en")
article.nlp() article.nlp()
logger.info(f"Successfully extracted article with Playwright from {url}")
return article.text return article.text
except Exception as e: except Exception as e:
print(f"Playwright failed for {url}: {str(e)}") logger.error(f"Playwright failed for {url}: {str(e)}")
return "" return ""
@ -191,7 +217,7 @@ def pull_article(link, source, title=None, save_to_file=True):
# Check if already cached # Check if already cached
if os.path.exists(os.path.join("articles", source, safe_filename)): if os.path.exists(os.path.join("articles", source, safe_filename)):
print(f"Article already cached: {filename}") logger.info(f"Article already cached: {filename}")
with open( with open(
os.path.join("articles", source, safe_filename), "r", encoding="utf-8" os.path.join("articles", source, safe_filename), "r", encoding="utf-8"
) as f: ) as f:
@ -210,31 +236,31 @@ def pull_article(link, source, title=None, save_to_file=True):
raise ValueError( raise ValueError(
"\tArticle text too short, falling back to Playwright/Selenium." "\tArticle text too short, falling back to Playwright/Selenium."
) )
print(f"\tSuccessfully pulled article with newspaper4k from {link}") logger.info(f"Successfully pulled article with newspaper4k from {link}")
except Exception as e: except Exception as e:
print( logger.warning(
f"\tnewspaper4k extraction failed for {link}: {e}, falling back to Playwright." f"newspaper4k extraction failed for {link}: {e}, falling back to Playwright."
) )
try: try:
text = get_article_with_playwright(link) text = get_article_with_playwright(link)
print(f"\t\tSuccessfully pulled article from {link} with Playwright") logger.info(f"Successfully pulled article from {link} with Playwright")
if not text or len(text) < 200: if not text or len(text) < 200:
print(f"\t\tPlaywright article too short, falling back to Selenium.") logger.warning(f"Playwright article too short, falling back to Selenium.")
# Fallback to Selenium with better error handling # Fallback to Selenium with better error handling
text = get_article_with_selenium(link) text = get_article_with_selenium(link)
print(f"\t\t\tSuccessfully pulled article from {link} with Selenium") logger.info(f"Successfully pulled article from {link} with Selenium")
except Exception as e: except Exception as e:
print(f"\t\tPlaywright failed for {link}: {e}") logger.error(f"Playwright failed for {link}: {e}")
# Fallback to Selenium # Fallback to Selenium
try: try:
text = get_article_with_selenium(link) text = get_article_with_selenium(link)
print(f"\t\tSuccessfully pulled article from {link} with Selenium") logger.info(f"Successfully pulled article from {link} with Selenium")
except Exception as e: except Exception as e:
print(f"\t\tSelenium failed for {link}: {e}") logger.error(f"Selenium failed for {link}: {e}")
return "" return ""
if save_to_file: if save_to_file:
@ -255,7 +281,7 @@ def safe_pull_articles(article_list):
for i in range(0, len(article_list), batch_size): for i in range(0, len(article_list), batch_size):
batch = article_list[i : i + batch_size] batch = article_list[i : i + batch_size]
print(f"Processing batch {i // batch_size + 1} with {len(batch)} articles") logger.info(f"Processing batch {i // batch_size + 1} with {len(batch)} articles")
# Use ThreadPoolExecutor instead of ProcessPoolExecutor to avoid # Use ThreadPoolExecutor instead of ProcessPoolExecutor to avoid
# process termination issues with browser automation # process termination issues with browser automation
@ -270,7 +296,7 @@ def safe_pull_articles(article_list):
result = future.result(timeout=120) # 2 minute timeout result = future.result(timeout=120) # 2 minute timeout
results.append(result) results.append(result)
except Exception as e: except Exception as e:
print(f"Error in pull_article: {e}") logger.error(f"Error in pull_article: {e}")
errors.append(e) errors.append(e)
# Add a small delay between batches to reduce system load # Add a small delay between batches to reduce system load
@ -284,8 +310,8 @@ def main():
Main scraping loop. Main scraping loop.
""" """
while True: while True:
print("=========================================") logger.info("=========================================")
print("Starting new scraping iteration...") logger.info("Starting new scraping iteration...")
try: try:
# Pull the RSS feed sources from the JSON file # Pull the RSS feed sources from the JSON file
@ -299,18 +325,18 @@ def main():
random.shuffle(rss_feed_links) random.shuffle(rss_feed_links)
print(f"Found {len(rss_feed_links)} articles to process") logger.info(f"Found {len(rss_feed_links)} articles to process")
if not rss_feed_links: if not rss_feed_links:
print("No articles found, sleeping for 15 minutes") logger.info("No articles found, sleeping for 15 minutes")
time.sleep(15 * 60) time.sleep(15 * 60)
continue continue
# Process articles with better error handling and resource management # Process articles with better error handling and resource management
results, errors = safe_pull_articles(rss_feed_links) results, errors = safe_pull_articles(rss_feed_links)
print(f"Attempted to Pull {len(results)} articles in parallel.") logger.info(f"Attempted to Pull {len(results)} articles in parallel.")
print( logger.info(
f"Encountered {len(errors)} errors during article pulling. " f"Encountered {len(errors)} errors during article pulling. "
+ "Outputting errors to a local file." + "Outputting errors to a local file."
) )
@ -320,20 +346,20 @@ def main():
with open("errors.txt", "w", encoding="utf-8") as f: with open("errors.txt", "w", encoding="utf-8") as f:
for error in errors: for error in errors:
f.write(str(error) + "\n") f.write(str(error) + "\n")
print(f"Errors logged to errors.txt") logger.info(f"Errors logged to errors.txt")
# Print all results to a log file # Print all results to a log file
with open("results.txt", "w", encoding="utf-8") as f: with open("results.txt", "w", encoding="utf-8") as f:
for result in results: for result in results:
f.write(result + "\n") f.write(result + "\n")
print("All articles pulled successfully.") logger.info("All articles pulled successfully.")
except Exception as e: except Exception as e:
print(f"Major error in main loop: {e}") logger.error(f"Major error in main loop: {e}")
# Continue to next iteration even if there's a major error # Continue to next iteration even if there's a major error
# Sleep for a while before the next iteration # Sleep for a while before the next iteration
print("Sleeping for 15 minutes before the next iteration...") logger.info("Sleeping for 15 minutes before the next iteration...")
time.sleep(15 * 60) # Sleep for 15 minutes time.sleep(15 * 60) # Sleep for 15 minutes