NewsArchiverV2/ap_processor.py
Jarian Cottingham a7936b8b11 fix: OPE hardening - logging, Docker, deps, scheduler, cleanup
- #12: Remove duplicate logging.basicConfig() from 10 modules
- #15: Remove redundant import re in rebuild_database.py
- #17: rglob('*') → rglob('*.html/json/txt/xml/md') for speed
- #18: Dockerfile individual COPY → glob COPY *.py/*.json + .dockerignore
- #19: Remove deprecated docker-compose version field
- #20: Pin requirements.txt versions (flask, requests, etc.)
- #22: SIGALRM → threading.Timer for multi-threaded safety
- #23: AP regex parsing → BeautifulSoup selectors
2026-07-05 04:14:05 +00:00

222 lines
6.3 KiB
Python

#!/usr/bin/env python3
"""AP News Processor for NewsArchiver - Phase 2.6
Processes AP News front page to extract article URLs and archive them.
Uses direct HTML parsing since AP doesn't provide RSS feeds.
"""
import argparse
import json
import logging
import sys
import time
from datetime import datetime
from pathlib import Path
from typing import List
try:
import requests
except ImportError:
print("WARNING: requests not installed. URL fetching may not work.")
print("Install with: pip install requests")
requests = None
try:
from bs4 import BeautifulSoup
except ImportError:
print("WARNING: beautifulsoup4 not installed. AP parsing may not work.")
print("Install with: pip install beautifulsoup4")
BeautifulSoup = None
try:
import sqlite3
except ImportError:
print("ERROR: sqlite3 is required (should be built-in)")
sys.exit(1)
try:
import storage_manager
except ImportError:
print("ERROR: storage_manager module not found")
sys.exit(1)
try:
from rss_processor import is_duplicate
except ImportError:
print("ERROR: rss_processor module not found")
sys.exit(1)
SCRIPT_DIR = Path(__file__).parent
ARCHIVE_DIR = SCRIPT_DIR / 'archival_data'
ARCHIVE_DIR.mkdir(exist_ok=True)
logger = logging.getLogger(__name__)
def fetch_ap_frontpage(timeout: int = 30) -> str:
"""Fetch AP News front page HTML.
Args:
timeout: Request timeout in seconds
Returns:
HTML string or empty string on failure
"""
if requests is None:
logger.error("requests library not available")
return ""
url = "https://apnews.com"
try:
headers = {
'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36'
}
response = requests.get(url, timeout=timeout, headers=headers)
response.raise_for_status()
logger.info("Fetched AP front page: %s", url)
return response.text
except Exception as e:
logger.error("Failed to fetch AP front page: %s", str(e))
return ""
def extract_article_links(html: str) -> List[str]:
"""Extract AP article URLs from HTML using BeautifulSoup.
Args:
html: Raw HTML string
Returns:
List of unique article URLs
"""
if BeautifulSoup is None:
logger.error("BeautifulSoup not available for link extraction")
return []
soup = BeautifulSoup(html, 'html.parser')
urls = []
for link in soup.find_all('a', href=True):
href = link['href']
if href.startswith('https://apnews.com/article/'):
if href not in urls:
urls.append(href)
logger.info("Extracted %d unique article links", len(urls))
return urls
def process_ap_articles(
output_dir: Path,
db_path: Path = ARCHIVE_DIR / 'cache.db',
dry_run: bool = False
) -> dict:
"""Process AP News articles from front page.
Args:
output_dir: Output directory for archived content
db_path: SQLite cache database path
dry_run: If True, preview without archiving
Returns:
Dictionary with results summary
"""
results = {
'source': 'Associated Press',
'processed': 0,
'archived': 0,
'skipped': 0,
'failed': 0,
'urls': []
}
html = fetch_ap_frontpage()
if not html:
logger.error("Failed to fetch AP front page")
return results
article_urls = extract_article_links(html)
for article_url in article_urls:
results['processed'] += 1
results['urls'].append(article_url)
if is_duplicate(article_url, 'Associated Press', db_path):
logger.debug("Skipping duplicate: %s", article_url[:60])
results['skipped'] += 1
continue
if dry_run:
logger.info("[DRY-RUN] Would archive: %s", article_url[:60])
results['archived'] += 1
continue
try:
from archive_engine import archive_and_extract
extraction_result = archive_and_extract(
article_url,
'Associated Press',
output_dir
)
if extraction_result['success']:
results['archived'] += 1
storage_manager.save_article('Associated Press', extraction_result['article_data'])
else:
results['failed'] += 1
logger.error("Failed to archive %s: %s", article_url[:60],
extraction_result.get('error', 'Unknown error'))
except Exception as e:
results['failed'] += 1
logger.error("Error processing %s: %s", article_url[:60], str(e))
time.sleep(0.5)
logger.info("AP processing complete: %d processed, %d archived, %d skipped, %d failed",
results['processed'], results['archived'], results['skipped'], results['failed'])
return results
def main():
parser = argparse.ArgumentParser(description='AP News Front Page Processor')
parser.add_argument('--output', type=Path, default=ARCHIVE_DIR,
help='Output directory for archived content')
parser.add_argument('--dry-run', action='store_true', help='Preview without archiving')
parser.add_argument('--verbose', action='store_true', help='Enable verbose logging')
args = parser.parse_args()
if args.verbose:
logger.setLevel(logging.DEBUG)
logger.info("=" * 60)
logger.info("AP News Front Page Processor - Phase 2.6")
logger.info("=" * 60)
storage_manager.initialize_storage()
results = process_ap_articles(args.output, ARCHIVE_DIR / 'cache.db', args.dry_run)
print("\n" + "=" * 60)
print("AP NEWS PROCESSING COMPLETE")
print("=" * 60)
print(f"Processed: {results['processed']}")
print(f"Archived: {results['archived']}")
print(f"Skipped: {results['skipped']}")
print(f"Failed: {results['failed']}")
if results['urls']:
print("\nArticle URLs:")
for url in results['urls'][:10]:
print(f" - {url[:70]}")
if len(results['urls']) > 10:
print(f" ... and {len(results['urls']) - 10} more")
print("=" * 60)
if __name__ == '__main__':
main()