NewsArchiverV2/rebuild_database.py
Jarian Cottingham a7936b8b11 fix: OPE hardening - logging, Docker, deps, scheduler, cleanup
- #12: Remove duplicate logging.basicConfig() from 10 modules
- #15: Remove redundant import re in rebuild_database.py
- #17: rglob('*') → rglob('*.html/json/txt/xml/md') for speed
- #18: Dockerfile individual COPY → glob COPY *.py/*.json + .dockerignore
- #19: Remove deprecated docker-compose version field
- #20: Pin requirements.txt versions (flask, requests, etc.)
- #22: SIGALRM → threading.Timer for multi-threaded safety
- #23: AP regex parsing → BeautifulSoup selectors
2026-07-05 04:14:05 +00:00

247 lines
8.7 KiB
Python

#!/usr/bin/env python3
"""Rebuild NewsArchiver database from existing HTML files."""
import json
import logging
import re
import sqlite3
import sys
from datetime import datetime
from pathlib import Path
from typing import Optional
try:
from content_extractor import parse_article_from_html, get_html_from_url
from storage_manager import initialize_storage, save_article as storage_save_article
except ImportError as e:
print(f"ERROR: Required module not found: {e}")
sys.exit(1)
SCRIPT_DIR = Path(__file__).parent
ARCHIVE_DIR = SCRIPT_DIR / 'archival_data'
RSS_FEEDS_PATH = SCRIPT_DIR / 'rss_feeds.json'
logger = logging.getLogger(__name__)
def extract_urls_from_html(html_file: Path) -> list:
"""Extract article URLs from an archived HTML file.
Args:
html_file: Path to archived HTML file
Returns:
List of article URLs found in the file
"""
try:
content = html_file.read_text(encoding='utf-8')
urls = []
# Only extract main article URLs - skip tracking links, RSS feeds, author pages, etc.
article_patterns = [
r'href=["\']https?://[^"\']+\/article\/[^"\']+["\']',
r'href=["\']https?://[^"\']+\/news\/[^"\']+["\']',
r'href=["\']https?://[^"\']+\/stories\/[^"\']+["\']',
r'href=["\']https?://[^"\']+\/archive\/[^"\']+["\']',
]
for pattern in article_patterns:
matches = re.findall(pattern, content)
for match in matches:
url_match = re.search(r'href=["\']([^"\']+)["\']', match)
if url_match:
url = url_match.group(1)
# Skip common non-article URLs
skip_patterns = [
'rss', 'feed', 'author', 'authors', 'tags', 'category', 'search',
'about', 'contact', 'privacy', 'terms', 'faq', 'subscribe',
'signin', 'login', 'register', 'account', 'profile'
]
if not any(skip in url.lower() for skip in skip_patterns):
urls.append(url)
return list(set(urls)) # Remove duplicates
except Exception as e:
logger.error("Error reading %s: %s", html_file, str(e))
return []
def find_html_files(archive_dir: Path) -> list:
"""Find all HTML files in the archive directory.
Args:
archive_dir: Root archive directory
Returns:
List of HTML file paths
"""
html_files = []
try:
websites_dir = archive_dir / 'websites'
if not websites_dir.exists():
logger.warning("Websites directory not found: %s", websites_dir)
return []
for source_dir in sorted(websites_dir.iterdir()):
if not source_dir.is_dir():
continue
html_dir = source_dir / 'html'
if not html_dir.exists():
continue
for date_dir in sorted(html_dir.iterdir()):
if not date_dir.is_dir():
continue
for html_file in sorted(date_dir.glob('article_*.html')):
html_files.append(html_file)
logger.info("Found %d HTML files to process", len(html_files))
except Exception as e:
logger.error("Error finding HTML files: %s", str(e))
return html_files
def rebuild_database(archive_dir: Path, rss_feeds_path: Path) -> dict:
"""Rebuild the database from existing HTML files.
Args:
archive_dir: Root archive directory
rss_feeds_path: Path to RSS feeds JSON file
Returns:
Dictionary with rebuild results
"""
results = {
'html_files_processed': 0,
'articles_extracted': 0,
'articles_saved': 0,
'articles_failed': 0,
'errors': []
}
# Initialize database
logger.info("Initializing database at: %s", ARCHIVE_DIR / 'cache.db')
initialize_storage()
logger.info("Database initialized")
# Load RSS feeds
rss_feeds = {}
if rss_feeds_path.exists():
with open(rss_feeds_path, 'r', encoding='utf-8') as f:
rss_feeds = json.load(f)
# Find all HTML files
html_files = find_html_files(archive_dir)
if not html_files:
logger.warning("No HTML files found to process")
return results
# Process each HTML file
for html_file in html_files:
try:
results['html_files_processed'] += 1
# Extract URLs from HTML
urls = extract_urls_from_html(html_file)
if not urls:
logger.debug("No URLs found in %s", html_file.name)
continue
results['articles_extracted'] += len(urls)
# Process each URL
for url in urls:
try:
# Try to get source name from RSS feeds
source_name = None
for feed_name, feed_info in rss_feeds.items():
if feed_info.get('rss_url') and url.startswith(feed_info.get('rss_url', '')):
source_name = feed_name
break
# Try to infer source from URL
if not source_name:
for feed_name, feed_info in rss_feeds.items():
website = feed_info.get('source_website', '')
if website and website in url:
source_name = feed_name
break
if not source_name:
# Try to extract domain from URL
domain_match = re.search(r'https?://([^/]+)', url)
if domain_match:
domain = domain_match.group(1)
for feed_name, feed_info in rss_feeds.items():
website = feed_info.get('source_website', '')
if website and website in domain:
source_name = feed_name
break
# If still no source, skip
if not source_name:
logger.debug("No source found for %s, skipping", url[:60])
continue
# Extract content from HTML file
raw_html = html_file.read_text(encoding='utf-8')
article_data = parse_article_from_html(raw_html, url)
if article_data.error:
logger.warning("Failed to extract content from %s: %s", url[:60], article_data.error)
results['articles_failed'] += 1
continue
# Save to database
logger.info("Saving article: %s -> %s", url[:80], source_name)
storage_save_article(source_name, article_data)
results['articles_saved'] += 1
if results['articles_saved'] % 100 == 0:
logger.info("Saved %d articles so far", results['articles_saved'])
except Exception as e:
logger.error("Error processing URL %s: %s", url, str(e))
results['articles_failed'] += 1
results['errors'].append({
'url': url,
'error': str(e)
})
except Exception as e:
logger.error("Error processing file %s: %s", html_file, str(e))
results['errors'].append({
'file': str(html_file),
'error': str(e)
})
return results
if __name__ == '__main__':
logger.info("=" * 60)
logger.info("Rebuilding NewsArchiver Database")
logger.info("=" * 60)
results = rebuild_database(ARCHIVE_DIR, RSS_FEEDS_PATH)
logger.info("=" * 60)
logger.info("Rebuild Complete")
logger.info("=" * 60)
logger.info("HTML files processed: %d", results['html_files_processed'])
logger.info("Articles extracted: %d", results['articles_extracted'])
logger.info("Articles saved: %d", results['articles_saved'])
logger.info("Articles failed: %d", results['articles_failed'])
if results['errors']:
logger.info("Errors:")
for error in results['errors'][:20]:
logger.info(" - %s", error)