import subprocess import os from pathlib import Path import logging from typing import Optional logger = logging.getLogger(__name__) try: from playwright.sync_api import sync_playwright PLAYWRIGHT_AVAILABLE = True except ImportError: PLAYWRIGHT_AVAILABLE = False logger.debug("Playwright not available") # Cache the SingleFile path _SINGLEFILE_PATH: Optional[str] = None def _get_singlefile_path() -> Optional[str]: """Get the path to SingleFile CLI executable.""" global _SINGLEFILE_PATH if _SINGLEFILE_PATH is not None: return _SINGLEFILE_PATH # Check PATH first single_file_path = os.environ.get('PATH', '').split(os.pathsep) for path in single_file_path: candidate = Path(path) / 'single-file' if candidate.is_file(): _SINGLEFILE_PATH = str(candidate) logger.debug(f"Found SingleFile in PATH: {_SINGLEFILE_PATH}") return _SINGLEFILE_PATH # Check common locations common_locations = [ '/home/user/.local/bin/single-file', '/usr/local/bin/single-file', '/usr/bin/single-file', '/home/user/.npm/_global/bin/single-file', ] for candidate in common_locations: if Path(candidate).is_file(): _SINGLEFILE_PATH = candidate logger.debug(f"Found SingleFile at: {_SINGLEFILE_PATH}") return _SINGLEFILE_PATH logger.error("SingleFile CLI not found. Please install with: npm install -g single-file") return None def check_singlefile_available() -> bool: """Check if SingleFile CLI is available""" single_file_path = _get_singlefile_path() if not single_file_path: return False try: result = subprocess.run( [single_file_path, '--version'], capture_output=True, text=True, timeout=10 ) return result.returncode == 0 except (subprocess.TimeoutExpired, FileNotFoundError, OSError): logger.error(f"SingleFile CLI at {single_file_path} is not executable") return False def is_error_page(html_content: str) -> bool: """Check if HTML content is an error page (403, 404, etc.) Args: html_content: HTML content to check Returns: True if error page detected, False otherwise """ error_patterns = [ '403 error', '403 forbidden', 'access denied', 'request blocked', 'cloudfront', '404 error', 'page not found', 'error 404', 'server error', '503 service unavailable', ] html_lower = html_content.lower() return any(pattern in html_lower for pattern in error_patterns) def validate_archived_html(output_path: Path) -> bool: """Validate that archived HTML is not an error page. Args: output_path: Path to the archived HTML file Returns: True if valid, False if error page detected """ try: if not output_path.exists(): logger.warning(f"Archived file not found: {output_path}") return False content = output_path.read_text(encoding='utf-8', errors='ignore') if is_error_page(content): logger.warning(f"Archived file contains error page: {output_path}") return False if len(content) < 1000: logger.warning(f"Archived file too small (likely incomplete): {output_path}") return False return True except Exception as e: logger.error(f"Error validating archived HTML {output_path}: {e}") return False def archive_page_with_singlefile( url: str, output_path: Path, extract_content: bool = True ) -> bool: """Archive a web page using SingleFile CLI Args: url: URL to archive output_path: Output file path for the archived HTML extract_content: Whether to use extract-content mode (ignored - SingleFile always extracts) Returns: True if successful, False otherwise """ single_file_path = _get_singlefile_path() if not single_file_path: logger.error("SingleFile CLI not available") return False cmd = [ single_file_path, url, str(output_path), '--browser-headless=true', '--browser-wait-delay=5000', '--browser-load-max-time=120000' ] logger.debug(f"Archiving {url} with SingleFile at {single_file_path}") try: result = subprocess.run(cmd, capture_output=True, text=True, timeout=180) if result.returncode == 0: logger.info(f"Successfully archived {url} to {output_path}") if result.stdout: logger.debug(f"SingleFile output: {result.stdout}") # Validate the archived HTML if not validate_archived_html(output_path): logger.warning(f"Archived HTML validation failed for {url}, will use fallback") return False return True else: error_msg = result.stderr if result.stderr else result.stdout logger.error(f"SingleFile failed for {url}: {error_msg}") return False except subprocess.TimeoutExpired: logger.error(f"SingleFile timed out for {url}") return False except FileNotFoundError: logger.error(f"SingleFile CLI executable not found at: {single_file_path}") return False except Exception as e: logger.error(f"Unexpected error archiving {url} with SingleFile: {e}") return False def archive_page_with_singlefile_no_extraction(url: str, output_path: Path) -> bool: """Archive a web page using SingleFile CLI without content extraction This preserves the full original HTML structure including navigation, ads, etc. Args: url: URL to archive output_path: Output file path for the archived HTML Returns: True if successful, False otherwise """ return archive_page_with_singlefile(url, output_path, extract_content=False) def archive_page_with_singlefile_extract(url: str, output_path: Path) -> bool: """Archive a web page using SingleFile CLI with content extraction This extracts only the main content, removing navigation, ads, and sidebars. Args: url: URL to archive output_path: Output file path for the archived HTML Returns: True if successful, False otherwise """ return archive_page_with_singlefile(url, output_path, extract_content=True) def archive_page_with_playwright(url: str, output_path: Path) -> bool: """Archive a web page using Playwright This visits the URL with a headless browser, waits for content to load, and saves the full HTML page. Args: url: URL to archive output_path: Output file path for the archived HTML Returns: True if successful, False otherwise """ if not PLAYWRIGHT_AVAILABLE: logger.error("Playwright not available. Install with: pip install playwright") return False try: with sync_playwright() as p: browser = p.chromium.launch(headless=True) page = browser.new_page() page.goto(url, wait_until='networkidle', timeout=120000) content = page.content() browser.close() output_path.parent.mkdir(parents=True, exist_ok=True) output_path.write_text(content, encoding='utf-8') logger.info("Successfully archived %s to %s using Playwright", url[:60], output_path) return True except Exception as e: logger.error("Playwright archiving failed for %s: %s", url, str(e)) return False