#!/usr/bin/env python3 """ Health check script for the embedding pipeline with Prometheus metrics This script verifies that all components of the embedding pipeline are working correctly and exports metrics for Grafana visualization """ import os import json import requests import logging from datetime import datetime import sys from prometheus_client import start_http_server, Gauge, Counter, Histogram # Setup logging logging.basicConfig( level=logging.INFO, format='%(asctime)s - %(name)s - %(levelname)s - %(message)s' ) logger = logging.getLogger(__name__) # Prometheus metrics # Health status metrics health_status = Gauge('embedding_pipeline_health_status', 'Health status of pipeline components (1=healthy, 0=unhealthy)', ['component']) # Processing metrics articles_processed_total = Counter('embedding_pipeline_articles_processed_total', 'Total number of articles processed') articles_failed_total = Counter('embedding_pipeline_articles_failed_total', 'Total number of articles failed to process') processing_time_seconds = Histogram('embedding_pipeline_processing_time_seconds', 'Time spent processing articles') # Database metrics chromadb_collections_count = Gauge('embedding_pipeline_chromadb_collections', 'Number of collections in ChromaDB') # Cache metrics cache_entries_count = Gauge('embedding_pipeline_cache_entries', 'Number of entries in cache file') def check_chromadb_connection(): """Check if ChromaDB is accessible""" try: import chromadb CHROMADB_HOST = os.getenv("CHROMADB_HOST", "example.com") CHROMADB_PORT = int(os.getenv("CHROMADB_PORT", "8000")) client = chromadb.HttpClient(host=CHROMADB_HOST, port=CHROMADB_PORT) # Test connection by getting all collections collections = client.list_collections() chromadb_collections_count.set(len(collections)) logger.info(f"✓ ChromaDB connection successful. Found {len(collections)} collections") health_status.labels(component='chromadb').set(1) return True except Exception as e: logger.error(f"✗ ChromaDB connection failed: {e}") health_status.labels(component='chromadb').set(0) return False def check_ai_server_connection(): """Check if AI server is accessible""" try: AI_SERVER_HOST = os.getenv("AI_SERVER_HOST", "example.com") AI_SERVER_PORT = int(os.getenv("AI_SERVER_PORT", "4000")) AI_SERVER_URL = f"http://{AI_SERVER_HOST}:{AI_SERVER_PORT}/v1/embeddings" # Test by sending a simple request start_time = datetime.now() response = requests.post( AI_SERVER_URL, json={ "input": "test", "model": "text-embedding-3-small" }, timeout=10 ) response.raise_for_status() end_time = datetime.now() response_time = (end_time - start_time).total_seconds() logger.info(f"✓ AI server connection successful. Response time: {response_time:.2f}s") health_status.labels(component='ai_server').set(1) return True except Exception as e: logger.error(f"✗ AI server connection failed: {e}") health_status.labels(component='ai_server').set(0) return False def check_scraper_directory(): """Check if scraper directory exists and has articles""" try: scraper_articles_dir = "/scraper/articles" if not os.path.exists(scraper_articles_dir): logger.error(f"✗ Scraper directory does not exist: {scraper_articles_dir}") health_status.labels(component='scraper_dir').set(0) return False # Check if there are any JSON files json_files = [] for root, dirs, files in os.walk(scraper_articles_dir): for file in files: if file.endswith('.json'): json_files.append(os.path.join(root, file)) if json_files: logger.info(f"✓ Scraper directory accessible. Found {len(json_files)} article files") health_status.labels(component='scraper_dir').set(1) return True else: logger.warning(f"⚠ Scraper directory exists but no JSON files found") health_status.labels(component='scraper_dir').set(1) # Directory exists, just no files yet return True except Exception as e: logger.error(f"✗ Error checking scraper directory: {e}") health_status.labels(component='scraper_dir').set(0) return False def check_cache_file(): """Check if cache file is accessible""" try: CACHE_FILE = os.getenv("CACHE_FILE", "processed_articles_cache.json") if os.path.exists(CACHE_FILE): logger.info("✓ Cache file exists") # Try to read it with open(CACHE_FILE, 'r') as f: cache_data = json.load(f) cache_entries_count.set(len(cache_data)) logger.info(f"✓ Cache file readable. Contains {len(cache_data)} entries") return True else: logger.info("✓ Cache file does not exist (this is normal for first run)") cache_entries_count.set(0) return True except Exception as e: logger.error(f"✗ Error reading cache file: {e}") return False def collect_pipeline_metrics(): """Collect and export all pipeline metrics""" try: # This function would be called by the main pipeline to collect metrics logger.info("Collecting pipeline metrics...") # In a real implementation, this would be called by the main processing loop return True except Exception as e: logger.error(f"Error collecting pipeline metrics: {e}") return False def main(): """Run all health checks and start Prometheus server""" # Start Prometheus metrics server on port 8001 try: start_http_server(8001) logger.info("Prometheus metrics server started on port 8001") except Exception as e: logger.error(f"Failed to start Prometheus server: {e}") logger.info("Starting embedding pipeline health check...") checks = [ check_chromadb_connection, check_ai_server_connection, check_scraper_directory, check_cache_file ] results = [] for check in checks: try: result = check() results.append(result) except Exception as e: logger.error(f"Error running {check.__name__}: {e}") results.append(False) # Summary passed = sum(results) total = len(results) logger.info(f"\nHealth Check Summary: {passed}/{total} checks passed") # Set overall health status overall_health = 1 if passed == total else 0 health_status.labels(component='overall').set(overall_health) if passed == total: logger.info("✓ All health checks passed - pipeline is ready to run") return 0 else: logger.error("✗ Some health checks failed - pipeline may not work correctly") return 1 if __name__ == "__main__": sys.exit(main())