""" File processing module for FactsDB service Handles detection and transformation of different file types to text """ import os import pdfminer.high_level from bs4 import BeautifulSoup from newspaper import Article from typing import Optional, Dict, Any import re class FileProcessor: """Handles file type detection and text extraction""" def __init__(self): pass def detect_file_type(self, file_path: str) -> str: """Detect file type based on extension""" _, ext = os.path.splitext(file_path) return ext.lower()[1:] # Remove the dot def extract_text_from_file(self, file_path: str) -> str: """Extract text from file based on its type""" file_type = self.detect_file_type(file_path) try: if file_type in ['txt', 'md', 'log']: return self._extract_text_from_text_file(file_path) elif file_type in ['html', 'htm']: return self._extract_text_from_html_file(file_path) elif file_type == 'pdf': return self._extract_text_from_pdf_file(file_path) elif file_type in ['xml', 'json']: return self._extract_text_from_structured_file(file_path) else: raise Exception(f"Unsupported file type: .{file_type}") except Exception as e: raise Exception(f"Error processing file {file_path}: {str(e)}") def _extract_text_from_text_file(self, file_path: str) -> str: """Extract text from plain text file""" with open(file_path, 'r', encoding='utf-8', errors='ignore') as f: return f.read() def _extract_text_from_html_file(self, file_path: str) -> str: """Extract text from HTML file""" with open(file_path, 'r', encoding='utf-8', errors='ignore') as f: content = f.read() soup = BeautifulSoup(content, 'html.parser') # Remove script and style elements for script in soup(["script", "style"]): script.decompose() # Get text and clean it up text = soup.get_text() # Break into lines and remove leading/trailing space lines = (line.strip() for line in text.splitlines()) # Break multi-headlines into a line each chunks = (phrase.strip() for line in lines for phrase in line.split(" ")) # Drop blank lines text = ' '.join(chunk for chunk in chunks if chunk) return text def _extract_text_from_pdf_file(self, file_path: str) -> str: """Extract text from PDF file""" try: # Use pdfminer to extract text with open(file_path, 'rb') as file: text = pdfminer.high_level.extract_text(file) return text except Exception as e: # Fallback to basic PDF reading if pdfminer fails raise Exception(f"PDF extraction failed: {str(e)}") def _extract_text_from_structured_file(self, file_path: str) -> str: """Extract text from structured files like XML or JSON""" with open(file_path, 'r', encoding='utf-8', errors='ignore') as f: content = f.read() # For JSON, extract key fields if self.detect_file_type(file_path) == 'json': import json try: data = json.loads(content) # Convert to text representation return str(data) except: return content else: # For XML, extract text content soup = BeautifulSoup(content, 'xml') return soup.get_text() def is_supported_file_type(self, file_path: str) -> bool: """Check if file type is supported""" supported_types = ['txt', 'md', 'log', 'html', 'htm', 'pdf', 'xml', 'json'] file_type = self.detect_file_type(file_path) return file_type in supported_types def get_file_info(self, file_path: str) -> Dict[str, Any]: """Get information about a file""" return { 'path': file_path, 'name': os.path.basename(file_path), 'size': os.path.getsize(file_path), 'type': self.detect_file_type(file_path), 'is_supported': self.is_supported_file_type(file_path) }