getting rid of fault preview logic

This commit is contained in:
Jarian Cottingham 2025-10-14 21:58:35 -05:00
parent 5941230412
commit eefe459952
2 changed files with 3 additions and 144 deletions

View File

@ -1,25 +1,18 @@
import os import os
import subprocess import subprocess
import json import json
import requests
from PIL import Image
from io import BytesIO
from typing import List, Dict, Any from typing import List, Dict, Any
import uuid import uuid
import urllib.parse import urllib.parse
from PIL import Image, ImageDraw, ImageFont
import requests
try:
import mercury
except ImportError:
mercury = None
from io import BytesIO
class ArchiveParser: class ArchiveParser:
def __init__(self): def __init__(self):
"""Initialize the ArchiveParser with archive_dir from environment variable.""" """Initialize the ArchiveParser with archive_dir from environment variable."""
self.archive_dir = os.environ.get("ARCHIVE_DIR", "/default/archive/path") self.archive_dir = os.environ.get("ARCHIVE_DIR", "/default/archive/path")
self.cache_dir = os.path.join(os.path.dirname(__file__), "..", "..", "cache")
os.makedirs(self.cache_dir, exist_ok=True)
def get_posts(self, count: int, start_index: int = 0) -> List[Dict[str, Any]]: def get_posts(self, count: int, start_index: int = 0) -> List[Dict[str, Any]]:
""" """
@ -140,142 +133,9 @@ class ArchiveParser:
"/Reddit_Logo.webp" # Default to the logo in website directory "/Reddit_Logo.webp" # Default to the logo in website directory
) )
# Generate HTML preview for the post URL
post["image"] = self._generate_preview_image(post["url"], post["id"])
return post return post
except Exception as e: except Exception as e:
print(f"Error converting index to post: {e}") print(f"Error converting index to post: {e}")
# Return a minimal post if there's an error # Return a minimal post if there's an error
raise e raise e
def _generate_preview_image(self, url: str, post_id: str) -> str:
"""
Generate a preview image for the HTML page at the given URL.
This implementation extracts the largest image from the HTML page content,
falling back to a placeholder if no suitable images are found.
Args:
url (str): The URL to generate preview for
post_id (str): The ID of the post
Returns:
str: Path to the generated preview image
"""
try:
# Validate the URL
parsed_url = urllib.parse.urlparse(url)
if not parsed_url.scheme or not parsed_url.netloc:
return "/Reddit_Logo.webp"
# Generate filename based on post ID
filename = f"{post_id}.webp"
image_path = os.path.join(self.cache_dir, filename)
# If image already exists, return its path
if os.path.exists(image_path):
return f"/cache/{filename}"
# Try to extract and use the largest image from the webpage content
image_extracted = self._extract_largest_image_from_webpage(url, image_path)
if not image_extracted:
# Fall back to creating a placeholder image
self._create_placeholder_image(url, image_path)
return f"/cache/{filename}"
except Exception as e:
print(f"Error generating preview for URL {url}: {e}")
# Return default fallback image in case of any error
return "/Reddit_Logo.webp"
def _create_placeholder_image(self, url: str, save_path: str):
"""
Create a simple placeholder image when real preview isn't possible.
Args:
url (str): The URL that was attempted
save_path (str): Path to save the image
"""
try:
# Create a simple placeholder image
img = Image.new("RGB", (800, 600), color=(73, 109, 137))
draw = ImageDraw.Draw(img)
# Try to get default font or use None (which will fall back to default)
try:
font = ImageFont.load_default()
except Exception:
font = None
# Draw some text on the image
draw.text(
(10, 10), "HTML Preview Placeholder", fill=(255, 255, 0), font=font
)
draw.text(
(10, 40),
f"URL: {url[:50]}{'...' if len(url) > 50 else ''}",
fill=(255, 255, 255),
font=font,
)
# Add instructions
draw.text(
(10, 80),
"Largest image from webpage would appear here",
fill=(255, 255, 255),
font=font,
)
draw.text(
(10, 100),
"This is a placeholder image",
fill=(255, 255, 255),
font=font,
)
# Save as webp image
img.save(save_path, "WEBP", quality=80)
except Exception as e:
print(f"Error creating placeholder image: {e}")
# If everything fails, we still have the default fallback
return
def _extract_largest_image_from_webpage(self, url: str, save_path: str) -> bool:
"""
Extract the largest image from a webpage using Mercury parser.
Args:
url (str): URL of the webpage to extract images from
save_path (str): Path where to save the extracted image
Returns:
bool: True if successful, False otherwise
"""
try:
if not mercury:
return False
# Use mercury to parse the webpage
parsed = mercury.parse(url)
# Check if mercury found a lead image
if hasattr(parsed, "lead_image_url") and parsed.lead_image_url:
image_url = parsed.lead_image_url
# Download and save the image
img_response = requests.get(image_url, timeout=10)
img_response.raise_for_status()
# Save as webp with quality 80
img = Image.open(BytesIO(img_response.content))
img.save(save_path, "WEBP", quality=80)
return True
except Exception as e:
print(f"Error extracting image using Mercury: {e}")
return False
return False

View File

@ -3,4 +3,3 @@ Flask-CORS
Pillow Pillow
beautifulsoup4 beautifulsoup4
requests requests
mercury