233 lines
7.4 KiB
Python
233 lines
7.4 KiB
Python
import os
|
||
import json
|
||
import datetime
|
||
from pathlib import Path
|
||
from flask import Flask, request, jsonify
|
||
from datetime import timedelta
|
||
|
||
app = Flask(__name__)
|
||
|
||
try:
|
||
# Read the RSS feeds to get all available news outlets
|
||
rss_feeds_path = os.path.join(
|
||
os.path.dirname(__file__), "..", "scraper", "rss_feeds.json"
|
||
)
|
||
with open(rss_feeds_path, "r") as f:
|
||
rss_feeds = json.load(f)
|
||
|
||
# Get all available news outlets from the RSS feeds
|
||
NEWS_OUTLETS = list(rss_feeds["rss_feeds"].keys())
|
||
except Exception as e:
|
||
print(f"Error loading RSS feeds: {e}")
|
||
NEWS_OUTLETS = []
|
||
|
||
# Configuration - can be overridden by environment variable
|
||
ARTICLE_DIR = os.environ.get("ARTICLE_DIR", "/Volumes/WORKDIR/articles/")
|
||
|
||
# Validate article directory exists
|
||
if not os.path.exists(ARTICLE_DIR):
|
||
print(f"Warning: Article directory does not exist: {ARTICLE_DIR}")
|
||
|
||
|
||
def get_files_in_directory(directory_path):
|
||
"""Get all files in a directory recursively."""
|
||
files = []
|
||
for root, _, filenames in os.walk(directory_path):
|
||
for filename in filenames:
|
||
file_path = os.path.join(root, filename)
|
||
files.append(file_path)
|
||
return files
|
||
|
||
|
||
def get_file_create_time(file_path):
|
||
"""Get the creation time of a file."""
|
||
stat = os.stat(file_path)
|
||
# On Unix systems, we use the creation time (ctime) or modification time (mtime)
|
||
# On some systems like macOS, ctime might be more appropriate
|
||
return datetime.datetime.fromtimestamp(stat.st_ctime)
|
||
|
||
|
||
def is_file_in_time_range(file_path, start_time):
|
||
"""Check if a file was created after the start_time."""
|
||
try:
|
||
create_time = get_file_create_time(file_path)
|
||
return create_time >= start_time
|
||
except Exception:
|
||
# If we can't get the creation time, assume it's not in range
|
||
return False
|
||
|
||
|
||
def filter_articles_by_outlets(articles, outlets):
|
||
"""Filter articles based on specified outlets."""
|
||
if not outlets or not isinstance(outlets, list):
|
||
return articles
|
||
|
||
# Normalize outlet names for comparison (remove extra spaces, make lowercase)
|
||
normalized_outlets = [outlet.strip().lower() for outlet in outlets]
|
||
|
||
filtered_articles = []
|
||
for article in articles:
|
||
# Extract the news outlet from the file path
|
||
# Article paths are like: /path/to/articles/Reuters – Business News/article_name.txt
|
||
path_parts = Path(article).parts
|
||
if len(path_parts) >= 2:
|
||
outlet_name = path_parts[-2] # Outlet name is second to last part
|
||
if outlet_name.lower() in normalized_outlets:
|
||
filtered_articles.append(article)
|
||
|
||
return filtered_articles
|
||
|
||
|
||
def get_articles_in_time_range(time_range, outlets=None):
|
||
"""
|
||
Get articles within a specified time range from the article directory.
|
||
|
||
Args:
|
||
time_range (str): Time range ('hour', 'day', 'week', 'month')
|
||
outlets (list, optional): List of news outlets to filter by
|
||
|
||
Returns:
|
||
list: List of article file paths matching criteria
|
||
"""
|
||
# Validate that the article directory exists
|
||
if not os.path.exists(ARTICLE_DIR):
|
||
return []
|
||
|
||
# Get the start time based on the time range
|
||
now = datetime.datetime.now()
|
||
|
||
if time_range == "hour":
|
||
start_time = now - timedelta(hours=1)
|
||
elif time_range == "day":
|
||
start_time = now - timedelta(days=1)
|
||
elif time_range == "week":
|
||
start_time = now - timedelta(weeks=1)
|
||
elif time_range == "month":
|
||
start_time = now - timedelta(days=30)
|
||
else:
|
||
# Default to last hour if not specified correctly
|
||
start_time = now - timedelta(hours=1)
|
||
|
||
# Get all files under the article directory
|
||
all_articles = get_files_in_directory(ARTICLE_DIR)
|
||
|
||
# Filter for articles within time range
|
||
filtered_articles = []
|
||
for article_path in all_articles:
|
||
if is_file_in_time_range(article_path, start_time):
|
||
filtered_articles.append(article_path)
|
||
|
||
# Filter by outlets if provided
|
||
if outlets:
|
||
filtered_articles = filter_articles_by_outlets(filtered_articles, outlets)
|
||
|
||
return filtered_articles
|
||
|
||
|
||
@app.route("/articles", methods=["GET"])
|
||
def articles_endpoint():
|
||
"""HTTP endpoint to get articles."""
|
||
try:
|
||
# Get query parameters
|
||
time_range = request.args.get("time_range", "hour").lower()
|
||
outlet_param = request.args.get("outlets")
|
||
|
||
# Parse outlets if provided
|
||
outlets = None
|
||
if outlet_param:
|
||
outlets = [o.strip() for o in outlet_param.split(",") if o.strip()]
|
||
|
||
# Validate time range
|
||
valid_time_ranges = ["hour", "day", "week", "month"]
|
||
if time_range not in valid_time_ranges:
|
||
return jsonify(
|
||
{"error": f"Invalid time_range. Must be one of {valid_time_ranges}"}
|
||
), 400
|
||
|
||
# Get articles
|
||
articles = get_articles_in_time_range(time_range, outlets)
|
||
|
||
response_data = {
|
||
"articles": [
|
||
{
|
||
"path": article,
|
||
"name": os.path.basename(article),
|
||
"outlet": os.path.basename(os.path.dirname(article)),
|
||
"created_at": get_file_create_time(article).isoformat(),
|
||
}
|
||
for article in articles
|
||
],
|
||
"count": len(articles),
|
||
"time_range": time_range,
|
||
"outlets": outlets if outlets else "all",
|
||
}
|
||
|
||
return jsonify(response_data)
|
||
|
||
except Exception as e:
|
||
return jsonify({"error": str(e)}), 500
|
||
|
||
|
||
@app.route("/outlets", methods=["GET"])
|
||
def outlets_endpoint():
|
||
"""HTTP endpoint to get all available news outlets."""
|
||
try:
|
||
response_data = {"news_outlets": NEWS_OUTLETS, "count": len(NEWS_OUTLETS)}
|
||
return jsonify(response_data)
|
||
except Exception as e:
|
||
return jsonify({"error": str(e)}), 500
|
||
|
||
|
||
@app.route("/health", methods=["GET"])
|
||
def health_check():
|
||
"""Health check endpoint."""
|
||
return jsonify({"status": "healthy"})
|
||
|
||
|
||
@app.route("/article/content", methods=["GET"])
|
||
def article_content_endpoint():
|
||
"""HTTP endpoint to get the full content of a specific article by file path."""
|
||
try:
|
||
# Get the file path from query parameters
|
||
file_path = request.args.get("path")
|
||
|
||
if not file_path:
|
||
return jsonify({"error": "Missing 'path' parameter"}), 400
|
||
|
||
# Validate that the file exists and is within our article directory
|
||
# We'll ensure the file path is safe by checking it's under ARTICLE_DIR
|
||
file_path = os.path.abspath(file_path)
|
||
article_dir = os.path.abspath(ARTICLE_DIR)
|
||
|
||
if not file_path.startswith(article_dir):
|
||
return jsonify(
|
||
{"error": "Invalid file path - must be within article directory"}
|
||
), 400
|
||
|
||
if not os.path.exists(file_path):
|
||
return jsonify({"error": "Article file not found"}), 404
|
||
|
||
# Read the content of the article file
|
||
with open(file_path, "r", encoding="utf-8") as f:
|
||
content = f.read()
|
||
|
||
# Get basic info about the article
|
||
outlet = os.path.basename(os.path.dirname(file_path))
|
||
filename = os.path.basename(file_path)
|
||
|
||
response_data = {
|
||
"path": file_path,
|
||
"name": filename,
|
||
"outlet": outlet,
|
||
"content": content,
|
||
}
|
||
|
||
return jsonify(response_data)
|
||
|
||
except Exception as e:
|
||
return jsonify({"error": str(e)}), 500
|
||
|
||
|
||
if __name__ == "__main__":
|
||
app.run(host="0.0.0.0", port=5008, debug=True)
|