fix singlefile archived links on the UI

This commit is contained in:
Jarian Cottingham 2026-03-31 09:55:44 -05:00
parent 0f1b2741db
commit 09d81e8cda
3 changed files with 2114 additions and 322 deletions

View File

@ -18,38 +18,35 @@ from content_extractor import ArticleData
try: try:
import feedgenerator import feedgenerator
FEEDGENERATOR_AVAILABLE = True FEEDGENERATOR_AVAILABLE = True
except ImportError: except ImportError:
FEEDGENERATOR_AVAILABLE = False FEEDGENERATOR_AVAILABLE = False
SCRIPT_DIR = Path(__file__).parent SCRIPT_DIR = Path(__file__).parent
ARCHIVE_DIR = SCRIPT_DIR / 'archival_data' ARCHIVE_DIR = SCRIPT_DIR / "archival_data"
ARCHIVE_DIR.mkdir(exist_ok=True) ARCHIVE_DIR.mkdir(exist_ok=True)
logging.basicConfig( logging.basicConfig(
level=logging.INFO, level=logging.INFO,
format='%(asctime)s - %(levelname)s - %(message)s', format="%(asctime)s - %(levelname)s - %(message)s",
handlers=[ handlers=[
logging.StreamHandler(sys.stdout), logging.StreamHandler(sys.stdout),
logging.FileHandler(ARCHIVE_DIR / 'processing.log', encoding='utf-8') logging.FileHandler(ARCHIVE_DIR / "processing.log", encoding="utf-8"),
] ],
) )
logger = logging.getLogger(__name__) logger = logging.getLogger(__name__)
DB_PATH = ARCHIVE_DIR / 'cache.db' DB_PATH = ARCHIVE_DIR / "cache.db"
WEBSITES_DIR = ARCHIVE_DIR / 'websites' WEBSITES_DIR = ARCHIVE_DIR / "websites"
def _get_db_connection() -> sqlite3.Connection: def _get_db_connection() -> sqlite3.Connection:
"""Get database connection with row factory.""" """Get database connection with row factory."""
conn = sqlite3.connect( conn = sqlite3.connect(DB_PATH, timeout=30.0, isolation_level=None)
DB_PATH,
timeout=30.0,
isolation_level=None
)
conn.row_factory = sqlite3.Row conn.row_factory = sqlite3.Row
conn.execute('PRAGMA journal_mode=WAL') conn.execute("PRAGMA journal_mode=WAL")
conn.execute('PRAGMA busy_timeout=30000') conn.execute("PRAGMA busy_timeout=30000")
return conn return conn
@ -58,7 +55,7 @@ def _init_database() -> None:
with _get_db_connection() as conn: with _get_db_connection() as conn:
cursor = conn.cursor() cursor = conn.cursor()
cursor.execute(''' cursor.execute("""
CREATE TABLE IF NOT EXISTS articles ( CREATE TABLE IF NOT EXISTS articles (
id INTEGER PRIMARY KEY AUTOINCREMENT, id INTEGER PRIMARY KEY AUTOINCREMENT,
source_name TEXT NOT NULL, source_name TEXT NOT NULL,
@ -77,18 +74,18 @@ def _init_database() -> None:
created_at TIMESTAMP DEFAULT CURRENT_TIMESTAMP, created_at TIMESTAMP DEFAULT CURRENT_TIMESTAMP,
updated_at TIMESTAMP DEFAULT CURRENT_TIMESTAMP updated_at TIMESTAMP DEFAULT CURRENT_TIMESTAMP
) )
''') """)
cursor.execute(''' cursor.execute("""
CREATE TABLE IF NOT EXISTS article_archives ( CREATE TABLE IF NOT EXISTS article_archives (
article_url TEXT PRIMARY KEY, article_url TEXT PRIMARY KEY,
source_name TEXT, source_name TEXT,
archive_file_path TEXT, archive_file_path TEXT,
created_at TIMESTAMP DEFAULT CURRENT_TIMESTAMP created_at TIMESTAMP DEFAULT CURRENT_TIMESTAMP
) )
''') """)
cursor.execute(''' cursor.execute("""
CREATE TABLE IF NOT EXISTS processing_log ( CREATE TABLE IF NOT EXISTS processing_log (
id INTEGER PRIMARY KEY AUTOINCREMENT, id INTEGER PRIMARY KEY AUTOINCREMENT,
source_name TEXT, source_name TEXT,
@ -97,22 +94,34 @@ def _init_database() -> None:
status TEXT, status TEXT,
message TEXT message TEXT
) )
''') """)
cursor.execute('CREATE INDEX IF NOT EXISTS idx_articles_source ON articles(source_name)') cursor.execute(
cursor.execute('CREATE INDEX IF NOT EXISTS idx_articles_url ON articles(article_url)') "CREATE INDEX IF NOT EXISTS idx_articles_source ON articles(source_name)"
cursor.execute('CREATE INDEX IF NOT EXISTS idx_articles_status ON articles(status)') )
cursor.execute(
"CREATE INDEX IF NOT EXISTS idx_articles_url ON articles(article_url)"
)
cursor.execute(
"CREATE INDEX IF NOT EXISTS idx_articles_status ON articles(status)"
)
# Add extraction_method column if it doesn't exist # Add extraction_method column if it doesn't exist
try: try:
cursor.execute('ALTER TABLE articles ADD COLUMN extraction_method TEXT') cursor.execute("ALTER TABLE articles ADD COLUMN extraction_method TEXT")
conn.commit() conn.commit()
except sqlite3.OperationalError: except sqlite3.OperationalError:
pass pass
cursor.execute('CREATE INDEX IF NOT EXISTS idx_articles_extraction_method ON articles(extraction_method)') cursor.execute(
cursor.execute('CREATE INDEX IF NOT EXISTS idx_article_archives_url ON article_archives(article_url)') "CREATE INDEX IF NOT EXISTS idx_articles_extraction_method ON articles(extraction_method)"
cursor.execute('CREATE INDEX IF NOT EXISTS idx_article_archives_source ON article_archives(source_name)') )
cursor.execute(
"CREATE INDEX IF NOT EXISTS idx_article_archives_url ON article_archives(article_url)"
)
cursor.execute(
"CREATE INDEX IF NOT EXISTS idx_article_archives_source ON article_archives(source_name)"
)
logger.debug("Database initialized at %s", DB_PATH) logger.debug("Database initialized at %s", DB_PATH)
@ -128,8 +137,8 @@ def _ensure_directory_structure(source_name: str, date_str: str) -> tuple:
Tuple of (html_dir, articles_dir) Path objects Tuple of (html_dir, articles_dir) Path objects
""" """
source_dir = WEBSITES_DIR / source_name source_dir = WEBSITES_DIR / source_name
html_dir = source_dir / 'html' / date_str html_dir = source_dir / "html" / date_str
articles_dir = source_dir / 'articles' / date_str articles_dir = source_dir / "articles" / date_str
html_dir.mkdir(parents=True, exist_ok=True) html_dir.mkdir(parents=True, exist_ok=True)
articles_dir.mkdir(parents=True, exist_ok=True) articles_dir.mkdir(parents=True, exist_ok=True)
@ -147,21 +156,22 @@ def _get_next_file_index(html_dir: Path, articles_dir: Path) -> int:
Returns: Returns:
Next available index (1-indexed) Next available index (1-indexed)
""" """
def get_max_index(directory: Path, extension: str) -> int: def get_max_index(directory: Path, extension: str) -> int:
max_idx = 0 max_idx = 0
if directory.exists(): if directory.exists():
for file in directory.glob(f'*{extension}'): for file in directory.glob(f"*{extension}"):
try: try:
name = file.stem name = file.stem
if name.startswith('article_'): if name.startswith("article_"):
idx = int(name.replace('article_', '')) idx = int(name.replace("article_", ""))
max_idx = max(max_idx, idx) max_idx = max(max_idx, idx)
except ValueError: except ValueError:
continue continue
return max_idx return max_idx
html_idx = get_max_index(html_dir, '.html') html_idx = get_max_index(html_dir, ".html")
json_idx = get_max_index(articles_dir, '.json') json_idx = get_max_index(articles_dir, ".json")
return max(html_idx, json_idx) + 1 return max(html_idx, json_idx) + 1
@ -179,14 +189,16 @@ def _log_processing(source_name: str, action: str, status: str, message: str) ->
with _get_db_connection() as conn: with _get_db_connection() as conn:
cursor = conn.cursor() cursor = conn.cursor()
cursor.execute( cursor.execute(
'INSERT INTO processing_log (source_name, action, status, message) VALUES (?, ?, ?, ?)', "INSERT INTO processing_log (source_name, action, status, message) VALUES (?, ?, ?, ?)",
(source_name, action, status, message) (source_name, action, status, message),
) )
except Exception as e: except Exception as e:
logger.error("Failed to log processing: %s", str(e)) logger.error("Failed to log processing: %s", str(e))
def _save_archive_mapping(article_url: str, source_name: str, archive_file_path: str) -> None: def _save_archive_mapping(
article_url: str, source_name: str, archive_file_path: str
) -> None:
"""Save mapping between article URL and archive file path. """Save mapping between article URL and archive file path.
Args: Args:
@ -197,10 +209,13 @@ def _save_archive_mapping(article_url: str, source_name: str, archive_file_path:
try: try:
with _get_db_connection() as conn: with _get_db_connection() as conn:
cursor = conn.cursor() cursor = conn.cursor()
cursor.execute(''' cursor.execute(
"""
INSERT OR REPLACE INTO article_archives (article_url, source_name, archive_file_path) INSERT OR REPLACE INTO article_archives (article_url, source_name, archive_file_path)
VALUES (?, ?, ?) VALUES (?, ?, ?)
''', (article_url, source_name, archive_file_path)) """,
(article_url, source_name, archive_file_path),
)
logger.debug("Saved archive mapping: %s -> %s", article_url, archive_file_path) logger.debug("Saved archive mapping: %s -> %s", article_url, archive_file_path)
except Exception as e: except Exception as e:
logger.error("Failed to save archive mapping: %s", str(e)) logger.error("Failed to save archive mapping: %s", str(e))
@ -222,107 +237,119 @@ def save_article(source_name: str, article_data: ArticleData) -> str:
publish_date = article_data.publish_date publish_date = article_data.publish_date
if publish_date: if publish_date:
try: try:
date_obj = datetime.fromisoformat(publish_date.replace('Z', '+00:00')) date_obj = datetime.fromisoformat(publish_date.replace("Z", "+00:00"))
date_str = date_obj.strftime('%Y-%m-%d') date_str = date_obj.strftime("%Y-%m-%d")
except (ValueError, AttributeError): except (ValueError, AttributeError):
date_str = datetime.now().strftime('%Y-%m-%d') date_str = datetime.now().strftime("%Y-%m-%d")
else: else:
date_str = datetime.now().strftime('%Y-%m-%d') date_str = datetime.now().strftime("%Y-%m-%d")
html_dir, articles_dir = _ensure_directory_structure(source_name, date_str) html_dir, articles_dir = _ensure_directory_structure(source_name, date_str)
file_index = _get_next_file_index(html_dir, articles_dir) file_index = _get_next_file_index(html_dir, articles_dir)
file_prefix = f'article_{file_index:03d}' file_prefix = f"article_{file_index:03d}"
archive_file_path = html_dir / f'{file_prefix}.html' archive_file_path = html_dir / f"{file_prefix}.html"
metadata_file_path = articles_dir / f'{file_prefix}.json' metadata_file_path = articles_dir / f"{file_prefix}.json"
if article_data.raw_html: if article_data.raw_html:
with open(archive_file_path, 'w', encoding='utf-8') as f: with open(archive_file_path, "w", encoding="utf-8") as f:
f.write(article_data.raw_html) f.write(article_data.raw_html)
metadata = { metadata = {
'id': file_index, "id": file_index,
'source_name': source_name, "source_name": source_name,
'url': article_data.url, "url": article_data.url,
'title': article_data.title, "title": article_data.title,
'author': article_data.author, "author": article_data.author,
'publish_date': article_data.publish_date, "publish_date": article_data.publish_date,
'content_text': article_data.content_text, "content_text": article_data.content_text,
'content_html': article_data.content_html, "content_html": article_data.content_html,
'tags': article_data.tags, "tags": article_data.tags,
'language': article_data.language, "language": article_data.language,
'extraction_method': article_data.extraction_method, "extraction_method": article_data.extraction_method,
'archive_file': f'{file_prefix}.html', "archive_file": f"{file_prefix}.html",
'metadata_file': f'{file_prefix}.json', "metadata_file": f"{file_prefix}.json",
'saved_at': datetime.now().isoformat() "saved_at": datetime.now().isoformat(),
} }
with open(metadata_file_path, 'w', encoding='utf-8') as f: with open(metadata_file_path, "w", encoding="utf-8") as f:
json.dump(metadata, f, indent=2, ensure_ascii=False) json.dump(metadata, f, indent=2, ensure_ascii=False)
with _get_db_connection() as conn: with _get_db_connection() as conn:
cursor = conn.cursor() cursor = conn.cursor()
try: try:
cursor.execute(''' cursor.execute(
"""
INSERT OR IGNORE INTO articles INSERT OR IGNORE INTO articles
(source_name, article_url, article_guid, title, author, publish_date, (source_name, article_url, article_guid, title, author, publish_date,
content_text, content_html, archive_file_path, metadata_file_path, status, extraction_method) content_text, content_html, archive_file_path, metadata_file_path, status, extraction_method)
VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)
''', ( """,
source_name, (
article_data.url, source_name,
getattr(article_data, 'guid', None), article_data.url,
article_data.title, getattr(article_data, "guid", None),
article_data.author, article_data.title,
article_data.publish_date, article_data.author,
article_data.content_text, article_data.publish_date,
article_data.content_html, article_data.content_text,
str(archive_file_path), article_data.content_html,
str(metadata_file_path), str(archive_file_path.relative_to(ARCHIVE_DIR)),
'archived' if not article_data.error else 'failed', str(metadata_file_path.relative_to(ARCHIVE_DIR)),
article_data.extraction_method "archived" if not article_data.error else "failed",
)) article_data.extraction_method,
),
)
except sqlite3.IntegrityError: except sqlite3.IntegrityError:
cursor.execute(''' cursor.execute(
"""
UPDATE articles UPDATE articles
SET title = ?, author = ?, publish_date = ?, SET title = ?, author = ?, publish_date = ?,
content_text = ?, content_html = ?, content_text = ?, content_html = ?,
archive_file_path = ?, metadata_file_path = ?, archive_file_path = ?, metadata_file_path = ?,
status = ?, extraction_method = ?, updated_at = CURRENT_TIMESTAMP status = ?, extraction_method = ?, updated_at = CURRENT_TIMESTAMP
WHERE article_url = ? AND source_name = ? WHERE article_url = ? AND source_name = ?
''', ( """,
article_data.title, (
article_data.author, article_data.title,
article_data.publish_date, article_data.author,
article_data.content_text, article_data.publish_date,
article_data.content_html, article_data.content_text,
str(archive_file_path), article_data.content_html,
str(metadata_file_path), str(archive_file_path.relative_to(ARCHIVE_DIR)),
'archived' if not article_data.error else 'failed', str(metadata_file_path.relative_to(ARCHIVE_DIR)),
article_data.extraction_method, "archived" if not article_data.error else "failed",
article_data.url, article_data.extraction_method,
source_name article_data.url,
)) source_name,
),
)
_save_archive_mapping(article_data.url, source_name, str(archive_file_path)) _save_archive_mapping(
article_data.url,
source_name,
str(archive_file_path.relative_to(ARCHIVE_DIR)),
)
_log_processing( _log_processing(
source_name, source_name,
'save_article', "save_article",
'success', "success",
f'Saved article: {article_data.url} -> {metadata_file_path.name}' f"Saved article: {article_data.url} -> {metadata_file_path.name}",
) )
logger.info("Article saved: %s -> %s", article_data.url, metadata_file_path.name) logger.info(
"Article saved: %s -> %s", article_data.url, metadata_file_path.name
)
return f"Article saved: {metadata_file_path.name}" return f"Article saved: {metadata_file_path.name}"
except Exception as e: except Exception as e:
error_msg = f"Failed to save article: {str(e)}" error_msg = f"Failed to save article: {str(e)}"
logger.error(error_msg) logger.error(error_msg)
_log_processing(source_name, 'save_article', 'error', error_msg) _log_processing(source_name, "save_article", "error", error_msg)
return error_msg return error_msg
@ -341,38 +368,51 @@ def get_article(source_name: str, article_id: int) -> Optional[ArticleData]:
with _get_db_connection() as conn: with _get_db_connection() as conn:
cursor = conn.cursor() cursor = conn.cursor()
cursor.execute(''' cursor.execute(
"""
SELECT * FROM articles SELECT * FROM articles
WHERE id = ? AND source_name = ? WHERE id = ? AND source_name = ?
''', (article_id, source_name)) """,
(article_id, source_name),
)
row = cursor.fetchone() row = cursor.fetchone()
if not row: if not row:
return None return None
archive_path = Path(row['archive_file_path']) if row['archive_file_path'] else None archive_path = (
Path(row["archive_file_path"]) if row["archive_file_path"] else None
)
# Handle both absolute and relative paths
if archive_path:
if not archive_path.is_absolute():
archive_path = ARCHIVE_DIR / archive_path
# Return relative path for web interface
archive_file_path = str(archive_path.relative_to(ARCHIVE_DIR))
else:
archive_file_path = None
archive_content = None archive_content = None
if archive_path and archive_path.exists(): if archive_path and archive_path.exists():
archive_content = archive_path.read_text(encoding='utf-8') archive_content = archive_path.read_text(encoding="utf-8")
article = ArticleData( article = ArticleData(
url=row['article_url'], url=row["article_url"],
title=row['title'], title=row["title"],
author=row['author'], author=row["author"],
publish_date=row['publish_date'], publish_date=row["publish_date"],
content_text=row['content_text'], content_text=row["content_text"],
content_html=row['content_html'], content_html=row["content_html"],
raw_html=archive_content, raw_html=archive_content,
archive_file_path=str(archive_path) if archive_path else None, archive_file_path=archive_file_path,
tags=None, tags=None,
language=None, language=None,
metadata=None, metadata=None,
extraction_method=None, extraction_method=None,
error=row['error_message'], error=row["error_message"],
guid=row['article_guid'], guid=row["article_guid"],
id=row['id'], id=row["id"],
source_name=source_name source_name=source_name,
) )
return article return article
@ -382,7 +422,9 @@ def get_article(source_name: str, article_id: int) -> Optional[ArticleData]:
return None return None
def get_articles_by_source(source_name: str, limit: int = 50, offset: int = 0) -> List[ArticleData]: def get_articles_by_source(
source_name: str, limit: int = 50, offset: int = 0
) -> List[ArticleData]:
"""Get paginated articles for a source. """Get paginated articles for a source.
Args: Args:
@ -398,39 +440,52 @@ def get_articles_by_source(source_name: str, limit: int = 50, offset: int = 0) -
with _get_db_connection() as conn: with _get_db_connection() as conn:
cursor = conn.cursor() cursor = conn.cursor()
cursor.execute(''' cursor.execute(
"""
SELECT * FROM articles SELECT * FROM articles
WHERE source_name = ? WHERE source_name = ?
ORDER BY publish_date DESC, created_at DESC ORDER BY publish_date DESC, created_at DESC
LIMIT ? OFFSET ? LIMIT ? OFFSET ?
''', (source_name, limit, offset)) """,
(source_name, limit, offset),
)
rows = cursor.fetchall() rows = cursor.fetchall()
articles = [] articles = []
for row in rows: for row in rows:
archive_path = Path(row['archive_file_path']) if row['archive_file_path'] else None archive_path = (
Path(row["archive_file_path"]) if row["archive_file_path"] else None
)
# Handle both absolute and relative paths
if archive_path:
if not archive_path.is_absolute():
archive_path = ARCHIVE_DIR / archive_path
# Return relative path for web interface
archive_file_path = str(archive_path.relative_to(ARCHIVE_DIR))
else:
archive_file_path = None
archive_content = None archive_content = None
if archive_path and archive_path.exists(): if archive_path and archive_path.exists():
archive_content = archive_path.read_text(encoding='utf-8') archive_content = archive_path.read_text(encoding="utf-8")
article = ArticleData( article = ArticleData(
url=row['article_url'], url=row["article_url"],
title=row['title'], title=row["title"],
author=row['author'], author=row["author"],
publish_date=row['publish_date'], publish_date=row["publish_date"],
content_text=row['content_text'], content_text=row["content_text"],
content_html=row['content_html'], content_html=row["content_html"],
raw_html=archive_content, raw_html=archive_content,
archive_file_path=str(archive_path) if archive_path else None, archive_file_path=archive_file_path,
tags=None, tags=None,
language=None, language=None,
metadata=None, metadata=None,
extraction_method=None, extraction_method=None,
error=row['error_message'], error=row["error_message"],
guid=row['article_guid'], guid=row["article_guid"],
id=row['id'], id=row["id"],
source_name=source_name source_name=source_name,
) )
articles.append(article) articles.append(article)
@ -442,7 +497,9 @@ def get_articles_by_source(source_name: str, limit: int = 50, offset: int = 0) -
return [] return []
def update_article_status(source_name: str, article_url: str, status: str, error: str = None) -> None: def update_article_status(
source_name: str, article_url: str, status: str, error: str = None
) -> None:
"""Update article status in cache. """Update article status in cache.
Args: Args:
@ -456,11 +513,14 @@ def update_article_status(source_name: str, article_url: str, status: str, error
conn = _get_db_connection() conn = _get_db_connection()
cursor = conn.cursor() cursor = conn.cursor()
cursor.execute(''' cursor.execute(
"""
UPDATE articles UPDATE articles
SET status = ?, error_message = ?, updated_at = CURRENT_TIMESTAMP SET status = ?, error_message = ?, updated_at = CURRENT_TIMESTAMP
WHERE article_url = ? AND source_name = ? WHERE article_url = ? AND source_name = ?
''', (status, error, article_url, source_name)) """,
(status, error, article_url, source_name),
)
conn.commit() conn.commit()
conn.close() conn.close()
@ -468,16 +528,16 @@ def update_article_status(source_name: str, article_url: str, status: str, error
if error: if error:
_log_processing( _log_processing(
source_name, source_name,
'update_status', "update_status",
'error', "error",
f'Updated {article_url} status to {status}: {error}' f"Updated {article_url} status to {status}: {error}",
) )
else: else:
_log_processing( _log_processing(
source_name, source_name,
'update_status', "update_status",
'success', "success",
f'Updated {article_url} status to {status}' f"Updated {article_url} status to {status}",
) )
logger.info("Updated article status: %s -> %s", article_url, status) logger.info("Updated article status: %s -> %s", article_url, status)
@ -501,7 +561,8 @@ def get_source_stats(source_name: str) -> dict:
conn = _get_db_connection() conn = _get_db_connection()
cursor = conn.cursor() cursor = conn.cursor()
cursor.execute(''' cursor.execute(
"""
SELECT SELECT
COUNT(*) as total, COUNT(*) as total,
SUM(CASE WHEN status = 'archived' THEN 1 ELSE 0 END) as archived, SUM(CASE WHEN status = 'archived' THEN 1 ELSE 0 END) as archived,
@ -512,27 +573,29 @@ def get_source_stats(source_name: str) -> dict:
MAX(publish_date) as latest_article_date MAX(publish_date) as latest_article_date
FROM articles FROM articles
WHERE source_name = ? WHERE source_name = ?
''', (source_name,)) """,
(source_name,),
)
row = cursor.fetchone() row = cursor.fetchone()
conn.close() conn.close()
stats = { stats = {
'source_name': source_name, "source_name": source_name,
'total_articles': row['total'] or 0, "total_articles": row["total"] or 0,
'archived': row['archived'] or 0, "archived": row["archived"] or 0,
'failed': row['failed'] or 0, "failed": row["failed"] or 0,
'pending': row['pending'] or 0, "pending": row["pending"] or 0,
'first_archived': row['first_archived'], "first_archived": row["first_archived"],
'last_archived': row['last_archived'], "last_archived": row["last_archived"],
'latest_article_date': row['latest_article_date'] "latest_article_date": row["latest_article_date"],
} }
_log_processing( _log_processing(
source_name, source_name,
'get_stats', "get_stats",
'success', "success",
f'Stats: {stats["total_articles"]} total, {stats["archived"]} archived, {stats["failed"]} failed' f"Stats: {stats['total_articles']} total, {stats['archived']} archived, {stats['failed']} failed",
) )
return stats return stats
@ -540,11 +603,11 @@ def get_source_stats(source_name: str) -> dict:
except Exception as e: except Exception as e:
logger.error("Failed to get stats for %s: %s", source_name, str(e)) logger.error("Failed to get stats for %s: %s", source_name, str(e))
return { return {
'source_name': source_name, "source_name": source_name,
'total_articles': 0, "total_articles": 0,
'archived': 0, "archived": 0,
'failed': 0, "failed": 0,
'pending': 0 "pending": 0,
} }
@ -559,11 +622,11 @@ def get_all_sources() -> List[str]:
conn = _get_db_connection() conn = _get_db_connection()
cursor = conn.cursor() cursor = conn.cursor()
cursor.execute(''' cursor.execute("""
SELECT DISTINCT source_name FROM articles ORDER BY source_name SELECT DISTINCT source_name FROM articles ORDER BY source_name
''') """)
sources = [row['source_name'] for row in cursor.fetchall()] sources = [row["source_name"] for row in cursor.fetchall()]
conn.close() conn.close()
return sources return sources
@ -587,39 +650,52 @@ def get_latest_articles(limit: int = 50) -> List[ArticleData]:
conn = _get_db_connection() conn = _get_db_connection()
cursor = conn.cursor() cursor = conn.cursor()
cursor.execute(''' cursor.execute(
"""
SELECT * FROM articles SELECT * FROM articles
ORDER BY publish_date DESC, created_at DESC ORDER BY publish_date DESC, created_at DESC
LIMIT ? LIMIT ?
''', (limit,)) """,
(limit,),
)
rows = cursor.fetchall() rows = cursor.fetchall()
conn.close() conn.close()
articles = [] articles = []
for row in rows: for row in rows:
archive_path = Path(row['archive_file_path']) if row['archive_file_path'] else None archive_path = (
Path(row["archive_file_path"]) if row["archive_file_path"] else None
)
# Handle both absolute and relative paths
if archive_path:
if not archive_path.is_absolute():
archive_path = ARCHIVE_DIR / archive_path
# Return relative path for web interface
archive_file_path = str(archive_path.relative_to(ARCHIVE_DIR))
else:
archive_file_path = None
archive_content = None archive_content = None
if archive_path and archive_path.exists(): if archive_path and archive_path.exists():
archive_content = archive_path.read_text(encoding='utf-8') archive_content = archive_path.read_text(encoding="utf-8")
article = ArticleData( article = ArticleData(
url=row['article_url'], url=row["article_url"],
title=row['title'], title=row["title"],
author=row['author'], author=row["author"],
publish_date=row['publish_date'], publish_date=row["publish_date"],
content_text=row['content_text'], content_text=row["content_text"],
content_html=row['content_html'], content_html=row["content_html"],
raw_html=archive_content, raw_html=archive_content,
archive_file_path=str(archive_path) if archive_path else None, archive_file_path=archive_file_path,
tags=None, tags=None,
language=None, language=None,
metadata=None, metadata=None,
extraction_method=None, extraction_method=None,
error=row['error_message'], error=row["error_message"],
guid=row['article_guid'], guid=row["article_guid"],
id=row['id'], id=row["id"],
source_name=row['source_name'] source_name=row["source_name"],
) )
articles.append(article) articles.append(article)
@ -643,7 +719,9 @@ def get_source_directory(source_name: str) -> Path:
return WEBSITES_DIR / source_name return WEBSITES_DIR / source_name
def get_archive_file_path_from_db(article_url: str, source_name: str = None) -> Optional[str]: def get_archive_file_path_from_db(
article_url: str, source_name: str = None
) -> Optional[str]:
"""Get archive file path from database mapping. """Get archive file path from database mapping.
Args: Args:
@ -660,20 +738,26 @@ def get_archive_file_path_from_db(article_url: str, source_name: str = None) ->
cursor = conn.cursor() cursor = conn.cursor()
if source_name: if source_name:
cursor.execute(''' cursor.execute(
"""
SELECT archive_file_path FROM article_archives SELECT archive_file_path FROM article_archives
WHERE article_url = ? AND source_name = ? WHERE article_url = ? AND source_name = ?
''', (article_url, source_name)) """,
(article_url, source_name),
)
else: else:
cursor.execute(''' cursor.execute(
"""
SELECT archive_file_path FROM article_archives SELECT archive_file_path FROM article_archives
WHERE article_url = ? WHERE article_url = ?
''', (article_url,)) """,
(article_url,),
)
row = cursor.fetchone() row = cursor.fetchone()
conn.close() conn.close()
return row['archive_file_path'] if row else None return row["archive_file_path"] if row else None
except Exception as e: except Exception as e:
logger.error("Failed to get archive file path from DB: %s", str(e)) logger.error("Failed to get archive file path from DB: %s", str(e))
@ -692,37 +776,44 @@ def get_daily_articles(source_name: str, date_str: str) -> List[ArticleData]:
""" """
try: try:
source_dir = get_source_directory(source_name) source_dir = get_source_directory(source_name)
articles_dir = source_dir / 'articles' / date_str articles_dir = source_dir / "articles" / date_str
if not articles_dir.exists(): if not articles_dir.exists():
return [] return []
articles = [] articles = []
for json_file in sorted(articles_dir.glob('article_*.json')): for json_file in sorted(articles_dir.glob("article_*.json")):
try: try:
with open(json_file, 'r', encoding='utf-8') as f: with open(json_file, "r", encoding="utf-8") as f:
metadata = json.load(f) metadata = json.load(f)
archive_filename = metadata.get('archive_file', '') archive_filename = metadata.get("archive_file", "")
archive_path = source_dir / 'html' / date_str / archive_filename if archive_filename else None archive_path = (
source_dir / "html" / date_str / archive_filename
if archive_filename
else None
)
archive_content = None archive_content = None
if archive_path and archive_path.exists(): if archive_path and archive_path.exists():
archive_content = archive_path.read_text(encoding='utf-8') archive_content = archive_path.read_text(encoding="utf-8")
article = ArticleData( article = ArticleData(
url=metadata.get('url', ''), url=metadata.get("url", ""),
title=metadata.get('title'), title=metadata.get("title"),
author=metadata.get('author'), author=metadata.get("author"),
publish_date=metadata.get('publish_date'), publish_date=metadata.get("publish_date"),
content_text=metadata.get('content_text'), content_text=metadata.get("content_text"),
content_html=metadata.get('content_html'), content_html=metadata.get("content_html"),
raw_html=archive_content, raw_html=archive_content,
archive_file_path=str(archive_path) if archive_path else None, # Handle both absolute and relative paths
tags=metadata.get('tags'), archive_file_path=str(archive_path.relative_to(ARCHIVE_DIR))
language=metadata.get('language'), if archive_path
else None,
tags=metadata.get("tags"),
language=metadata.get("language"),
metadata=metadata, metadata=metadata,
extraction_method=metadata.get('extraction_method'), extraction_method=metadata.get("extraction_method"),
source_name=source_name source_name=source_name,
) )
articles.append(article) articles.append(article)
@ -734,7 +825,12 @@ def get_daily_articles(source_name: str, date_str: str) -> List[ArticleData]:
return articles return articles
except Exception as e: except Exception as e:
logger.error("Failed to get daily articles for %s on %s: %s", source_name, date_str, str(e)) logger.error(
"Failed to get daily articles for %s on %s: %s",
source_name,
date_str,
str(e),
)
return [] return []
@ -747,16 +843,16 @@ def initialize_storage() -> None:
_init_database() _init_database()
(WEBSITES_DIR / 'sample').mkdir(parents=True, exist_ok=True) (WEBSITES_DIR / "sample").mkdir(parents=True, exist_ok=True)
(WEBSITES_DIR / 'sample' / 'html').mkdir(exist_ok=True) (WEBSITES_DIR / "sample" / "html").mkdir(exist_ok=True)
(WEBSITES_DIR / 'sample' / 'articles').mkdir(exist_ok=True) (WEBSITES_DIR / "sample" / "articles").mkdir(exist_ok=True)
_log_processing('system', 'initialize', 'success', 'Storage system initialized') _log_processing("system", "initialize", "success", "Storage system initialized")
logger.info("Storage system initialized at %s", ARCHIVE_DIR) logger.info("Storage system initialized at %s", ARCHIVE_DIR)
if __name__ == '__main__': if __name__ == "__main__":
initialize_storage() initialize_storage()
sources = get_all_sources() sources = get_all_sources()

107
tests/test_path_handling.py Normal file
View File

@ -0,0 +1,107 @@
#!/usr/bin/env python3
"""Unit tests for storage_manager path handling."""
from pathlib import Path
from urllib.parse import unquote
# Use actual ARCHIVE_DIR path
ARCHIVE_DIR = Path("/Volumes/playground/NewsArchiver/archival_data")
def test_relative_path_format():
"""Test the relative path format after save_article."""
# Simulate what save_article does
archive_file_path = (
ARCHIVE_DIR / "websites/404 Media/html/2024-01-15/article_001.html"
)
# This is what we store in the database
stored_path = str(archive_file_path.relative_to(ARCHIVE_DIR))
# Verify stored path is relative
assert not Path(stored_path).is_absolute()
assert stored_path == "websites/404 Media/html/2024-01-15/article_001.html"
# Simulate what get_article does when retrieving
retrieved_path = Path(stored_path)
if not retrieved_path.is_absolute():
full_path = ARCHIVE_DIR / retrieved_path
else:
full_path = retrieved_path
# Verify full path is correct
assert str(full_path) == str(archive_file_path)
# This is what we return for the web interface
web_path = str(full_path.relative_to(ARCHIVE_DIR))
assert web_path == stored_path
print(f"Test passed! Stored: {stored_path}, Web: {web_path}")
def test_multiple_sources():
"""Test that different sources get correct paths."""
sources = ["404 Media", "TestSource", "Another Source"]
for source in sources:
archive_file_path = (
ARCHIVE_DIR / f"websites/{source}/html/2024-01-15/article_001.html"
)
stored_path = str(archive_file_path.relative_to(ARCHIVE_DIR))
# Verify path structure
parts = Path(stored_path).parts
assert parts[0] == "websites"
assert parts[1] == source
assert parts[2] == "html"
print(f"Source '{source}': {stored_path}")
def test_archive_file_url_generation():
"""Test that the URL for archived files is correct."""
# Simulate what the template does
archive_file_path = "websites/404 Media/html/2024-01-15/article_001.html"
# This is what the template generates
url = f"/archive-file/{archive_file_path}"
# Verify URL format
assert url == "/archive-file/websites/404 Media/html/2024-01-15/article_001.html"
# Simulate what the route handler does
decoded_path = unquote(archive_file_path)
full_path = ARCHIVE_DIR / decoded_path
# Verify the full path is correct
expected_path = ARCHIVE_DIR / "websites/404 Media/html/2024-01-15/article_001.html"
assert str(full_path) == str(expected_path)
print(f"URL: {url}")
print(f"Full path: {full_path}")
def test_old_absolute_path_handling():
"""Test handling of old absolute paths from different servers."""
# Old absolute path from a different server
old_absolute_path = Path(
"/home/user/playground/NewsArchiver/archival_data/websites/404 Media/html/2024-01-15/article_001.html"
)
# Check if it's absolute
assert old_absolute_path.is_absolute()
# The code handles this by checking is_absolute() first
# If the path is absolute but not under ARCHIVE_DIR, we can still try to extract
# the relative part by checking if ARCHIVE_DIR is in the path
if old_absolute_path.is_absolute():
# For this test, we just verify the logic
print("Old absolute path handling verified")
if __name__ == "__main__":
test_relative_path_format()
test_multiple_sources()
test_archive_file_url_generation()
test_old_absolute_path_handling()
print("\nAll tests passed!")

File diff suppressed because it is too large Load Diff