From cd67adaae41a5aaff2e4b5ca936b0515d7cc399f Mon Sep 17 00:00:00 2001 From: Jarian Cottingham Date: Sat, 5 Jul 2025 05:35:29 +0000 Subject: [PATCH] Initial Commit --- .gitignore | 18 +++ MCPServer/docker-compose.yml | 20 ++++ MCPServer/dockerfile | 20 ++++ MCPServer/openapi.json | 74 ++++++++++++ MCPServer/pyvenv.cfg | 5 + MCPServer/requirements.txt | 119 +++++++++++++++++++ MCPServer/server.py | 110 ++++++++++++++++++ ai_processor/ai_processor.py | 132 +++++++++++++++++++++ ai_processor/pyvenv.cfg | 5 + embedding/embedder.py | 90 ++++++++++++++ embedding/pyvenv.cfg | 5 + embedding/requirements.txt | 82 +++++++++++++ scraper/pyvenv.cfg | 5 + scraper/requirements.txt | 46 ++++++++ scraper/rss_feeds.json | 205 ++++++++++++++++++++++++++++++++ scraper/rss_short_feed.json | 13 +++ scraper/scraper.py | 219 +++++++++++++++++++++++++++++++++++ 17 files changed, 1168 insertions(+) create mode 100644 .gitignore create mode 100644 MCPServer/docker-compose.yml create mode 100644 MCPServer/dockerfile create mode 100644 MCPServer/openapi.json create mode 100644 MCPServer/pyvenv.cfg create mode 100644 MCPServer/requirements.txt create mode 100644 MCPServer/server.py create mode 100644 ai_processor/ai_processor.py create mode 100644 ai_processor/pyvenv.cfg create mode 100644 embedding/embedder.py create mode 100644 embedding/pyvenv.cfg create mode 100644 embedding/requirements.txt create mode 100644 scraper/pyvenv.cfg create mode 100644 scraper/requirements.txt create mode 100644 scraper/rss_feeds.json create mode 100644 scraper/rss_short_feed.json create mode 100644 scraper/scraper.py diff --git a/.gitignore b/.gitignore new file mode 100644 index 0000000..9d16dee --- /dev/null +++ b/.gitignore @@ -0,0 +1,18 @@ +# Anything from venv +**/bin/* +**/include/* +**/lib/* +**/lib64/* +**/lib64/ +**/lib64 +**/share/* + +# Anything from MCPServer +MCPServer/cache/* +MCPServer/lib64/* + +# Anything from Scraper +scraper/articles/* + +# Anything from AI Process +ai_processor/output/* diff --git a/MCPServer/docker-compose.yml b/MCPServer/docker-compose.yml new file mode 100644 index 0000000..8f1630c --- /dev/null +++ b/MCPServer/docker-compose.yml @@ -0,0 +1,20 @@ +services: + flask-app: + build: . + container_name: stockdocs-mcp + networks: + - ainetwork + ports: + - "5005:5005" + environment: + - FLASK_ENV=development + deploy: + replicas: 1 # You can scale the number of instances (replicas) as needed + resources: + limits: + cpus: '1' + memory: 1024M +networks: + ainetwork: + external: true + name: ainetwork diff --git a/MCPServer/dockerfile b/MCPServer/dockerfile new file mode 100644 index 0000000..d94223d --- /dev/null +++ b/MCPServer/dockerfile @@ -0,0 +1,20 @@ +# Use an official Python runtime as a base image +FROM python:3.12.3-slim + +# Set the working directory in the container +WORKDIR /app + +# Install Dependencies +COPY requirements.txt /app/ + +RUN pip install --no-cache-dir -r requirements.txt + +# Copy the current directory contents into the container at /app +COPY . /app + +# Expose the port the app runs on +EXPOSE 5005 + +# Run the Flask app +CMD ["python", "server.py"] + diff --git a/MCPServer/openapi.json b/MCPServer/openapi.json new file mode 100644 index 0000000..1f3335e --- /dev/null +++ b/MCPServer/openapi.json @@ -0,0 +1,74 @@ +{ + "openapi": "3.0.0", + "info": { + "title": "StockDocs", + "version": "1.0.0" + }, + "paths": { + "/query": { + "post": { + "summary": "Query the vector database", + "requestBody": { + "required": true, + "content": { + "application/json": { + "schema": { + "type": "object", + "properties": { + "question": { "type": "string" } + }, + "required": ["question"] + } + } + } + }, + "responses": { + "200": { + "description": "Query results", + "content": { + "application/json": { + "schema": { + "type": "object", + "properties": { + "results": { + "type": "array", + "items": { + "type": "object", + "properties": { + "document": { "type": "string" }, + "score": { "type": "number" }, + "metadata": { "type": "object" } + } + } + } + } + } + } + } + } + } + } + }, + "/health": { + "get": { + "summary": "Health check", + "responses": { + "200": { + "description": "OK" + } + } + } + }, + "/info": { + "get": { + "summary": "Service info", + "responses": { + "200": { + "description": "Info" + } + } + } + } + } +} + diff --git a/MCPServer/pyvenv.cfg b/MCPServer/pyvenv.cfg new file mode 100644 index 0000000..a71ffc0 --- /dev/null +++ b/MCPServer/pyvenv.cfg @@ -0,0 +1,5 @@ +home = /usr/bin +include-system-site-packages = false +version = 3.12.3 +executable = /usr/bin/python3.12 +command = /usr/bin/python3 -m venv /home/user/StockDocs/MCPServer diff --git a/MCPServer/requirements.txt b/MCPServer/requirements.txt new file mode 100644 index 0000000..8ee3540 --- /dev/null +++ b/MCPServer/requirements.txt @@ -0,0 +1,119 @@ +annotated-types==0.7.0 +anyio==4.9.0 +attrs==25.3.0 +backoff==2.2.1 +bcrypt==4.3.0 +blinker==1.9.0 +build==1.2.2.post1 +cachetools==5.5.2 +certifi==2025.6.15 +charset-normalizer==3.4.2 +chromadb==1.0.13 +click==8.2.1 +coloredlogs==15.0.1 +distro==1.9.0 +durationpy==0.10 +einops==0.8.1 +filelock==3.18.0 +Flask==3.1.1 +flatbuffers==25.2.10 +fsspec==2025.5.1 +google-auth==2.40.3 +googleapis-common-protos==1.70.0 +grpcio==1.73.1 +h11==0.16.0 +hf-xet==1.1.5 +httpcore==1.0.9 +httptools==0.6.4 +httpx==0.28.1 +huggingface-hub==0.33.2 +humanfriendly==10.0 +idna==3.10 +importlib_metadata==8.7.0 +importlib_resources==6.5.2 +itsdangerous==2.2.0 +Jinja2==3.1.6 +joblib==1.5.1 +jsonschema==4.24.0 +jsonschema-specifications==2025.4.1 +kubernetes==33.1.0 +markdown-it-py==3.0.0 +MarkupSafe==3.0.2 +mdurl==0.1.2 +mmh3==5.1.0 +mpmath==1.3.0 +networkx==3.5 +numpy==2.3.1 +nvidia-cublas-cu12==12.6.4.1 +nvidia-cuda-cupti-cu12==12.6.80 +nvidia-cuda-nvrtc-cu12==12.6.77 +nvidia-cuda-runtime-cu12==12.6.77 +nvidia-cudnn-cu12==9.5.1.17 +nvidia-cufft-cu12==11.3.0.4 +nvidia-cufile-cu12==1.11.1.6 +nvidia-curand-cu12==10.3.7.77 +nvidia-cusolver-cu12==11.7.1.2 +nvidia-cusparse-cu12==12.5.4.2 +nvidia-cusparselt-cu12==0.6.3 +nvidia-nccl-cu12==2.26.2 +nvidia-nvjitlink-cu12==12.6.85 +nvidia-nvtx-cu12==12.6.77 +oauthlib==3.3.1 +onnxruntime==1.22.0 +opentelemetry-api==1.34.1 +opentelemetry-exporter-otlp-proto-common==1.34.1 +opentelemetry-exporter-otlp-proto-grpc==1.34.1 +opentelemetry-proto==1.34.1 +opentelemetry-sdk==1.34.1 +opentelemetry-semantic-conventions==0.55b1 +orjson==3.10.18 +overrides==7.7.0 +packaging==25.0 +pillow==11.3.0 +posthog==6.0.1 +protobuf==5.29.5 +pyasn1==0.6.1 +pyasn1_modules==0.4.2 +pybase64==1.4.1 +pydantic==2.11.7 +pydantic_core==2.33.2 +Pygments==2.19.2 +PyPika==0.48.9 +pyproject_hooks==1.2.0 +python-dateutil==2.9.0.post0 +python-dotenv==1.1.1 +PyYAML==6.0.2 +referencing==0.36.2 +regex==2024.11.6 +requests==2.32.4 +requests-oauthlib==2.0.0 +rich==14.0.0 +rpds-py==0.26.0 +rsa==4.9.1 +safetensors==0.5.3 +scikit-learn==1.7.0 +scipy==1.16.0 +sentence-transformers==5.0.0 +setuptools==80.9.0 +shellingham==1.5.4 +six==1.17.0 +sniffio==1.3.1 +sympy==1.14.0 +tenacity==9.1.2 +threadpoolctl==3.6.0 +tokenizers==0.21.2 +torch==2.7.1 +tqdm==4.67.1 +transformers==4.53.0 +triton==3.3.1 +typer==0.16.0 +typing-inspection==0.4.1 +typing_extensions==4.14.0 +urllib3==2.5.0 +uvicorn==0.35.0 +uvloop==0.21.0 +watchfiles==1.1.0 +websocket-client==1.8.0 +websockets==15.0.1 +Werkzeug==3.1.3 +zipp==3.23.0 diff --git a/MCPServer/server.py b/MCPServer/server.py new file mode 100644 index 0000000..663fdf2 --- /dev/null +++ b/MCPServer/server.py @@ -0,0 +1,110 @@ +import chromadb +from sentence_transformers import SentenceTransformer +from flask import Flask, request, jsonify, send_from_directory + +model = SentenceTransformer( + "Snowflake/snowflake-arctic-embed-m-long", + device="cpu", # <-- the only line that changes + trust_remote_code=True +) + +# Test Test Test +client = chromadb.HttpClient(host="chromadb", port=8000) + +collection = client.get_or_create_collection("news") + + +def query(question): + + query_embedding = model.encode(question, normalize_embeddings=True) + + results = collection.query( + query_embeddings=query_embedding, + n_results=5, + ) + + return results + +app = Flask(__name__) + +@app.route("/query", methods=["POST"]) +def query_endpoint(): + data = request.get_json() + question = data.get("question") + if not question: + return jsonify({"error": "Missing 'question' in request body"}), 400 + + results = query(question) + + # Adapt response to MCP Model standards + # Example MCP format: + # { + # "results": [ + # { + # "document": ..., + # "score": ..., + # "metadata": {...} + # }, + # ... + # ] + # } + mcp_results = [] + for doc, score, meta in zip( + results.get("documents", [[]])[0], + results.get("distances", [[]])[0], + results.get("metadatas", [[]])[0]): + mcp_results.append({ + "document": doc, + "score": float(score), + "metadata": meta + }) + + return jsonify({"results": mcp_results}) + +@app.route("/health", methods=["GET"]) +def health(): + return jsonify({"status": "ok"}) + +@app.route("/info", methods=["GET"]) +def info(): + return jsonify({ + "provider": "StockDoc", + "model": "Snowflake/snowflake-arctic-embed-m-long", + "embedding_dim": 1024, # or your actual dimension + "collection": "news" + }) + +@app.route("/tools", methods=["GET"]) +def tools(): + return jsonify({ + "endpoints": [ + { + "path": "/query", + "method": "POST", + "description": "Query the vector database with a question.", + "request_format": {"question": "string"}, + "response_format": { + "results": [ + {"document": "string", "score": "float", "metadata": "object"} + ] + } + }, + { + "path": "/health", + "method": "GET", + "description": "Health check endpoint." + }, + { + "path": "/info", + "method": "GET", + "description": "Returns model and service metadata." + } + ] + }) + +@app.route("/openapi.json", methods=["GET"]) +def openapi(): + return send_from_directory('.', 'openapi.json') + +if __name__ == "__main__": + app.run(host="0.0.0.0", port=5005, threaded=True, debug=True) diff --git a/ai_processor/ai_processor.py b/ai_processor/ai_processor.py new file mode 100644 index 0000000..8e0e173 --- /dev/null +++ b/ai_processor/ai_processor.py @@ -0,0 +1,132 @@ +import os +import requests +import json + +LOCAL_AI_SERVICE_URL = "http://192.168.8.124:11434" + +class ArticleFile: + def __init__(self, filename, content, source="Unfiltered"): + self.filename = filename + self.content = content + self.source = source + + def __str__(self): + return f"ArticleFile(filename={self.filename}, source={self.source}, content_length={len(self.content)})" + +def load_articles_from_folder(folder_path): + """ + Loads all articles from text files in the specified folder and returns them as a list of ArticleFile objects. + """ + articles = [] + for newspaper in os.listdir(folder_path): + for filename in os.listdir(os.path.join(folder_path, newspaper)): + file_path = os.path.join(folder_path, newspaper, filename) + if os.path.isfile(file_path): + with open(file_path, 'r', encoding='utf-8') as f: + first_line = f.readline() + if first_line.startswith("SOURCE:"): + source = first_line[len("SOURCE:"):].strip() + content = f.read() + else: + source = "Unfiltered" + content = first_line + f.read() + articles.append(ArticleFile(filename, content, source)) + + print(f"Loaded {len(articles)} articles from folder {folder_path}.") + return articles + +def call_local_ai_service(article): + """ + Sends the article to the local Ollama AI server and returns the parsed response. + Trims any 'thinking' section from the DeepSeek model response. + """ + url = LOCAL_AI_SERVICE_URL + "/api/generate" + prompt = ( + "Give a detailed summary of the article and then list out the tickers and their movement in the article. Do not say anything at the end of your response." + "Format the response as JSON with keys 'summary' and 'tickersAndMovements', where 'tickersAndMovements' is a list of " + "['Ticker', 'Movement direction', 'Percent change']. Here's an example of the expected output:\n" + "{'summary': '...', 'tickersAndMovements': [['AAPL', '+', '2.5%'], ['GOOGL', '-', '1.2%']]}\n\n" \ + "Validate the JSON format and ensure it is parsable before giving the answer. For every entry in " + "tickersAndMovements, it must have 3 entries. If you don't know the ticker symbol, don't make one up, just give " + "the full name, you must have either +, - or = for the second entry, and you must have a decimal with percent (like 2.4%) or (?%) if you're unsure" \ + "Do not say anything at the end of your response." + ) + payload = { + "model": "llama3.2:latest", + "prompt": f"{prompt}\n\nArticle:\n{article}", + "stream": False + # TODO: In the future, we should force the response to be json and adjust the temperature + } + + print(f"Sending request to local AI service ...") + + # Send the request to the local AI service + response = requests.post(url, json=payload, timeout=180) # Increased timeout to 300 seconds + response.raise_for_status() + # The AI should return a JSON string, so parse it + raw_response = response.json()["response"] + # Remove the 'thinking' section if present + think_marker = "" + if think_marker in raw_response: + raw_response = raw_response.split(think_marker, 1)[1] # Take the second half after the marker + + # Remove any leading code block markers or whitespace (e.g., ```json or ``` or whitespace) + raw_response = raw_response.lstrip("` \njson") + # Remove any trailing code block marker ``` + if raw_response.endswith("```"): + raw_response = raw_response[:-3].rstrip() + + #print(f"Received response (trimmed): {raw_response}") + print(f"Received response from local AI service, processing...") + ai_processed_results = json.loads(raw_response) + return ai_processed_results + +def process_article(article_file): + processed_article = None + + # Check if we already processed this article + # TODO : protentially expensive, consider using a database or cache + if os.path.exists(f"output/{article_file.filename}.json"): + print(f"Article {article_file.filename} already processed, skipping.") + + # open file and return the content + with open(f"output/{article_file.filename}.json", 'r', encoding='utf-8') as f: + processed_article = json.load(f) + return processed_article + + for attempt in range(3): + try: + # Pass the article content to the AI service + processed_article = call_local_ai_service(article_file.content) + print(f"Processed article: {processed_article['summary'][:100]}...") + + # Add metadata to the processed article + processed_article['filename'] = article_file.filename + processed_article['original_content'] = article_file.content + processed_article['source'] = article_file.source # <-- Add source here + break # Success, exit retry loop + except Exception as e: + print(f"\tError processing article (attempt {attempt+1}/3): {e}") + if processed_article is not None: + return processed_article + else: + print(f"\tSkipping article after 3 failed attempts.") + + return processed_article + +# Retrieve the current archive of pulled articles +articles_folder = os.path.join(os.path.dirname(__file__), "../scraper/articles") +print("Loading articles from folder " + articles_folder + " ...") +articles = load_articles_from_folder(articles_folder) + +# Process the articles retrieved +results = [] +for article in articles: + try: + result = process_article(article) + + # Save the processed result to a JSON file + with open("output/" + result['filename'] + ".json", "w", encoding="utf-8") as f: + json.dump(result, f, ensure_ascii=False, indent=2) + except Exception as e: + print(f"Error processing or saving article {article.filename}: {e}") diff --git a/ai_processor/pyvenv.cfg b/ai_processor/pyvenv.cfg new file mode 100644 index 0000000..1536dac --- /dev/null +++ b/ai_processor/pyvenv.cfg @@ -0,0 +1,5 @@ +home = /usr/bin +include-system-site-packages = false +version = 3.12.3 +executable = /usr/bin/python3.12 +command = /usr/bin/python3 -m venv /home/user/embedding diff --git a/embedding/embedder.py b/embedding/embedder.py new file mode 100644 index 0000000..63480c2 --- /dev/null +++ b/embedding/embedder.py @@ -0,0 +1,90 @@ +import os +import json +import chromadb +import uuid + +from sentence_transformers import SentenceTransformer + +model = SentenceTransformer( + "Snowflake/snowflake-arctic-embed-m-long", + device="cpu", # <-- the only line that changes + trust_remote_code=True +) + + +import math +def embed_text(text, specific_context, max_tokens=2048, overlap=256): + """ + Embeds the given text using the SentenceTransformer model. + Splits the text into chunks if it exceeds max_tokens. + """ + if len(text) + len(specific_context) <= max_tokens: + return (text + specific_context, model.encode([text, specific_context], normalize_embeddings=True)) + + # Split text into chunks + chunks = [] + start = 0 + + chunkNum = math.ceil(len(text) / (max_tokens - len(specific_context))) + # 1000 + 20 = 1020 + # 3000 + 20 = 3020 -> 3000 / (2048 - 20) = 1.47 ~ 2 + + for i in range(chunkNum): + if start >= len(text): + break + # Calculate end index for the chunk + end = min(start + max_tokens - len(specific_context), len(text)) + chunk = text[start:end] + specific_context + chunks.append(chunk) + start += (max_tokens - overlap) # Overlap for next chunk + + return (chunks, model.encode(chunks, normalize_embeddings=True)) + + +client = chromadb.HttpClient(host="localhost", port=8000) + +# client.delete_collection("news") # Replace "news" with your collection name + +collection = client.get_or_create_collection("news") + +# Load all the processed articles from the output folder +output_folder = os.path.join(os.path.dirname(__file__), "../ai_processor/output") +output_articles = [] +# When embedding and upserting, use the source from the processed result +for filename in os.listdir(output_folder): + if filename.endswith(".json"): + with open(os.path.join(output_folder, filename), 'r', encoding='utf-8') as f: + prev_proc = json.load(f) + output_articles.append(prev_proc) + + def makeTickerMovement(ticker, movement, percent_change): + """ + Helper function to create a ticker movement entry. + """ + ticker = "" if ticker is None else ticker + movement = "" if movement is None else movement + percent_change = "" if percent_change is None else percent_change + return [ticker, movement, percent_change] + + # Embed the article content + try: + specific_context = prev_proc['summary'] + ",".join([' '.join(makeTickerMovement(*x)) for x in prev_proc['tickersAndMovements']]) + chunks, embedded_content = embed_text(prev_proc['original_content'], prev_proc['summary']) + + print(f"Embedded content for {filename}: \n {embedded_content[:10]}...") # Print first 10 values for preview + + collection.upsert( + ids = [ str(uuid.uuid4()) for d in chunks ], # Generate unique IDs for each embedded content + documents = chunks, # optional but nice for debugging + embeddings = embedded_content.tolist(), # Chroma expects List[List[float]] + metadatas =[{ + "source": prev_proc.get('source', 'Unknown'), # Use the source from the processed result + "published": prev_proc.get('published', 'Unknown'), + "filename": prev_proc.get('filename', 'Unknown') # Add filename for traceability + } for d in chunks], # Metadata for each embedded content + ) + + except Exception as e: + print(f"Error embedding content for {filename}: {e}") + continue + diff --git a/embedding/pyvenv.cfg b/embedding/pyvenv.cfg new file mode 100644 index 0000000..1536dac --- /dev/null +++ b/embedding/pyvenv.cfg @@ -0,0 +1,5 @@ +home = /usr/bin +include-system-site-packages = false +version = 3.12.3 +executable = /usr/bin/python3.12 +command = /usr/bin/python3 -m venv /home/user/embedding diff --git a/embedding/requirements.txt b/embedding/requirements.txt new file mode 100644 index 0000000..15c98f7 --- /dev/null +++ b/embedding/requirements.txt @@ -0,0 +1,82 @@ +attrs==23.2.0 +Automat==22.10.0 +Babel==2.10.3 +bcc==0.29.1 +bcrypt==3.2.2 +blinker==1.7.0 +boto3==1.34.46 +botocore==1.34.46 +certifi==2023.11.17 +chardet==5.2.0 +click==8.1.6 +cloud-init==25.1.2 +colorama==0.4.6 +command-not-found==0.3 +configobj==5.0.8 +constantly==23.10.4 +cryptography==41.0.7 +dbus-python==1.3.2 +distro==1.9.0 +distro-info==1.7+build1 +gpg==1.18.0 +h11==0.14.0 +httplib2==0.20.4 +hyperlink==21.0.0 +idna==3.6 +incremental==22.10.0 +Jinja2==3.1.2 +jmespath==1.0.1 +jsonpatch==1.32 +jsonpointer==2.0 +jsonschema==4.10.3 +launchpadlib==1.11.0 +lazr.restfulclient==0.14.6 +lazr.uri==1.0.6 +Markdown==3.5.2 +markdown-it-py==3.0.0 +MarkupSafe==2.1.5 +mdurl==0.1.2 +netaddr==0.8.0 +netifaces==0.11.0 +oauthlib==3.2.2 +packaging==24.0 +pexpect==4.9.0 +ptyprocess==0.7.0 +pyasn1==0.4.8 +pyasn1-modules==0.2.8 +Pygments==2.17.2 +PyGObject==3.48.2 +PyHamcrest==2.1.0 +PyJWT==2.7.0 +pyOpenSSL==23.2.0 +pyparsing==3.1.1 +pyrsistent==0.20.0 +pyserial==3.5 +python-apt==2.7.7+ubuntu4 +python-dateutil==2.8.2 +python-debian==0.1.49+ubuntu2 +python-magic==0.4.27 +pytz==2024.1 +PyYAML==6.0.1 +requests==2.31.0 +rich==13.7.1 +s3transfer==0.10.1 +service-identity==24.1.0 +setuptools==68.1.2 +six==1.16.0 +sos==4.7.2 +ssh-import-id==5.11 +systemd-python==235 +Twisted==24.3.0 +ubuntu-drivers-common==0.0.0 +ubuntu-pro-client==8001 +ufw==0.36.2 +unattended-upgrades==0.1 +urllib3==2.0.7 +uvicorn==0.27.1 +uvloop==0.19.0 +wadllib==1.3.6 +wheel==0.42.0 +wsproto==1.2.0 +xkit==0.0.0 +zope.interface==6.1 diff --git a/scraper/pyvenv.cfg b/scraper/pyvenv.cfg new file mode 100644 index 0000000..e160f35 --- /dev/null +++ b/scraper/pyvenv.cfg @@ -0,0 +1,5 @@ +home = /usr/bin +include-system-site-packages = false +version = 3.12.3 +executable = /usr/bin/python3.12 +command = /usr/bin/python3 -m venv /home/user/scraper diff --git a/scraper/requirements.txt b/scraper/requirements.txt new file mode 100644 index 0000000..21c339e --- /dev/null +++ b/scraper/requirements.txt @@ -0,0 +1,46 @@ +attrs==25.3.0 +beautifulsoup4==4.13.4 +certifi==2025.6.15 +charset-normalizer==3.4.2 +click==8.2.1 +dnspython==2.7.0 +feedparser==6.0.11 +filelock==3.18.0 +gnews==0.4.1 +greenlet==3.2.3 +h11==0.16.0 +idna==3.10 +joblib==1.5.1 +lxml==5.4.0 +lxml-html-clean==0.4.2 +newspaper4k==0.9.3.1 +nltk==3.9.1 +numpy==2.3.0 +outcome==1.3.0.post0 +pandas==2.3.0 +pillow==11.2.1 +playwright==1.52.0 +pyee==13.0.0 +pysocks==1.7.1 +python-dateutil==2.9.0.post0 +pytz==2025.2 +pyyaml==6.0.2 +regex==2024.11.6 +requests==2.32.4 +requests-file==2.1.0 +selenium==4.33.0 +sgmllib3k==1.0.0 +six==1.17.0 +sniffio==1.3.1 +sortedcontainers==2.4.0 +soupsieve==2.7 +tldextract==5.3.0 +tqdm==4.67.1 +trio==0.30.0 +trio-websocket==0.12.2 +typing-extensions==4.14.0 +tzdata==2025.2 +urllib3==2.5.0 +websocket-client==1.8.0 +wsproto==1.2.0 + diff --git a/scraper/rss_feeds.json b/scraper/rss_feeds.json new file mode 100644 index 0000000..6026f7b --- /dev/null +++ b/scraper/rss_feeds.json @@ -0,0 +1,205 @@ +{ + "rss_feeds": { + "Reuters – Business News": { + "source_website": "reuters.com", + "rss_url": "https://news.google.com/rss/search?q=site:reuters.com+business&hl=en-US&gl=US&ceid=US:en" + }, + "Associated Press – Business": { + "source_website": "apnews.com", + "rss_url": "https://news.google.com/rss/search?q=site:apnews.com&hl=en-US&gl=US&ceid=US:en" + }, + "Financial Times": { + "source_website": "ft.com", + "rss_url": "https://www.ft.com/rss/home" + }, + "Fortune – Top Stories": { + "source_website": "fortune.com", + "rss_url": "https://fortune.com/feed/fortune-feeds/?id=3230629" + }, + "Seeking Alpha – Market News": { + "source_website": "seekingalpha.com", + "rss_url": "https://seekingalpha.com/feed.xml" + }, + "The Motley Fool – Stock News & Analysis": { + "source_website": "fool.com", + "rss_url": "https://www.fool.com/a/feeds/partner/googlechromefollow?apikey=5e092c1f-c5f9-4428-9219-908a47d2e2de" + }, + "Nasdaq – Markets News": { + "source_website": "nasdaq.com", + "rss_url": "https://www.nasdaq.com/feed/rssoutbound?category=Nasdaq" + }, + "Business Standard – Latest News": { + "source_website": "business-standard.com", + "rss_url": "https://www.business-standard.com/rss/latest.rss" + }, + "TheStreet – Full Articles": { + "source_website": "thestreet.com", + "rss_url": "https://www.thestreet.com/.rss/full" + }, + "Benzinga – Financial News": { + "source_website": "benzinga.com", + "rss_url": "https://feeds.benzinga.com/benzinga" + }, + "MarketBeat – Market News": { + "source_website": "marketbeat.com", + "rss_url": "https://www.marketbeat.com/feed/" + }, + "Money (Time) – Personal Finance": { + "source_website": "money.com", + "rss_url": "https://money.com/money/feed/" + }, + "Global Finance Magazine": { + "source_website": "gfmag.com", + "rss_url": "https://www.gfmag.com/feed" + }, + "Financial Samurai": { + "source_website": "financialsamurai.com", + "rss_url": "https://www.financialsamurai.com/feed/" + }, + "MoneyWeek": { + "source_website": "moneyweek.com", + "rss_url": "https://moneyweek.com/feed/all" + }, + "Finance Monthly": { + "source_website": "finance-monthly.com", + "rss_url": "https://www.finance-monthly.com/feed/" + }, + "European Financial Review": { + "source_website": "europeanfinancialreview.com", + "rss_url": "https://www.europeanfinancialreview.com/feed" + }, + "Money Morning": { + "source_website": "moneymorning.com", + "rss_url": "https://moneymorning.com/feed" + }, + "Dealbreaker": { + "source_website": "dealbreaker.com", + "rss_url": "https://dealbreaker.com/.rss/full" + }, + "World Finance": { + "source_website": "worldfinance.com", + "rss_url": "https://www.worldfinance.com/feed" + }, + "Fox Business – Headlines": { + "source_website": "foxbusiness.com", + "rss_url": "https://moxie.foxbusiness.com/google-publisher/latest.xml" + }, + "FinanceAsia": { + "source_website": "financeasia.com", + "rss_url": "https://www.financeasia.com/rss/latest" + }, + "CNBC – Business": { + "source_website": "cnbc.com", + "rss_url": "https://www.cnbc.com/id/100003114/device/rss/rss.html" + }, + "CNN Money": { + "source_website": "cnn.com", + "rss_url": "http://rss.cnn.com/rss/money_topstories.rss" + }, + "Markets Insider": { + "source_website": "markets.businessinsider.com", + "rss_url": "https://markets.businessinsider.com/rss/news" + }, + "U.S. News – Money": { + "source_website": "usnews.com", + "rss_url": "https://www.usnews.com/rss/money" + }, + "The Economist – Business & Finance": { + "source_website": "economist.com", + "rss_url": "https://www.economist.com/business/rss.xml" + }, + "Barchart News": { + "source_website": "barchart.com", + "rss_url": "http://feeds.feedburner.com/BarchartNews" + }, + "The Guardian – Business": { + "source_website": "theguardian.com", + "rss_url": "http://feeds.theguardian.com/theguardian/uk/business/rss" + }, + "Economy Watch": { + "source_website": "economywatch.com", + "rss_url": "https://www.economywatch.com/feed" + }, + "CFI.co": { + "source_website": "cfi.co", + "rss_url": "https://cfi.co/feed" + }, + "BBC News – Business": { + "source_website": "bbc.co.uk", + "rss_url": "http://feeds.bbci.co.uk/news/business/rss.xml" + }, + "Investor’s Business Daily": { + "source_website": "investors.com", + "rss_url": "https://www.investors.com/feed/" + }, + "Forbes – Real-Time": { + "source_website": "forbes.com", + "rss_url": "https://www.forbes.com/real-time/feed2/" + }, + "The Financial Express": { + "source_website": "financialexpress.com", + "rss_url": "https://www.financialexpress.com/feed/" + }, + "MarketWatch – Top Stories": { + "source_website": "marketwatch.com", + "rss_url": "http://feeds.marketwatch.com/marketwatch/topstories/" + }, + "Wall Street Journal – U.S. Business": { + "source_website": "wsj.com", + "rss_url": "https://feeds.a.dj.com/rss/WSJcomUSBusiness.xml" + }, + "Sky News – Business": { + "source_website": "news.sky.com", + "rss_url": "http://news.sky.com/feeds/rss/business.xml" + }, + "Bloomberg – Surveillance Podcast": { + "source_website": "bloomberg.com", + "rss_url": "https://www.bloomberg.com/feed/podcast/bloomberg-surveillance.xml" + }, + "Barron’s – Markets": { + "source_website": "barrons.com", + "rss_url": "https://www.barrons.com/xml/rss/markets.xml" + }, + "Yahoo Finance": { + "source_website": "finance.yahoo.com", + "rss_url": "https://www.yahoo.com/news/rss/finance" + }, + "Investing.com – News": { + "source_website": "investing.com", + "rss_url": "https://www.investing.com/rss/news.rss" + }, + "Investopedia – Headlines": { + "source_website": "investopedia.com", + "rss_url": "https://www.investopedia.com/feedbuilder/feed/getfeed/?feedName=rss_headline" + }, + "NerdWallet – Finance": { + "source_website": "nerdwallet.com", + "rss_url": "https://www.nerdwallet.com/news/finance/feed" + }, + "Newsmax Finance": { + "source_website": "newsmax.com", + "rss_url": "https://www.newsmax.com/rss/finance" + }, + "Bankrate – News": { + "source_website": "bankrate.com", + "rss_url": "https://www.bankrate.com/rss/" + }, + "Morningstar – Articles": { + "source_website": "morningstar.com", + "rss_url": "https://www.morningstar.com/articles.rss" + }, + "Kiplinger": { + "source_website": "kiplinger.com", + "rss_url": "https://www.kiplinger.com/kiplinger.rss" + }, + "International Business Times": { + "source_website": "ibtimes.com", + "rss_url": "https://www.ibtimes.com/rss" + }, + "Policygenius – News": { + "source_website": "policygenius.com", + "rss_url": "https://www.policygenius.com/news/feed/" + } + } +} + diff --git a/scraper/rss_short_feed.json b/scraper/rss_short_feed.json new file mode 100644 index 0000000..5b56bcb --- /dev/null +++ b/scraper/rss_short_feed.json @@ -0,0 +1,13 @@ +{ + "rss_feeds": { + "Reuters – Business News": { + "source_website": "reuters.com", + "rss_url": "https://news.google.com/rss/search?q=site:reuters.com+business&hl=en-US&gl=US&ceid=US:en" + }, + "Associated Press – Business": { + "source_website": "apnews.com", + "rss_url": "https://news.google.com/rss/search?q=site:apnews.com&hl=en-US&gl=US&ceid=US:en" + } +} +} + diff --git a/scraper/scraper.py b/scraper/scraper.py new file mode 100644 index 0000000..1b4b222 --- /dev/null +++ b/scraper/scraper.py @@ -0,0 +1,219 @@ +import newspaper +import json +import feedparser +import time +import os +from selenium import webdriver +from selenium.webdriver.firefox.options import Options as FirefoxOptions +from newspaper.google_news import GoogleNewsSource +from concurrent.futures import ProcessPoolExecutor +import nltk + +# Ensure necessary NLTK resources are downloaded / Needed for selenium + newspaper4k article parsing +nltk.download('punkt_tab') + +articles = [] + +def load_rss_feed_sources(): + """ + Loads the RSS feed sources from a JSON file. + """ + print("Loading RSS feed sources from rss_feeds.json...") + + try: + with open("rss_short_feed.json", "r", encoding="utf-8") as f: + return json.load(f) + except FileNotFoundError: + print("rss_feeds.json not found, returning empty list.") + return [] + except json.JSONDecodeError: + print("Error decoding rss_feeds.json, returning empty list.") + return [] + +def mine_all_articles(rss_feed_sources, limit=None): + """ + Mines all articles from the given RSS feed sources. + Returns a list of (site, title, link) tuples. + """ + all_links = [] + sources = rss_feed_sources['rss_feeds'] + + for site, data in sources.items(): + print(f"Parsing RSS feed: {data['rss_url']}") + try: + feed = feedparser.parse(data["rss_url"]) + feed_entries = feed.entries[:limit] if limit else feed.entries + + for entry in feed_entries: + if "link" in entry and "title" in entry: + all_links.append((site, entry.title, entry.link)) + except Exception as e: + print(f"Error parsing RSS feed: {site} Error: {str(e)}") + return all_links + +def generate_filename_from_url(url): + """ + Generates a filename from the given URL by replacing slashes with underscores. + """ + # Use only the last part of the URL or replace slashes + return url.replace("https://", "").replace("http://", "").replace("/", "_") + +def generate_safe_filename(name): + # Remove/replace characters not allowed in filenames + import re + safe = re.sub(r'[\\/*?:"<>|]', "_", name) + return safe + +def save_article_to_file(article, filename, source="Unfiltered"): + """ + Saves the given article text to a file with the specified filename. + """ + os.makedirs("articles", exist_ok=True) + + outputDir = "articles/"+source if source else "articles" + os.makedirs(outputDir, exist_ok=True) if source else None + + # Sanitize filename: use only the last part of the URL or replace slashes + safe_filename = generate_filename_from_url(filename) + file_path = os.path.join(outputDir, safe_filename) + # Save the source as the first line in the file for later retrieval + with open(file_path, "w", encoding="utf-8") as f: + f.write(f"SOURCE:{source}\n") + f.write(article) + print(f"Article saved to {file_path}") + +def get_article_with_selenium(url): + options = FirefoxOptions() + options.add_argument("--headless") + driver = webdriver.Firefox(options=options) + try: + driver.get(url) + time.sleep(5) # Wait for JS to load + html = driver.page_source + + # Parse with Newspaper4k + article = newspaper.article(url, input_html=html, language='en') + article.nlp() + return article.text + finally: + driver.quit() + +def get_article_with_playwright(url): + import asyncio + from playwright.sync_api import sync_playwright + from newspaper import Article + + with sync_playwright() as p: + browser = p.chromium.launch(headless=True) + page = browser.new_page() + page.goto(url) + + # Optional: wait for specific content to load + time.sleep(5) # Adjust as needed for the page to load completely + + html = page.content() + ''' + # Check for iframes and use the first one's content if present + frames = page.frames + main_frame = page.main_frame + for frame in frames: + if frame != main_frame: + try: + frame.wait_for_load_state("domcontentloaded", timeout=5000) + html = frame.content() + print("Extracted content from iframe.") + break + except Exception as e: + print(f"Could not extract iframe content: {e}") + ''' + + browser.close() + + # Parse with Newspaper4k + article = newspaper.article(url, input_html=html, language='en') + article.nlp() + return article.text + + +def pull_articles(link, source, title=None, save_to_file=True): + filename = title if title else link + safe_filename = generate_filename_from_url(filename) + if os.path.exists(os.path.join("articles", safe_filename)): + print(f"Article already cached: {filename}") + with open(os.path.join("articles", safe_filename), "r", encoding="utf-8") as f: + return f.read() + + text = "" + + time.sleep(5) # Since we spawn lots of processes, we need to sleep at the start + + try: + # --- Google News Source handling temporarily disabled --- + # if "news.google.com" in link: + # gn = GoogleNewsSource(link) + # gn.build() + # # Pull the first article from the Google News cluster + # if gn.articles: + # article = gn.articles[0] + # article.download() + # article.parse() + # text = article.text + # if not text or len(text) < 200: + # raise ValueError("\tGoogle News article text too short, falling back to Selenium.") + # print(f"\tSuccessfully pulled Google News article from {link}") + # else: + article = newspaper.article(link) + article.download() + article.parse() + text = article.text + if not text or len(text) < 200: + raise ValueError("\tArticle text too short, falling back to Playwright/Selenium.") + print(f"\tSuccessfully pulled article with newspaper4k from {link}") + + except Exception as e: + print(f"\tnewspaper4k extraction failed for {link}: {e}, falling back to Playwright.") + try: + text = get_article_with_playwright(link) + print(f"\t\tSuccessfully pulled article from {link} with Playwright") + if not text or len(text) < 200: + print(f"\t\tPlaywright article too short, falling back to Selenium.") + try: + text = get_article_with_selenium(link) + print(f"\t\t\tSuccessfully pulled article from {link} with Selenium") + except Exception as e: + print(f"\t\t\tSelenium failed for {link}: {e}") + return "" + except Exception as e: + print(f"\t\tPlaywright failed for {link}: {e}") + try: + text = get_article_with_selenium(link) + print(f"\t\tSuccessfully pulled article from {link} with Selenium") + except Exception as e: + print(f"\t\tSelenium failed for {link}: {e}") + return "" + + if save_to_file: + save_article_to_file(text, filename, source) + return text + +# Pull the RSS feed sources from the JSON file +rss_feed_sources = load_rss_feed_sources() + +# Mine all articles from the RSS feed sources +rss_feed_links = mine_all_articles(rss_feed_sources) + +# Randomize the order of the links to help with load balancing +import random +random.shuffle(rss_feed_links) + +# Pull the latest article from a variety of news sources +#[pull_articles(link, source, title) for source, title, link in rss_feed_links] + +link_list = [link for _, title, link in rss_feed_links] +source_list = [source for source, _, _ in rss_feed_links] +title_list = [title for _, title, link in rss_feed_links] + +with ProcessPoolExecutor() as executor: + results = list(executor.map(pull_articles, link_list, source_list, title_list)) + print(f"Pulled {len(results)} articles in parallel.") + print("All articles pulled successfully.")