init: initial commit
This commit is contained in:
38
.github/workflows/update_feed.yml
vendored
Normal file
38
.github/workflows/update_feed.yml
vendored
Normal file
@@ -0,0 +1,38 @@
|
||||
name: Update Hugging Face Papers RSS Feed
|
||||
|
||||
on:
|
||||
schedule:
|
||||
# Runs daily at midnight UTC
|
||||
- cron: '0 0 * * *'
|
||||
workflow_dispatch: # Allows manual triggering
|
||||
|
||||
jobs:
|
||||
update-feed:
|
||||
runs-on: ubuntu-latest
|
||||
permissions:
|
||||
contents: write # Allow the job to push changes
|
||||
steps:
|
||||
- name: Checkout repository
|
||||
uses: actions/checkout@v4
|
||||
|
||||
- name: Set up Python
|
||||
uses: actions/setup-python@v5
|
||||
with:
|
||||
python-version: '3.11' # Specify Python version
|
||||
|
||||
- name: Install dependencies
|
||||
run: |
|
||||
python -m pip install --upgrade pip
|
||||
pip install -r requirements.txt
|
||||
|
||||
- name: Run parser and generate feed
|
||||
run: python parser.py --source https://huggingface.co/papers --output feed.xml
|
||||
|
||||
- name: Commit and push if feed changed
|
||||
run: |
|
||||
git config --global user.name 'github-actions[bot]'
|
||||
git config --global user.email 'github-actions[bot]@users.noreply.github.com'
|
||||
git add feed.xml
|
||||
# Commit only if there are changes staged
|
||||
git diff --staged --quiet || git commit -m "Update RSS feed"
|
||||
git push
|
||||
46
.github/workflows/update_feed_monthly.yml
vendored
Normal file
46
.github/workflows/update_feed_monthly.yml
vendored
Normal file
@@ -0,0 +1,46 @@
|
||||
name: Update Hugging Face Papers Monthly RSS Feed
|
||||
|
||||
on:
|
||||
schedule:
|
||||
# Runs on the 1st of every month at midnight UTC
|
||||
- cron: '0 0 1 * *'
|
||||
workflow_dispatch: # Allows manual triggering
|
||||
|
||||
jobs:
|
||||
update-feed-monthly:
|
||||
runs-on: ubuntu-latest
|
||||
permissions:
|
||||
contents: write # Allow the job to push changes
|
||||
steps:
|
||||
- name: Checkout repository
|
||||
uses: actions/checkout@v4
|
||||
|
||||
- name: Set up Python
|
||||
uses: actions/setup-python@v5
|
||||
with:
|
||||
python-version: '3.11' # Specify Python version
|
||||
|
||||
- name: Install dependencies
|
||||
run: |
|
||||
python -m pip install --upgrade pip
|
||||
pip install -r requirements.txt
|
||||
|
||||
- name: Generate monthly URL
|
||||
id: generate_url
|
||||
run: |
|
||||
MONTHLY_TAG=$(date +%Y-%m) # Format: YYYY-MM
|
||||
echo "Generated monthly tag: $MONTHLY_TAG"
|
||||
URL="https://huggingface.co/papers/month/${MONTHLY_TAG}"
|
||||
echo "URL=$URL" >> $GITHUB_OUTPUT
|
||||
|
||||
- name: Run parser and generate monthly feed
|
||||
run: python parser.py --source ${{ steps.generate_url.outputs.URL }} --output feed_monthly.xml
|
||||
|
||||
- name: Commit and push if monthly feed changed
|
||||
run: |
|
||||
git config --global user.name 'github-actions[bot]'
|
||||
git config --global user.email 'github-actions[bot]@users.noreply.github.com'
|
||||
git add feed_monthly.xml
|
||||
# Commit only if there are changes staged
|
||||
git diff --staged --quiet || git commit -m "Update Monthly RSS feed"
|
||||
git push
|
||||
46
.github/workflows/update_feed_weekly.yml
vendored
Normal file
46
.github/workflows/update_feed_weekly.yml
vendored
Normal file
@@ -0,0 +1,46 @@
|
||||
name: Update Hugging Face Papers Weekly RSS Feed
|
||||
|
||||
on:
|
||||
schedule:
|
||||
# Runs every Monday at midnight UTC
|
||||
- cron: '0 0 * * 1'
|
||||
workflow_dispatch: # Allows manual triggering
|
||||
|
||||
jobs:
|
||||
update-feed-weekly:
|
||||
runs-on: ubuntu-latest
|
||||
permissions:
|
||||
contents: write # Allow the job to push changes
|
||||
steps:
|
||||
- name: Checkout repository
|
||||
uses: actions/checkout@v4
|
||||
|
||||
- name: Set up Python
|
||||
uses: actions/setup-python@v5
|
||||
with:
|
||||
python-version: '3.11' # Specify Python version
|
||||
|
||||
- name: Install dependencies
|
||||
run: |
|
||||
python -m pip install --upgrade pip
|
||||
pip install -r requirements.txt
|
||||
|
||||
- name: Generate weekly URL
|
||||
id: generate_url # Give this step an ID to reference its output
|
||||
run: |
|
||||
WEEKLY_TAG=$(date +%Y-W%V) # Format: YYYY-Www (ISO 8601 week)
|
||||
echo "Generated weekly tag: $WEEKLY_TAG"
|
||||
URL="https://huggingface.co/papers/week/${WEEKLY_TAG}"
|
||||
echo "URL=$URL" >> $GITHUB_OUTPUT # Set output for use in next step
|
||||
|
||||
- name: Run parser and generate weekly feed
|
||||
run: python parser.py --source ${{ steps.generate_url.outputs.URL }} --output feed_weekly.xml
|
||||
|
||||
- name: Commit and push if weekly feed changed
|
||||
run: |
|
||||
git config --global user.name 'github-actions[bot]'
|
||||
git config --global user.email 'github-actions[bot]@users.noreply.github.com'
|
||||
git add feed_weekly.xml
|
||||
# Commit only if there are changes staged
|
||||
git diff --staged --quiet || git commit -m "Update Weekly RSS feed"
|
||||
git push
|
||||
37
.gitignore
vendored
Normal file
37
.gitignore
vendored
Normal file
@@ -0,0 +1,37 @@
|
||||
# Byte-compiled / optimized files
|
||||
__pycache__/
|
||||
*.py[cod]
|
||||
*$py.class
|
||||
|
||||
# Distribution / packaging
|
||||
*.egg-info/
|
||||
dist/
|
||||
build/
|
||||
|
||||
# Virtual environments
|
||||
venv/
|
||||
env/
|
||||
.env/
|
||||
.venv/
|
||||
|
||||
# IDE specific files
|
||||
.idea/
|
||||
.vscode/
|
||||
*.swp
|
||||
*.swo
|
||||
|
||||
# OS specific files
|
||||
.DS_Store
|
||||
Thumbs.db
|
||||
|
||||
# Local testing/development
|
||||
*.bak
|
||||
*.tmp
|
||||
.coverage
|
||||
htmlcov/
|
||||
|
||||
# Exclude the HTML file downloaded from HF (if you keep test files locally)
|
||||
*Daily Papers - Hugging Face.html
|
||||
|
||||
# Uncomment the line below if you don't want to commit generated feed files
|
||||
# *.xml
|
||||
21
LICENSE
Normal file
21
LICENSE
Normal file
@@ -0,0 +1,21 @@
|
||||
MIT License
|
||||
|
||||
Copyright (c) 2024
|
||||
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
of this software and associated documentation files (the "Software"), to deal
|
||||
in the Software without restriction, including without limitation the rights
|
||||
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||
copies of the Software, and to permit persons to whom the Software is
|
||||
furnished to do so, subject to the following conditions:
|
||||
|
||||
The above copyright notice and this permission notice shall be included in all
|
||||
copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
||||
SOFTWARE.
|
||||
100
README.md
Normal file
100
README.md
Normal file
@@ -0,0 +1,100 @@
|
||||
# Hugging Face Papers RSS Feed Generator
|
||||
|
||||
This project automatically generates RSS feeds for the daily, weekly, and monthly curated papers listed on the Hugging Face website.
|
||||
Since Hugging Face doesn't provide official RSS feeds for these pages, this project bridges that gap.
|
||||
|
||||
## Features
|
||||
|
||||
* Fetches paper data (title, link, authors, summary, publication date, thumbnail, upvotes) directly from Hugging Face.
|
||||
* Generates standard RSS 2.0 feeds.
|
||||
* Automated updates using GitHub Actions.
|
||||
* Provides separate feeds for daily, weekly, and monthly papers.
|
||||
|
||||
## Generated Feeds
|
||||
|
||||
The following feed files are automatically generated and updated in this repository:
|
||||
|
||||
* **Daily:** [`feed.xml`](./feed.xml)
|
||||
* Updates daily around midnight UTC.
|
||||
* Sources from: `https://huggingface.co/papers`
|
||||
* **Weekly:** [`feed_weekly.xml`](./feed_weekly.xml)
|
||||
* Updates every Monday around midnight UTC.
|
||||
* Sources from: `https://huggingface.co/papers/week/YYYY-Www` (dynamic URL)
|
||||
* **Monthly:** [`feed_monthly.xml`](./feed_monthly.xml)
|
||||
* Updates on the 1st of every month around midnight UTC.
|
||||
* Sources from: `https://huggingface.co/papers/month/YYYY-MM` (dynamic URL)
|
||||
|
||||
You can subscribe to these feeds using your favorite RSS reader by using the raw file URL (e.g., `https://raw.githubusercontent.com/YOUR_USERNAME/YOUR_REPOSITORY/main/feed.xml`).
|
||||
|
||||
## How it Works
|
||||
|
||||
1. **Parsing:** A Python script (`parser.py`) fetches the HTML content of the relevant Hugging Face papers page.
|
||||
2. **Data Extraction:** It uses BeautifulSoup and JSON parsing to extract the paper details embedded within the page's HTML.
|
||||
3. **RSS Generation:** Another Python script (`rss_generator.py`) uses the `feedgen` library to construct the RSS feed from the extracted data.
|
||||
4. **Automation:** GitHub Actions workflows (`.github/workflows/`) are scheduled to run automatically:
|
||||
* The daily workflow runs `parser.py` targeting the main papers page.
|
||||
* The weekly/monthly workflows calculate the correct URL for the current week/month and then run `parser.py`.
|
||||
* If the generated feed file has changed, the workflow commits and pushes the update to the repository.
|
||||
|
||||
## Local Execution Tutorial
|
||||
|
||||
You can also run the script locally to generate the feeds manually.
|
||||
|
||||
**Prerequisites:**
|
||||
|
||||
* Python 3.7+
|
||||
* pip (Python package installer)
|
||||
* Git
|
||||
|
||||
**Steps:**
|
||||
|
||||
1. **Clone the repository:**
|
||||
```bash
|
||||
git clone https://github.com/YOUR_USERNAME/YOUR_REPOSITORY.git
|
||||
cd YOUR_REPOSITORY
|
||||
```
|
||||
(Replace `YOUR_USERNAME/YOUR_REPOSITORY` with the actual path to this repo).
|
||||
|
||||
2. **Set up a virtual environment (Recommended):**
|
||||
```bash
|
||||
python -m venv venv
|
||||
source venv/bin/activate # On Windows use `venv\Scripts\activate`
|
||||
```
|
||||
|
||||
3. **Install dependencies:**
|
||||
```bash
|
||||
pip install -r requirements.txt
|
||||
```
|
||||
|
||||
4. **Run the parser:**
|
||||
The `parser.py` script handles both fetching/parsing and calling the RSS generator.
|
||||
|
||||
* **Generate Daily Feed (Default):** Fetches from the main `/papers` URL.
|
||||
```bash
|
||||
python parser.py
|
||||
```
|
||||
This will create/update `feed.xml`.
|
||||
|
||||
* **Generate Weekly Feed:** You need to provide the specific weekly URL.
|
||||
```bash
|
||||
# Replace YYYY-Www with the desired week, e.g., 2025-W17
|
||||
python parser.py --source https://huggingface.co/papers/week/YYYY-Www --output feed_weekly.xml
|
||||
```
|
||||
|
||||
* **Generate Monthly Feed:** Provide the specific monthly URL.
|
||||
```bash
|
||||
# Replace YYYY-MM with the desired month, e.g., 2025-04
|
||||
python parser.py --source https://huggingface.co/papers/month/YYYY-MM --output feed_monthly.xml
|
||||
```
|
||||
|
||||
* **Use a local HTML file (for testing):**
|
||||
```bash
|
||||
# Make sure 'local_papers.html' exists
|
||||
python parser.py --source local_papers.html --output test_feed.xml
|
||||
```
|
||||
|
||||
5. **Find the output:** The generated RSS feed file (`feed.xml`, `feed_weekly.xml`, etc.) will be created in the project's root directory.
|
||||
|
||||
## Contributing
|
||||
|
||||
Feel free to open issues or pull requests if you find bugs or have suggestions for improvement.
|
||||
144
parser.py
Normal file
144
parser.py
Normal file
@@ -0,0 +1,144 @@
|
||||
import json
|
||||
import requests
|
||||
from bs4 import BeautifulSoup
|
||||
from rss_generator import generate_rss_feed
|
||||
import argparse
|
||||
|
||||
def fetch_html(url):
|
||||
"""Fetches HTML content from a given URL."""
|
||||
try:
|
||||
headers = {
|
||||
'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/91.0.4472.124 Safari/537.36'
|
||||
}
|
||||
response = requests.get(url, headers=headers, timeout=30)
|
||||
response.raise_for_status()
|
||||
return response.text
|
||||
except requests.exceptions.RequestException as e:
|
||||
print(f"Error fetching URL {url}: {e}")
|
||||
return None
|
||||
|
||||
def parse_daily_papers(source):
|
||||
"""Parses the Hugging Face Daily Papers HTML to extract paper data.
|
||||
|
||||
Args:
|
||||
source (str): URL of the daily papers page or path to a local HTML file.
|
||||
|
||||
Returns:
|
||||
list: A list of dictionaries, where each dictionary contains
|
||||
information about a paper.
|
||||
Returns an empty list if the data cannot be found or parsed.
|
||||
"""
|
||||
html_content = None
|
||||
if source.startswith('http://') or source.startswith('https://'):
|
||||
print(f"Fetching HTML from URL: {source}")
|
||||
html_content = fetch_html(source)
|
||||
else:
|
||||
print(f"Reading HTML from file: {source}")
|
||||
try:
|
||||
with open(source, 'r', encoding='utf-8') as f:
|
||||
html_content = f.read()
|
||||
except FileNotFoundError:
|
||||
print(f"Error: File not found at {source}")
|
||||
return []
|
||||
|
||||
if not html_content:
|
||||
print("Error: Could not get HTML content.")
|
||||
return []
|
||||
|
||||
try:
|
||||
soup = BeautifulSoup(html_content, 'lxml')
|
||||
|
||||
# Find the div containing the paper data
|
||||
papers_div = soup.find('div', attrs={'data-target': 'DailyPapers'})
|
||||
if not papers_div:
|
||||
print("Error: Could not find the 'DailyPapers' div.")
|
||||
return []
|
||||
|
||||
# Extract the JSON data from the 'data-props' attribute
|
||||
data_props_json = papers_div.get('data-props')
|
||||
if not data_props_json:
|
||||
print("Error: Could not find 'data-props' attribute.")
|
||||
return []
|
||||
|
||||
# Parse the JSON data
|
||||
data = json.loads(data_props_json)
|
||||
|
||||
# Extract paper information
|
||||
papers_list = []
|
||||
if 'dailyPapers' in data and isinstance(data['dailyPapers'], list):
|
||||
for item in data['dailyPapers']:
|
||||
paper_info = item.get('paper')
|
||||
if paper_info and isinstance(paper_info, dict):
|
||||
paper_id = paper_info.get('id')
|
||||
title = paper_info.get('title', 'N/A').replace('\n', ' ').strip()
|
||||
summary = paper_info.get('summary', 'N/A').replace('\n', ' ').strip()
|
||||
published_at = paper_info.get('publishedAt') # Keep as string for now
|
||||
link = f"https://arxiv.org/abs/{paper_id}" if paper_id else 'N/A'
|
||||
|
||||
authors_list = paper_info.get('authors', [])
|
||||
author_names = [author.get('name', 'Unknown') for author in authors_list if isinstance(author, dict)]
|
||||
authors_str = ", ".join(author_names)
|
||||
|
||||
# Extract thumbnail and upvotes from the parent 'item' dictionary
|
||||
thumbnail = item.get('thumbnail', None)
|
||||
upvotes = paper_info.get('upvotes', 0) # Upvotes seem to be inside paper_info
|
||||
|
||||
papers_list.append({
|
||||
'id': paper_id,
|
||||
'title': title,
|
||||
'link': link,
|
||||
'authors': authors_str,
|
||||
'summary': summary,
|
||||
'published_at': published_at,
|
||||
'thumbnail': thumbnail,
|
||||
'upvotes': upvotes
|
||||
})
|
||||
else:
|
||||
print("Error: 'dailyPapers' key not found or not a list in JSON data.")
|
||||
return []
|
||||
|
||||
return papers_list
|
||||
|
||||
except json.JSONDecodeError:
|
||||
print("Error: Could not decode JSON from data.")
|
||||
return []
|
||||
except Exception as e:
|
||||
print(f"An unexpected error occurred during parsing: {e}")
|
||||
return []
|
||||
|
||||
if __name__ == "__main__":
|
||||
# Setup argument parser
|
||||
parser = argparse.ArgumentParser(description='Parse Hugging Face Daily Papers and generate RSS feed.')
|
||||
parser.add_argument(
|
||||
'--source',
|
||||
type=str,
|
||||
default='https://huggingface.co/papers',
|
||||
help='URL of the Hugging Face papers page or path to a local HTML file.'
|
||||
)
|
||||
parser.add_argument(
|
||||
'--output',
|
||||
type=str,
|
||||
default='feed.xml',
|
||||
help='Path to save the generated RSS feed file.'
|
||||
)
|
||||
args = parser.parse_args()
|
||||
|
||||
html_source = args.source
|
||||
rss_file = args.output
|
||||
|
||||
extracted_papers = parse_daily_papers(html_source)
|
||||
|
||||
if extracted_papers:
|
||||
print(f"Successfully extracted {len(extracted_papers)} papers from {html_source}.")
|
||||
|
||||
# Print details of the first paper as a sample, including new fields
|
||||
if extracted_papers:
|
||||
print("\n--- Sample Paper --- ")
|
||||
for key, value in extracted_papers[0].items():
|
||||
print(f"{key.capitalize()}: {value}")
|
||||
print("-------------------")
|
||||
|
||||
# Generate and save the RSS feed
|
||||
generate_rss_feed(extracted_papers, rss_file)
|
||||
else:
|
||||
print(f"Failed to extract papers from {html_source}.")
|
||||
4
requirements.txt
Normal file
4
requirements.txt
Normal file
@@ -0,0 +1,4 @@
|
||||
beautifulsoup4
|
||||
lxml
|
||||
feedgen
|
||||
requests
|
||||
67
rss_generator.py
Normal file
67
rss_generator.py
Normal file
@@ -0,0 +1,67 @@
|
||||
from feedgen.feed import FeedGenerator
|
||||
from datetime import datetime
|
||||
import pytz # To handle timezone awareness
|
||||
|
||||
def generate_rss_feed(papers_list, feed_filepath):
|
||||
"""Generates an RSS feed from the list of papers and saves it to a file.
|
||||
|
||||
Args:
|
||||
papers_list (list): A list of paper dictionaries from the parser.
|
||||
feed_filepath (str): The path to save the generated RSS feed file.
|
||||
"""
|
||||
fg = FeedGenerator()
|
||||
fg.title('Hugging Face Daily Papers')
|
||||
fg.link(href='https://huggingface.co/papers', rel='alternate') # Link to the source page
|
||||
# Use the link of the first paper as the feed ID, assuming papers are sorted by date
|
||||
fg.id(papers_list[0]['link'] if papers_list else 'tag:huggingface.co,2024:papers/daily')
|
||||
fg.description('Daily research papers curated by the Hugging Face community.')
|
||||
fg.language('en')
|
||||
|
||||
# Sort papers by published_at date, newest first
|
||||
# Handle potential None values in published_at
|
||||
papers_list.sort(key=lambda p: p.get('published_at') or '1970-01-01T00:00:00.000Z', reverse=True)
|
||||
|
||||
|
||||
for paper in papers_list:
|
||||
fe = fg.add_entry()
|
||||
fe.title(paper['title'])
|
||||
fe.link(href=paper['link'])
|
||||
fe.id(paper['link']) # Use the ArXiv link as the unique identifier
|
||||
|
||||
# Build the description with thumbnail, authors, upvotes, and summary
|
||||
description = ""
|
||||
if paper.get('thumbnail'):
|
||||
description += f'<p><img src="{paper["thumbnail"]}" alt="Paper thumbnail" style="max-width: 300px; height: auto;" /></p>'
|
||||
|
||||
description += f"<p><b>Authors:</b> {paper.get('authors', 'N/A')}</p>"
|
||||
description += f"<p><b>Upvotes:</b> {paper.get('upvotes', 0)}</p>" # Added Upvotes
|
||||
description += f"<p><b>Summary:</b> {paper.get('summary', 'N/A')}</p>"
|
||||
|
||||
fe.description(description)
|
||||
|
||||
# Parse and set the publication date
|
||||
pub_date_str = paper.get('published_at')
|
||||
if pub_date_str:
|
||||
try:
|
||||
# Parse the ISO 8601 format string
|
||||
pub_date = datetime.fromisoformat(pub_date_str.replace('Z', '+00:00'))
|
||||
# Ensure it's timezone-aware (UTC)
|
||||
fe.pubDate(pub_date.astimezone(pytz.utc))
|
||||
except ValueError:
|
||||
print(f"Warning: Could not parse date '{pub_date_str}' for paper ID {paper.get('id')}")
|
||||
# Optionally set a default date or leave it out
|
||||
# fe.pubDate(datetime.now(pytz.utc)) # Example: set to now
|
||||
|
||||
# Add authors
|
||||
fe.author(name=paper.get('authors', 'N/A'))
|
||||
|
||||
# Generate the RSS feed as a string
|
||||
rss_feed = fg.rss_str(pretty=True)
|
||||
|
||||
# Save the feed to the specified file
|
||||
try:
|
||||
with open(feed_filepath, 'wb') as f: # Write in binary mode for UTF-8
|
||||
f.write(rss_feed)
|
||||
print(f"RSS feed successfully generated and saved to {feed_filepath}")
|
||||
except IOError as e:
|
||||
print(f"Error writing RSS feed to {feed_filepath}: {e}")
|
||||
Reference in New Issue
Block a user