Downloads · 30 days
0
Hanxchoi/bandcamp-update
bandcamp-update is a machine learning model from Hanxchoi. Use it for the machine learning task on the model card, and read the license before you ship it in a product.
import feedparser import requests from bs4 import BeautifulSoup import time
Downloads · 30 days
0
Access
Public
Updated Jun 3, 2023
Repo size
—
Likes
0
Public
Click a slice to open those files.
.md2.9 KB · 67%
From the Hugging Face model README
import feedparser import requests from bs4 import BeautifulSoup import time
bandcamp_url = 'https://artistname.bandcamp.com/'
rss_feed_url = 'https://www.example.com/rss'
headers = {'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/58.0.3029.110 Safari/537.3'}
d = feedparser.FeedParserDict()
def get_links_from_page(url): response = requests.get(url, headers=headers) soup = BeautifulSoup(response.text, 'html.parser') href_list = [] for a in soup.find_all('a', href=True): if '/track/' in a['href']: href_list.append('https://'+a['href'][2:]) return href_list
def process_links(href_list): items_content = '' for href in href_list: name = href.split('/')[-1].replace('-', ' ').title() items_content += f'<item><title>New Music from {name}</title><link>{href}</link><description>New music released by {name}. Check it out!</description></item>' return items_content
while True: href_list = get_links_from_page(bandcamp_url) if not d.entries: for href in href_list: d.entries.append({'title': href.split('/')[-1].replace('-', ' ').title(), 'link': href}) items_content = process_links(href_list) rss_content = f'''<?xml version="1.0" encoding="UTF-8"?> <rss version="2.0"> <channel> <title>New Music from {bandcamp_url}</title> <link>{bandcamp_url}</link> <description>New releases by your favorite Bandcamp artists.</description> {items_content} </channel> </rss>''' requests.post(rss_feed_url, data=rss_content.encode('utf-8'), headers={'Content-type': 'application/rss+xml'}) else: for href in href_list: if href not in [entry.link for entry in d.entries]: d.entries.append({'title': href.split('/')[-1].replace('-', ' ').title(), 'link': href}) items_content = process_links([href]) rss_content = f'''<?xml version="1.0" encoding="UTF-8"?> <rss version="2.0"> <channel> <title>New Music from {bandcamp_url}</title> <link>{bandcamp_url}</link> <description>New releases by your favorite Bandcamp artists.</description> {items_content} </channel> </rss>''' requests.post(rss_feed_url, data=rss_content.encode('utf-8'), headers={'Content-type': 'application/rss+xml'}) time.sleep(60*30) # Wait for 30 minutes and repeat