diff --git a/gen_feed.py b/gen_feed.py index 76f8a3f..0af08cf 100644 --- a/gen_feed.py +++ b/gen_feed.py @@ -1,16 +1,13 @@ -import requests -import re import datetime import xml.dom.minidom -from bs4 import BeautifulSoup from urllib.parse import urlparse from zoneinfo import ZoneInfo + +import requests +from bs4 import BeautifulSoup from rfeed import * -regex_date = r"/(\d{4})/(\d{2})/(\d{2})/" -regex_time = r"à (\d+):(\d{2})" - class Article: def __init__(self, title: str, link: str, is_paid: bool): self.title = title @@ -19,18 +16,20 @@ class Article: self.date = datetime.datetime.now() self._get_time_and_text() self.is_paid = is_paid - + def _get_time_and_text(self): print(f" Retrieving {self.link} to get pub time...") response = requests.get(self.link) print(f" Response is <{response.status_code}>") if response.status_code != 200: return - soup = BeautifulSoup(response.text, 'html.parser') + soup = BeautifulSoup(response.text, "html.parser") # Handle time - publish_date_time = soup.find("meta", property="article:published_time")["content"] - self.date = datetime.datetime.strptime(publish_date_time, '%Y-%m-%dT%H:%M:%SZ') + publish_date_time = soup.find("meta", property="article:published_time")[ + "content" + ] + self.date = datetime.datetime.strptime(publish_date_time, "%Y-%m-%dT%H:%M:%SZ") self.date = self.date.replace(tzinfo=ZoneInfo("UTC")) # Handle text @@ -45,21 +44,23 @@ class Article: # Handle image if any img_url = soup.find("meta", property="og:image") if img_url is not None: - self.text = f"\"Image\n" + self.text - + self.text = ( + f'Image illustrant l\'article\n' + + self.text + ) def full_title(self): paid = "" if self.is_paid: paid = "[€] " - return f'{paid}{self.title}' + return f"{paid}{self.title}" def __str__(self): paid = "" if self.is_paid: paid = "[€] " - return f'{paid}{self.title} [{self.date}]' - + return f"{paid}{self.title} [{self.date}]" + def __repr__(self): return self.__str__() @@ -73,10 +74,12 @@ def generate_feed(town_url, feed_url, feed_path): print(f"Failed to get url: {response.status_code}") return False - soup = BeautifulSoup(response.text, 'html.parser') + soup = BeautifulSoup(response.text, "html.parser") articles = [] - small_articles = soup.find(id="ListUneMain").find_all("div", class_="wrapperClickArticle") + soup.find(id="ListUneSecondary").find_all("div", class_="wrapperClickArticle") + small_articles = soup.find(id="ListUneMain").find_all( + "div", class_="wrapperClickArticle" + ) + soup.find(id="ListUneSecondary").find_all("div", class_="wrapperClickArticle") print(f"Found {len(small_articles)} articles...") for small_article in small_articles: t = small_article.find_all("span")[1].text @@ -93,27 +96,35 @@ def generate_feed(town_url, feed_url, feed_path): print("Generating feed...") items = [] for a in articles: - item = Item(title = a.full_title(), link= a.link, guid = Guid(a.link), pubDate = a.date, description = a.text) + item = Item( + title=a.full_title(), + link=a.link, + guid=Guid(a.link), + pubDate=a.date, + description=a.text, + ) items.append(item) - feed = Feed(title = soup.title.string, - link = feed_url, - description = soup.title.string + " - Derniers articles", - language = "fr-FR", - lastBuildDate = datetime.datetime.now(datetime.UTC), - items = items) - + feed = Feed( + title=soup.title.string, + link=feed_url, + description=soup.title.string + " - Derniers articles", + language="fr-FR", + lastBuildDate=datetime.datetime.now(datetime.UTC), + items=items, + ) + print("Writing feed...") - with open(feed_path, 'w') as out_rss: + with open(feed_path, "w") as out_rss: dom = xml.dom.minidom.parseString(feed.rss()) out_rss.write(dom.toprettyxml()) print("All done!") return True -if __name__ == '__main__': +if __name__ == "__main__": town_url = "https://www.leprogres.fr/c/rhone/69163-quincieux" town_url = "https://www.ledauphine.com/c/isere/38425-saint-maurice-l-exil" town_url = "https://www.ledauphine.com/c/isere/38124-corbelin" feed_url = "https://www.onnula.fr/infos/38124-corbelin.xml" feed_path = "38124-corbelin.xml" - generate_feed(town_url, feed_url, feed_path) \ No newline at end of file + generate_feed(town_url, feed_url, feed_path)