import datetime import xml.dom.minidom from urllib.parse import urlparse from zoneinfo import ZoneInfo import requests from bs4 import BeautifulSoup from rfeed import * class Article: def __init__(self, title: str, link: str, is_paid: bool): self.title = title self.link = link self.text = "" self.date = datetime.datetime.now() self._get_time_and_text() self.is_paid = is_paid def _get_time_and_text(self): print(f" Retrieving {self.link} to get pub time...") response = requests.get(self.link) print(f" Response is <{response.status_code}>") if response.status_code != 200: return soup = BeautifulSoup(response.text, "html.parser") # Handle time publish_date_time = soup.find("meta", property="article:published_time")[ "content" ] self.date = datetime.datetime.strptime(publish_date_time, "%Y-%m-%dT%H:%M:%SZ").replace(tzinfo=ZoneInfo("UTC")) # Handle text text = [] if soup.find("div", class_="chapo") is not None: text.append(soup.find("div", class_="chapo").text) paragraphs = soup.find_all("div", class_="textComponent") for p in paragraphs: text.append(p.text) self.text = "\n".join(text) # Handle image if any img_url = soup.find("meta", property="og:image") if img_url is not None: self.text = ( f'Image illustrant l\'article\n' + self.text ) def full_title(self): paid = "" if self.is_paid: paid = "[€] " return f"{paid}{self.title}" def __str__(self): paid = "" if self.is_paid: paid = "[€] " return f"{paid}{self.title} [{self.date}]" def __repr__(self): return self.__str__() def generate_feed(town_url, feed_url, feed_path): main_domain = urlparse(town_url).netloc print(f"Retrieving {town_url}...") response = requests.get(town_url) if response.status_code != 200: print(f"Failed to get url: {response.status_code}") return False soup = BeautifulSoup(response.text, "html.parser") articles = [] small_articles = soup.find(id="ListUneMain").find_all( "div", class_="wrapperClickArticle" ) + soup.find(id="ListUneSecondary").find_all("div", class_="wrapperClickArticle") print(f"Found {len(small_articles)} articles...") for small_article in small_articles: t = small_article.find_all("span")[1].text if small_article.find("a").get("href").startswith("https"): link = small_article.find("a").get("href") else: link = "https://" + main_domain + small_article.find("a").get("href") if "/videos/" in link: print(f" Skipping {link} since it's a video") continue is_paid = len(small_article.find_all("span", class_="flagPaid")) > 0 articles.append(Article(t, link, is_paid)) print("Generating feed...") items = [] for a in articles: item = Item( title=a.full_title(), link=a.link, guid=Guid(a.link), pubDate=a.date, description=a.text, ) items.append(item) feed = Feed( title=soup.title.string, link=feed_url, description=soup.title.string + " - Derniers articles", language="fr-FR", lastBuildDate=datetime.datetime.now(datetime.UTC), items=items, ) print("Writing feed...") with open(feed_path, "w") as out_rss: dom = xml.dom.minidom.parseString(feed.rss()) out_rss.write(dom.toprettyxml()) print("All done!") return True if __name__ == "__main__": town_url = "https://www.leprogres.fr/c/rhone/69163-quincieux" town_url = "https://www.ledauphine.com/c/isere/38425-saint-maurice-l-exil" town_url = "https://www.ledauphine.com/c/isere/38124-corbelin" feed_url = "https://www.onnula.fr/infos/38124-corbelin.xml" feed_path = "38124-corbelin.xml" generate_feed(town_url, feed_url, feed_path)