RSSEbra/gen_feed.py

130 lines
4.2 KiB
Python
Raw Normal View History

import datetime
import xml.dom.minidom
from urllib.parse import urlparse
from zoneinfo import ZoneInfo
2026-07-28 17:33:28 +02:00
import requests
from bs4 import BeautifulSoup
from rfeed import *
class Article:
def __init__(self, title: str, link: str, is_paid: bool):
self.title = title
self.link = link
2026-05-22 20:59:52 +02:00
self.text = ""
2026-07-28 13:34:44 +02:00
self.date = datetime.datetime.now()
self._get_time_and_text()
self.is_paid = is_paid
2026-07-28 17:33:28 +02:00
2026-05-22 20:59:52 +02:00
def _get_time_and_text(self):
print(f" Retrieving {self.link} to get pub time...")
2026-07-28 13:34:44 +02:00
response = requests.get(self.link)
print(f" Response is <{response.status_code}>")
if response.status_code != 200:
2026-07-28 13:34:44 +02:00
return
2026-07-28 17:33:28 +02:00
soup = BeautifulSoup(response.text, "html.parser")
2026-05-22 20:59:52 +02:00
2026-07-28 13:34:44 +02:00
# Handle time
2026-07-28 17:33:28 +02:00
publish_date_time = soup.find("meta", property="article:published_time")[
"content"
]
self.date = datetime.datetime.strptime(publish_date_time, "%Y-%m-%dT%H:%M:%SZ")
2026-07-28 13:34:44 +02:00
self.date = self.date.replace(tzinfo=ZoneInfo("UTC"))
2026-05-22 20:59:52 +02:00
# Handle text
text = []
if soup.find("div", class_="chapo") is not None:
text.append(soup.find("div", class_="chapo").text)
paragraphs = soup.find_all("div", class_="textComponent")
for p in paragraphs:
text.append(p.text)
self.text = "\n".join(text)
2026-07-28 13:34:44 +02:00
# Handle image if any
img_url = soup.find("meta", property="og:image")
if img_url is not None:
2026-07-28 17:33:28 +02:00
self.text = (
f'<img alt="Image illustrant l\'article" src="{img_url["content"]}" />\n'
+ self.text
)
def full_title(self):
paid = ""
if self.is_paid:
paid = "[€] "
2026-07-28 17:33:28 +02:00
return f"{paid}{self.title}"
def __str__(self):
paid = ""
if self.is_paid:
paid = "[€] "
2026-07-28 17:33:28 +02:00
return f"{paid}{self.title} [{self.date}]"
def __repr__(self):
return self.__str__()
def generate_feed(town_url, feed_url, feed_path):
main_domain = urlparse(town_url).netloc
print(f"Retrieving {town_url}...")
response = requests.get(town_url)
if response.status_code != 200:
print(f"Failed to get url: {response.status_code}")
return False
2026-07-28 17:33:28 +02:00
soup = BeautifulSoup(response.text, "html.parser")
articles = []
2026-07-28 17:33:28 +02:00
small_articles = soup.find(id="ListUneMain").find_all(
"div", class_="wrapperClickArticle"
) + soup.find(id="ListUneSecondary").find_all("div", class_="wrapperClickArticle")
print(f"Found {len(small_articles)} articles...")
for small_article in small_articles:
t = small_article.find_all("span")[1].text
2026-07-07 22:58:39 +02:00
if small_article.find("a").get("href").startswith("https"):
link = small_article.find("a").get("href")
else:
link = "https://" + main_domain + small_article.find("a").get("href")
if "/videos/" in link:
print(f" Skipping {link} since it's a video")
continue
is_paid = len(small_article.find_all("span", class_="flagPaid")) > 0
articles.append(Article(t, link, is_paid))
print("Generating feed...")
items = []
for a in articles:
2026-07-28 17:33:28 +02:00
item = Item(
title=a.full_title(),
link=a.link,
guid=Guid(a.link),
pubDate=a.date,
description=a.text,
)
items.append(item)
2026-07-28 17:33:28 +02:00
feed = Feed(
title=soup.title.string,
link=feed_url,
description=soup.title.string + " - Derniers articles",
language="fr-FR",
lastBuildDate=datetime.datetime.now(datetime.UTC),
items=items,
)
print("Writing feed...")
2026-07-28 17:33:28 +02:00
with open(feed_path, "w") as out_rss:
dom = xml.dom.minidom.parseString(feed.rss())
out_rss.write(dom.toprettyxml())
print("All done!")
return True
2026-07-28 17:33:28 +02:00
if __name__ == "__main__":
town_url = "https://www.leprogres.fr/c/rhone/69163-quincieux"
town_url = "https://www.ledauphine.com/c/isere/38425-saint-maurice-l-exil"
town_url = "https://www.ledauphine.com/c/isere/38124-corbelin"
feed_url = "https://www.onnula.fr/infos/38124-corbelin.xml"
feed_path = "38124-corbelin.xml"
2026-07-28 17:33:28 +02:00
generate_feed(town_url, feed_url, feed_path)