RSSEbra/gen_feed.py

119 lines
4.1 KiB
Python
Raw Normal View History

import requests
import re
import datetime
import xml.dom.minidom
from bs4 import BeautifulSoup
from urllib.parse import urlparse
from zoneinfo import ZoneInfo
from rfeed import *
regex_date = r"/(\d{4})/(\d{2})/(\d{2})/"
regex_time = r"à (\d+):(\d{2})"
class Article:
def __init__(self, title: str, link: str, is_paid: bool):
self.title = title
self.link = link
2026-05-22 20:59:52 +02:00
self.text = ""
2026-07-28 13:34:44 +02:00
self.date = datetime.datetime.now()
self._get_time_and_text()
self.is_paid = is_paid
2026-05-22 20:59:52 +02:00
def _get_time_and_text(self):
print(f" Retrieving {self.link} to get pub time...")
2026-07-28 13:34:44 +02:00
response = requests.get(self.link)
print(f" Response is <{response.status_code}>")
if response.status_code != 200:
2026-07-28 13:34:44 +02:00
return
soup = BeautifulSoup(response.text, 'html.parser')
2026-05-22 20:59:52 +02:00
2026-07-28 13:34:44 +02:00
# Handle time
publish_date_time = soup.find("meta", property="article:published_time")["content"]
self.date = datetime.datetime.strptime(publish_date_time, '%Y-%m-%dT%H:%M:%SZ')
self.date = self.date.replace(tzinfo=ZoneInfo("UTC"))
2026-05-22 20:59:52 +02:00
# Handle text
text = []
if soup.find("div", class_="chapo") is not None:
text.append(soup.find("div", class_="chapo").text)
paragraphs = soup.find_all("div", class_="textComponent")
for p in paragraphs:
text.append(p.text)
self.text = "\n".join(text)
2026-07-28 13:34:44 +02:00
# Handle image if any
img_url = soup.find("meta", property="og:image")
if img_url is not None:
self.text = f"<img alt=\"Image illustrant l'article\" src=\"{img_url["content"]}\" />\n" + self.text
def full_title(self):
paid = ""
if self.is_paid:
paid = "[€] "
return f'{paid}{self.title}'
def __str__(self):
paid = ""
if self.is_paid:
paid = "[€] "
return f'{paid}{self.title} [{self.date}]'
def __repr__(self):
return self.__str__()
def generate_feed(town_url, feed_url, feed_path):
main_domain = urlparse(town_url).netloc
print(f"Retrieving {town_url}...")
response = requests.get(town_url)
if response.status_code != 200:
print(f"Failed to get url: {response.status_code}")
return False
soup = BeautifulSoup(response.text, 'html.parser')
articles = []
small_articles = soup.find(id="ListUneMain").find_all("div", class_="wrapperClickArticle") + soup.find(id="ListUneSecondary").find_all("div", class_="wrapperClickArticle")
print(f"Found {len(small_articles)} articles...")
for small_article in small_articles:
t = small_article.find_all("span")[1].text
2026-07-07 22:58:39 +02:00
if small_article.find("a").get("href").startswith("https"):
link = small_article.find("a").get("href")
else:
link = "https://" + main_domain + small_article.find("a").get("href")
if "/videos/" in link:
print(f" Skipping {link} since it's a video")
continue
is_paid = len(small_article.find_all("span", class_="flagPaid")) > 0
articles.append(Article(t, link, is_paid))
print("Generating feed...")
items = []
for a in articles:
2026-05-22 20:59:52 +02:00
item = Item(title = a.full_title(), link= a.link, guid = Guid(a.link), pubDate = a.date, description = a.text)
items.append(item)
feed = Feed(title = soup.title.string,
link = feed_url,
description = soup.title.string + " - Derniers articles",
language = "fr-FR",
lastBuildDate = datetime.datetime.now(datetime.UTC),
items = items)
print("Writing feed...")
with open(feed_path, 'w') as out_rss:
dom = xml.dom.minidom.parseString(feed.rss())
out_rss.write(dom.toprettyxml())
print("All done!")
return True
if __name__ == '__main__':
town_url = "https://www.leprogres.fr/c/rhone/69163-quincieux"
town_url = "https://www.ledauphine.com/c/isere/38425-saint-maurice-l-exil"
town_url = "https://www.ledauphine.com/c/isere/38124-corbelin"
feed_url = "https://www.onnula.fr/infos/38124-corbelin.xml"
feed_path = "38124-corbelin.xml"
generate_feed(town_url, feed_url, feed_path)