Formatting
This commit is contained in:
parent
8f76a16e5a
commit
8a059a722d
1 changed files with 39 additions and 28 deletions
51
gen_feed.py
51
gen_feed.py
|
|
@ -1,16 +1,13 @@
|
|||
import requests
|
||||
import re
|
||||
import datetime
|
||||
import xml.dom.minidom
|
||||
from bs4 import BeautifulSoup
|
||||
from urllib.parse import urlparse
|
||||
from zoneinfo import ZoneInfo
|
||||
|
||||
import requests
|
||||
from bs4 import BeautifulSoup
|
||||
from rfeed import *
|
||||
|
||||
|
||||
regex_date = r"/(\d{4})/(\d{2})/(\d{2})/"
|
||||
regex_time = r"à (\d+):(\d{2})"
|
||||
|
||||
class Article:
|
||||
def __init__(self, title: str, link: str, is_paid: bool):
|
||||
self.title = title
|
||||
|
|
@ -26,11 +23,13 @@ class Article:
|
|||
print(f" Response is <{response.status_code}>")
|
||||
if response.status_code != 200:
|
||||
return
|
||||
soup = BeautifulSoup(response.text, 'html.parser')
|
||||
soup = BeautifulSoup(response.text, "html.parser")
|
||||
|
||||
# Handle time
|
||||
publish_date_time = soup.find("meta", property="article:published_time")["content"]
|
||||
self.date = datetime.datetime.strptime(publish_date_time, '%Y-%m-%dT%H:%M:%SZ')
|
||||
publish_date_time = soup.find("meta", property="article:published_time")[
|
||||
"content"
|
||||
]
|
||||
self.date = datetime.datetime.strptime(publish_date_time, "%Y-%m-%dT%H:%M:%SZ")
|
||||
self.date = self.date.replace(tzinfo=ZoneInfo("UTC"))
|
||||
|
||||
# Handle text
|
||||
|
|
@ -45,20 +44,22 @@ class Article:
|
|||
# Handle image if any
|
||||
img_url = soup.find("meta", property="og:image")
|
||||
if img_url is not None:
|
||||
self.text = f"<img alt=\"Image illustrant l'article\" src=\"{img_url['content']}\" />\n" + self.text
|
||||
|
||||
self.text = (
|
||||
f'<img alt="Image illustrant l\'article" src="{img_url["content"]}" />\n'
|
||||
+ self.text
|
||||
)
|
||||
|
||||
def full_title(self):
|
||||
paid = ""
|
||||
if self.is_paid:
|
||||
paid = "[€] "
|
||||
return f'{paid}{self.title}'
|
||||
return f"{paid}{self.title}"
|
||||
|
||||
def __str__(self):
|
||||
paid = ""
|
||||
if self.is_paid:
|
||||
paid = "[€] "
|
||||
return f'{paid}{self.title} [{self.date}]'
|
||||
return f"{paid}{self.title} [{self.date}]"
|
||||
|
||||
def __repr__(self):
|
||||
return self.__str__()
|
||||
|
|
@ -73,10 +74,12 @@ def generate_feed(town_url, feed_url, feed_path):
|
|||
print(f"Failed to get url: {response.status_code}")
|
||||
return False
|
||||
|
||||
soup = BeautifulSoup(response.text, 'html.parser')
|
||||
soup = BeautifulSoup(response.text, "html.parser")
|
||||
articles = []
|
||||
|
||||
small_articles = soup.find(id="ListUneMain").find_all("div", class_="wrapperClickArticle") + soup.find(id="ListUneSecondary").find_all("div", class_="wrapperClickArticle")
|
||||
small_articles = soup.find(id="ListUneMain").find_all(
|
||||
"div", class_="wrapperClickArticle"
|
||||
) + soup.find(id="ListUneSecondary").find_all("div", class_="wrapperClickArticle")
|
||||
print(f"Found {len(small_articles)} articles...")
|
||||
for small_article in small_articles:
|
||||
t = small_article.find_all("span")[1].text
|
||||
|
|
@ -93,24 +96,32 @@ def generate_feed(town_url, feed_url, feed_path):
|
|||
print("Generating feed...")
|
||||
items = []
|
||||
for a in articles:
|
||||
item = Item(title = a.full_title(), link= a.link, guid = Guid(a.link), pubDate = a.date, description = a.text)
|
||||
item = Item(
|
||||
title=a.full_title(),
|
||||
link=a.link,
|
||||
guid=Guid(a.link),
|
||||
pubDate=a.date,
|
||||
description=a.text,
|
||||
)
|
||||
items.append(item)
|
||||
feed = Feed(title = soup.title.string,
|
||||
feed = Feed(
|
||||
title=soup.title.string,
|
||||
link=feed_url,
|
||||
description=soup.title.string + " - Derniers articles",
|
||||
language="fr-FR",
|
||||
lastBuildDate=datetime.datetime.now(datetime.UTC),
|
||||
items = items)
|
||||
items=items,
|
||||
)
|
||||
|
||||
print("Writing feed...")
|
||||
with open(feed_path, 'w') as out_rss:
|
||||
with open(feed_path, "w") as out_rss:
|
||||
dom = xml.dom.minidom.parseString(feed.rss())
|
||||
out_rss.write(dom.toprettyxml())
|
||||
print("All done!")
|
||||
return True
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
if __name__ == "__main__":
|
||||
town_url = "https://www.leprogres.fr/c/rhone/69163-quincieux"
|
||||
town_url = "https://www.ledauphine.com/c/isere/38425-saint-maurice-l-exil"
|
||||
town_url = "https://www.ledauphine.com/c/isere/38124-corbelin"
|
||||
|
|
|
|||
Loading…
Reference in a new issue