70 lines
1.7 KiB
Python
70 lines
1.7 KiB
Python
"""
|
|
When there is an error there is no output but an error.
|
|
All in german.
|
|
"""
|
|
|
|
import json
|
|
import re
|
|
import sys
|
|
|
|
import requests
|
|
from bs4 import BeautifulSoup
|
|
|
|
DATE, MENSA = sys.argv[1:3]
|
|
URL = f"https://www.studierendenwerk-goettingen.de/fileadmin/templates/php/mensaspeiseplan/cached/de/{DATE}/{MENSA}.html"
|
|
|
|
r = requests.get(URL)
|
|
if r.status_code == 404:
|
|
sys.exit(1)
|
|
|
|
soup = BeautifulSoup(r.text, "html.parser")
|
|
table = soup.select_one("table.sp_tab")
|
|
|
|
if not table or "kein Angebot" in table.get_text():
|
|
sys.exit(1)
|
|
|
|
dishes = []
|
|
|
|
for row in table.select("tr")[1:]:
|
|
cells = row.select("td")
|
|
if len(cells) != 3:
|
|
continue
|
|
|
|
typ, bez, hin = cells
|
|
strong = bez.select_one("strong")
|
|
raw_title = strong.get_text(" ", strip=True) if strong else ""
|
|
|
|
title = re.sub(r"\s*\([^)]*\)", "", raw_title)
|
|
|
|
description = bez.get_text(" ", strip=True).removeprefix(raw_title).strip()
|
|
angebot = bez.select_one("i.smaller")
|
|
if angebot:
|
|
description = description.removesuffix(
|
|
angebot.get_text(" ", strip=True)
|
|
).strip()
|
|
|
|
allergens = [
|
|
x.strip()
|
|
for group in re.findall(r"\(([^()]*)\)", raw_title + " " + description)
|
|
for x in group.split(",")
|
|
]
|
|
|
|
category = [
|
|
str(img["src"]).rsplit("/", 1)[-1].rsplit(".", 1)[0]
|
|
for img in hin.select("img")
|
|
]
|
|
|
|
dishes.append(
|
|
{
|
|
"date": DATE,
|
|
"mensa": MENSA,
|
|
"type": typ.get_text(strip=True),
|
|
"title": title,
|
|
"description": description,
|
|
"category": category,
|
|
"allergens": allergens,
|
|
}
|
|
)
|
|
|
|
print(json.dumps(dishes, ensure_ascii=False))
|