Add historical radioeins parser support

This commit is contained in:
2026-08-09 12:19:14 +02:00
parent 355ea3e509
commit bbf0f1bf0a
5 changed files with 141 additions and 6 deletions
+45
View File
@@ -160,6 +160,47 @@ def extract_votes_from_table(
return [rows_by_rank[rank] for rank in range(1, 11)]
def extract_votes_from_historical_text(
soup,
juror: str,
source_url: str,
) -> list[Vote] | None:
rows_by_rank: dict[int, Vote] = {}
for element in soup.find_all(["tr", "li", "p"]):
text = clean(element.get_text(" ", strip=True))
match = re.fullmatch(r"(10|[1-9])\.\s*(.+)", text)
if not match:
continue
rank = int(match.group(1))
entry = match.group(2)
artist, separator, title = entry.partition(":")
if not separator:
artist, separator, title = entry.partition(" - ")
if not separator:
continue
artist = clean(artist)
title = clean(title)
if not artist or not title:
continue
rows_by_rank[rank] = Vote(
juror=juror,
rank=rank,
points=POINTS[rank],
artist=artist,
title=title,
source_url=source_url,
)
if set(rows_by_rank) != set(range(1, 11)):
return None
return [rows_by_rank[rank] for rank in range(1, 11)]
def parse_jury_page(
html: str,
fallback_name: str,
@@ -173,6 +214,10 @@ def parse_jury_page(
if votes is not None:
return votes
votes = extract_votes_from_historical_text(soup, juror, source_url)
if votes is not None:
return votes
raise RuntimeError("Keine vollständige Top-10-Tabelle gefunden")