Add historical radioeins parser support
This commit is contained in:
@@ -160,6 +160,47 @@ def extract_votes_from_table(
|
||||
return [rows_by_rank[rank] for rank in range(1, 11)]
|
||||
|
||||
|
||||
def extract_votes_from_historical_text(
|
||||
soup,
|
||||
juror: str,
|
||||
source_url: str,
|
||||
) -> list[Vote] | None:
|
||||
rows_by_rank: dict[int, Vote] = {}
|
||||
|
||||
for element in soup.find_all(["tr", "li", "p"]):
|
||||
text = clean(element.get_text(" ", strip=True))
|
||||
match = re.fullmatch(r"(10|[1-9])\.\s*(.+)", text)
|
||||
if not match:
|
||||
continue
|
||||
|
||||
rank = int(match.group(1))
|
||||
entry = match.group(2)
|
||||
artist, separator, title = entry.partition(":")
|
||||
if not separator:
|
||||
artist, separator, title = entry.partition(" - ")
|
||||
if not separator:
|
||||
continue
|
||||
|
||||
artist = clean(artist)
|
||||
title = clean(title)
|
||||
if not artist or not title:
|
||||
continue
|
||||
|
||||
rows_by_rank[rank] = Vote(
|
||||
juror=juror,
|
||||
rank=rank,
|
||||
points=POINTS[rank],
|
||||
artist=artist,
|
||||
title=title,
|
||||
source_url=source_url,
|
||||
)
|
||||
|
||||
if set(rows_by_rank) != set(range(1, 11)):
|
||||
return None
|
||||
|
||||
return [rows_by_rank[rank] for rank in range(1, 11)]
|
||||
|
||||
|
||||
def parse_jury_page(
|
||||
html: str,
|
||||
fallback_name: str,
|
||||
@@ -173,6 +214,10 @@ def parse_jury_page(
|
||||
if votes is not None:
|
||||
return votes
|
||||
|
||||
votes = extract_votes_from_historical_text(soup, juror, source_url)
|
||||
if votes is not None:
|
||||
return votes
|
||||
|
||||
raise RuntimeError("Keine vollständige Top-10-Tabelle gefunden")
|
||||
|
||||
|
||||
|
||||
Reference in New Issue
Block a user