Files
radioeins_charts/radioeins_top100.py
T

486 lines
14 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
# -*- coding: utf-8 -*-
from __future__ import annotations
import argparse
import re
import sys
import time
import unicodedata
from collections import Counter, defaultdict
from dataclasses import asdict, dataclass
from pathlib import Path
from urllib.parse import urljoin, urlparse
import pandas as pd
import requests
from bs4 import BeautifulSoup
DEFAULT_URL = "https://www.radioeins.de/musik/top_100/2026/deutscher-hip-hop/"
HEADERS = {
"User-Agent": (
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) "
"AppleWebKit/537.36 Chrome/142 Safari/537.36"
)
}
POINTS = {
1: 12,
2: 10,
3: 8,
4: 7,
5: 6,
6: 5,
7: 4,
8: 3,
9: 2,
10: 1,
}
@dataclass
class Vote:
juror: str
rank: int
points: int
artist: str
title: str
source_url: str
class IncompleteJuryListError(RuntimeError):
def __init__(self, votes: list[Vote]):
self.votes = votes
self.found_ranks = [vote.rank for vote in votes]
self.missing_ranks = sorted(set(range(1, 11)) - set(self.found_ranks))
super().__init__("Keine vollständige Top-10-Tabelle gefunden")
def clean(value: str) -> str:
return re.sub(r"\s+", " ", value.replace("\xa0", " ")).strip()
def normalize_title(value: str) -> str:
value = unicodedata.normalize("NFKC", clean(value)).casefold()
value = value.replace("’", "'").replace("`", "'").replace("´", "'")
value = re.sub(r"[‐‑‒–—―_-]+", " ", value)
value = re.sub(r"\s+", " ", value)
return value.strip()
def derive_output_dir(overview_url: str) -> Path:
page_name = Path(urlparse(overview_url).path.rstrip("/")).name
return Path(f"output_{page_name}") if page_name else Path("output")
def get_html(session: requests.Session, url: str, retries: int = 3) -> str:
last_error: Exception | None = None
for attempt in range(1, retries + 1):
try:
response = session.get(url, timeout=30)
response.raise_for_status()
response.encoding = response.apparent_encoding or "utf-8"
return response.text
except requests.RequestException as exc:
last_error = exc
if attempt < retries:
time.sleep(attempt)
raise RuntimeError(f"Seite konnte nicht geladen werden: {url}") from last_error
def discover_jury_pages(
session: requests.Session,
overview_url: str,
) -> list[tuple[str, str]]:
soup = BeautifulSoup(get_html(session, overview_url), "html.parser")
base_path = urlparse(overview_url).path.rstrip("/") + "/"
pages: dict[str, str] = {}
for link in soup.find_all("a", href=True):
url = urljoin(overview_url, str(link["href"]))
parsed = urlparse(url)
if not parsed.path.startswith(base_path):
continue
if not parsed.path.endswith(".html"):
continue
name = clean(link.get_text(" ", strip=True)).lstrip("- ").strip()
if not name:
name = Path(parsed.path).stem.replace("-", " ").title()
pages[url] = name
return sorted(pages.items(), key=lambda item: item[1].casefold())
def juror_name_from_page(soup: BeautifulSoup, fallback_name: str) -> str:
h1 = soup.find("h1")
if h1:
name = clean(h1.get_text(" ", strip=True))
if name:
return name
og_title = soup.find("meta", attrs={"property": "og:title"})
if og_title and og_title.get("content"):
name = clean(str(og_title["content"]))
if name:
return name
return fallback_name
def extract_votes_from_table(
table,
juror: str,
source_url: str,
) -> list[Vote] | None:
rows_by_rank: dict[int, Vote] = {}
for row in table.find_all("tr"):
cells = row.find_all(["td", "th"])
if len(cells) < 3:
continue
rank_text = clean(cells[0].get_text(" ", strip=True))
rank_match = re.fullmatch(r"(10|[1-9])(?:[.)])?", rank_text)
if not rank_match:
continue
rank = int(rank_match.group(1))
artist = clean(cells[1].get_text(" ", strip=True))
title = clean(cells[2].get_text(" ", strip=True))
if not artist or not title:
continue
rows_by_rank[rank] = Vote(
juror=juror,
rank=rank,
points=POINTS[rank],
artist=artist,
title=title,
source_url=source_url,
)
if set(rows_by_rank) != set(range(1, 11)):
return None
return [rows_by_rank[rank] for rank in range(1, 11)]
def extract_votes_from_historical_text(
soup,
juror: str,
source_url: str,
) -> list[Vote] | None:
rows_by_rank: dict[int, Vote] = {}
for element in soup.find_all(["tr", "li", "p"]):
text = clean(element.get_text(" ", strip=True))
match = re.fullmatch(r"(10|[1-9])(?:\.\s*|\s+)(.+)", text)
if not match:
continue
rank = int(match.group(1))
entry = match.group(2)
artist, separator, title = entry.partition(":")
if not separator:
artist, separator, title = entry.partition(" - ")
if not separator:
artist, separator, title = entry.partition(". ")
if not separator:
continue
artist = clean(artist)
title = clean(title)
if not artist or not title:
continue
rows_by_rank[rank] = Vote(
juror=juror,
rank=rank,
points=POINTS[rank],
artist=artist,
title=title,
source_url=source_url,
)
if not rows_by_rank:
return None
return [rows_by_rank[rank] for rank in sorted(rows_by_rank)]
def parse_jury_page(
html: str,
fallback_name: str,
source_url: str,
) -> list[Vote]:
soup = BeautifulSoup(html, "html.parser")
juror = juror_name_from_page(soup, fallback_name)
for table in soup.find_all("table"):
votes = extract_votes_from_table(table, juror, source_url)
if votes is not None:
return votes
votes = extract_votes_from_historical_text(soup, juror, source_url)
if votes and len(votes) == 10:
return votes
if votes:
raise IncompleteJuryListError(votes)
raise RuntimeError("Keine vollständige Top-10-Tabelle gefunden")
def decide_incomplete_list(
error: IncompleteJuryListError,
*,
accept_incomplete: bool,
interactive: bool,
decision_helper=input,
) -> bool:
vote = error.votes[0]
print("\nUnvollständige Juryliste:")
print(f" Juror: {vote.juror}")
print(f" URL: {vote.source_url}")
print(f" Gefundene Ränge: {', '.join(map(str, error.found_ranks))}")
print(f" Fehlende Ränge: {', '.join(map(str, error.missing_ranks))}")
if accept_incomplete:
return True
if not interactive:
return False
answer = decision_helper("Count this incomplete jury list anyway? [y/N]: ")
return answer.strip().casefold() in {"y", "yes"}
def aggregate(votes: list[Vote]) -> pd.DataFrame:
groups: dict[str, list[Vote]] = defaultdict(list)
for vote in votes:
groups[normalize_title(vote.title)].append(vote)
rows: list[dict[str, object]] = []
for group in groups.values():
title_counts = Counter(vote.title for vote in group)
artist_counts = Counter(vote.artist for vote in group)
display_title = sorted(
title_counts,
key=lambda title: (
-title_counts[title],
-len(title),
title.casefold(),
),
)[0]
rows.append(
{
"Titel": display_title,
"Punkte": sum(vote.points for vote in group),
"Nennungen": len(group),
"Erste Plätze": sum(vote.rank == 1 for vote in group),
"Beste Platzierung": min(vote.rank for vote in group),
"Durchschnittsplatz": round(
sum(vote.rank for vote in group) / len(group),
3,
),
"Künstlerangaben": " | ".join(
artist for artist, _ in artist_counts.most_common()
),
"Titelvarianten": " | ".join(
title for title, _ in title_counts.most_common()
),
}
)
ranking = pd.DataFrame(rows)
if ranking.empty:
return ranking
ranking = ranking.sort_values(
by=[
"Punkte",
"Nennungen",
"Erste Plätze",
"Beste Platzierung",
"Durchschnittsplatz",
"Titel",
],
ascending=[False, False, False, True, True, True],
kind="stable",
).reset_index(drop=True)
ranking.insert(0, "Rang", range(1, len(ranking) + 1))
return ranking
def write_outputs(
output_dir: Path,
votes: list[Vote],
ranking: pd.DataFrame,
errors: list[dict[str, str]],
incomplete_lists: list[dict[str, object]],
) -> None:
output_dir.mkdir(parents=True, exist_ok=True)
votes_df = pd.DataFrame(asdict(vote) for vote in votes)
errors_df = pd.DataFrame(errors)
incomplete_df = pd.DataFrame(
incomplete_lists,
columns=["juror", "url", "found_ranks", "missing_ranks", "accepted"],
)
ranking.head(100).to_csv(
output_dir / "top100.csv",
index=False,
encoding="utf-8-sig",
)
ranking.to_csv(
output_dir / "gesamtwertung.csv",
index=False,
encoding="utf-8-sig",
)
votes_df.to_csv(
output_dir / "einzelwertungen.csv",
index=False,
encoding="utf-8-sig",
)
errors_df.to_csv(
output_dir / "fehler.csv",
index=False,
encoding="utf-8-sig",
)
incomplete_df.to_csv(
output_dir / "incomplete_jury_lists.csv",
index=False,
encoding="utf-8-sig",
)
with pd.ExcelWriter(
output_dir / "radioeins_top100.xlsx",
engine="openpyxl",
) as writer:
ranking.head(100).to_excel(writer, sheet_name="Top 100", index=False)
ranking.to_excel(writer, sheet_name="Gesamtwertung", index=False)
votes_df.to_excel(writer, sheet_name="Einzelwertungen", index=False)
errors_df.to_excel(writer, sheet_name="Fehler", index=False)
for sheet in writer.book.worksheets:
sheet.freeze_panes = "A2"
sheet.auto_filter.ref = sheet.dimensions
for column in sheet.columns:
width = max(len(str(cell.value or "")) for cell in column) + 2
sheet.column_dimensions[column[0].column_letter].width = min(
max(width, 10),
60,
)
def build_argument_parser() -> argparse.ArgumentParser:
parser = argparse.ArgumentParser(
description="radioeins-Jurylisten zu einer Top 100 zusammenfassen"
)
parser.add_argument("url", nargs="?", default=DEFAULT_URL)
parser.add_argument("--limit", type=int, default=None)
parser.add_argument("--delay", type=float, default=0.2)
parser.add_argument("--output", type=Path, default=None)
parser.add_argument(
"--accept-incomplete",
action="store_true",
help="unvollständige historische Jurylisten ohne Nachfrage akzeptieren",
)
return parser
def main() -> int:
args = build_argument_parser().parse_args()
output_dir = args.output or derive_output_dir(args.url)
session = requests.Session()
session.headers.update(HEADERS)
pages = discover_jury_pages(session, args.url)
if args.limit is not None:
pages = pages[: args.limit]
print(f"{len(pages)} Juryseiten gefunden.\n")
all_votes: list[Vote] = []
errors: list[dict[str, str]] = []
incomplete_lists: list[dict[str, object]] = []
valid_jury_lists = 0
for index, (url, name) in enumerate(pages, start=1):
print(f"[{index:>3}/{len(pages)}] {name}")
try:
votes = parse_jury_page(
html=get_html(session, url),
fallback_name=name,
source_url=url,
)
except IncompleteJuryListError as exc:
accepted = decide_incomplete_list(
exc,
accept_incomplete=args.accept_incomplete,
interactive=sys.stdin.isatty(),
)
incomplete_lists.append(
{
"juror": exc.votes[0].juror,
"url": exc.votes[0].source_url,
"found_ranks": ",".join(map(str, exc.found_ranks)),
"missing_ranks": ",".join(map(str, exc.missing_ranks)),
"accepted": accepted,
}
)
if accepted:
all_votes.extend(exc.votes)
valid_jury_lists += 1
else:
errors.append(
{"juror": name, "url": url, "error": str(exc)}
)
except Exception as exc:
errors.append(
{
"juror": name,
"url": url,
"error": str(exc),
}
)
else:
all_votes.extend(votes)
valid_jury_lists += 1
if args.delay > 0:
time.sleep(args.delay)
ranking = aggregate(all_votes)
write_outputs(output_dir, all_votes, ranking, errors, incomplete_lists)
print("\nFertig:")
print(f" gefundene Seiten: {len(pages)}")
print(f" gültige Jurylisten: {valid_jury_lists}")
print(f" Wertungen: {len(all_votes)}")
print(f" verschiedene Titel: {len(ranking)}")
print(f" Fehler: {len(errors)}")
print(f" unvollständig gefunden: {len(incomplete_lists)}")
print(
" unvollständig akzeptiert: "
f"{sum(bool(item['accepted']) for item in incomplete_lists)}"
)
print(
" unvollständig abgelehnt: "
f"{sum(not bool(item['accepted']) for item in incomplete_lists)}"
)
print(f" Ausgabe: {output_dir.resolve()}")
return 1 if errors else 0
if __name__ == "__main__":
sys.exit(main())