# -*- coding: utf-8 -*- from __future__ import annotations import argparse import re import sys import time import unicodedata from collections import Counter, defaultdict from dataclasses import asdict, dataclass from pathlib import Path from urllib.parse import urljoin, urlparse import pandas as pd import requests from bs4 import BeautifulSoup DEFAULT_URL = "https://www.radioeins.de/musik/top_100/2026/deutscher-hip-hop/" HEADERS = { "User-Agent": ( "Mozilla/5.0 (Windows NT 10.0; Win64; x64) " "AppleWebKit/537.36 Chrome/142 Safari/537.36" ) } POINTS = { 1: 12, 2: 10, 3: 8, 4: 7, 5: 6, 6: 5, 7: 4, 8: 3, 9: 2, 10: 1, } @dataclass class Vote: juror: str rank: int points: int artist: str title: str source_url: str def clean(value: str) -> str: return re.sub(r"\s+", " ", value.replace("\xa0", " ")).strip() def normalize_title(value: str) -> str: value = unicodedata.normalize("NFKC", clean(value)).casefold() value = value.replace("’", "'").replace("`", "'").replace("´", "'") value = re.sub(r"[‐‑‒–—―_-]+", " ", value) value = re.sub(r"\s+", " ", value) return value.strip() def derive_output_dir(overview_url: str) -> Path: page_name = Path(urlparse(overview_url).path.rstrip("/")).name return Path(f"output_{page_name}") if page_name else Path("output") def get_html(session: requests.Session, url: str, retries: int = 3) -> str: last_error: Exception | None = None for attempt in range(1, retries + 1): try: response = session.get(url, timeout=30) response.raise_for_status() response.encoding = response.apparent_encoding or "utf-8" return response.text except requests.RequestException as exc: last_error = exc if attempt < retries: time.sleep(attempt) raise RuntimeError(f"Seite konnte nicht geladen werden: {url}") from last_error def discover_jury_pages( session: requests.Session, overview_url: str, ) -> list[tuple[str, str]]: soup = BeautifulSoup(get_html(session, overview_url), "html.parser") base_path = urlparse(overview_url).path.rstrip("/") + "/" pages: dict[str, str] = {} for link in soup.find_all("a", href=True): url = urljoin(overview_url, str(link["href"])) parsed = urlparse(url) if not parsed.path.startswith(base_path): continue if not parsed.path.endswith(".html"): continue name = clean(link.get_text(" ", strip=True)).lstrip("- ").strip() if not name: name = Path(parsed.path).stem.replace("-", " ").title() pages[url] = name return sorted(pages.items(), key=lambda item: item[1].casefold()) def juror_name_from_page(soup: BeautifulSoup, fallback_name: str) -> str: h1 = soup.find("h1") if h1: name = clean(h1.get_text(" ", strip=True)) if name: return name og_title = soup.find("meta", attrs={"property": "og:title"}) if og_title and og_title.get("content"): name = clean(str(og_title["content"])) if name: return name return fallback_name def extract_votes_from_table( table, juror: str, source_url: str, ) -> list[Vote] | None: rows_by_rank: dict[int, Vote] = {} for row in table.find_all("tr"): cells = row.find_all(["td", "th"]) if len(cells) < 3: continue rank_text = clean(cells[0].get_text(" ", strip=True)) rank_match = re.fullmatch(r"(10|[1-9])(?:[.)])?", rank_text) if not rank_match: continue rank = int(rank_match.group(1)) artist = clean(cells[1].get_text(" ", strip=True)) title = clean(cells[2].get_text(" ", strip=True)) if not artist or not title: continue rows_by_rank[rank] = Vote( juror=juror, rank=rank, points=POINTS[rank], artist=artist, title=title, source_url=source_url, ) if set(rows_by_rank) != set(range(1, 11)): return None return [rows_by_rank[rank] for rank in range(1, 11)] def extract_votes_from_historical_text( soup, juror: str, source_url: str, ) -> list[Vote] | None: rows_by_rank: dict[int, Vote] = {} for element in soup.find_all(["tr", "li", "p"]): text = clean(element.get_text(" ", strip=True)) match = re.fullmatch(r"(10|[1-9])\.\s*(.+)", text) if not match: continue rank = int(match.group(1)) entry = match.group(2) artist, separator, title = entry.partition(":") if not separator: artist, separator, title = entry.partition(" - ") if not separator: continue artist = clean(artist) title = clean(title) if not artist or not title: continue rows_by_rank[rank] = Vote( juror=juror, rank=rank, points=POINTS[rank], artist=artist, title=title, source_url=source_url, ) if set(rows_by_rank) != set(range(1, 11)): return None return [rows_by_rank[rank] for rank in range(1, 11)] def parse_jury_page( html: str, fallback_name: str, source_url: str, ) -> list[Vote]: soup = BeautifulSoup(html, "html.parser") juror = juror_name_from_page(soup, fallback_name) for table in soup.find_all("table"): votes = extract_votes_from_table(table, juror, source_url) if votes is not None: return votes votes = extract_votes_from_historical_text(soup, juror, source_url) if votes is not None: return votes raise RuntimeError("Keine vollständige Top-10-Tabelle gefunden") def aggregate(votes: list[Vote]) -> pd.DataFrame: groups: dict[str, list[Vote]] = defaultdict(list) for vote in votes: groups[normalize_title(vote.title)].append(vote) rows: list[dict[str, object]] = [] for group in groups.values(): title_counts = Counter(vote.title for vote in group) artist_counts = Counter(vote.artist for vote in group) display_title = sorted( title_counts, key=lambda title: ( -title_counts[title], -len(title), title.casefold(), ), )[0] rows.append( { "Titel": display_title, "Punkte": sum(vote.points for vote in group), "Nennungen": len(group), "Erste Plätze": sum(vote.rank == 1 for vote in group), "Beste Platzierung": min(vote.rank for vote in group), "Durchschnittsplatz": round( sum(vote.rank for vote in group) / len(group), 3, ), "Künstlerangaben": " | ".join( artist for artist, _ in artist_counts.most_common() ), "Titelvarianten": " | ".join( title for title, _ in title_counts.most_common() ), } ) ranking = pd.DataFrame(rows) if ranking.empty: return ranking ranking = ranking.sort_values( by=[ "Punkte", "Nennungen", "Erste Plätze", "Beste Platzierung", "Durchschnittsplatz", "Titel", ], ascending=[False, False, False, True, True, True], kind="stable", ).reset_index(drop=True) ranking.insert(0, "Rang", range(1, len(ranking) + 1)) return ranking def write_outputs( output_dir: Path, votes: list[Vote], ranking: pd.DataFrame, errors: list[dict[str, str]], ) -> None: output_dir.mkdir(parents=True, exist_ok=True) votes_df = pd.DataFrame(asdict(vote) for vote in votes) errors_df = pd.DataFrame(errors) ranking.head(100).to_csv( output_dir / "top100.csv", index=False, encoding="utf-8-sig", ) ranking.to_csv( output_dir / "gesamtwertung.csv", index=False, encoding="utf-8-sig", ) votes_df.to_csv( output_dir / "einzelwertungen.csv", index=False, encoding="utf-8-sig", ) errors_df.to_csv( output_dir / "fehler.csv", index=False, encoding="utf-8-sig", ) with pd.ExcelWriter( output_dir / "radioeins_top100.xlsx", engine="openpyxl", ) as writer: ranking.head(100).to_excel(writer, sheet_name="Top 100", index=False) ranking.to_excel(writer, sheet_name="Gesamtwertung", index=False) votes_df.to_excel(writer, sheet_name="Einzelwertungen", index=False) errors_df.to_excel(writer, sheet_name="Fehler", index=False) for sheet in writer.book.worksheets: sheet.freeze_panes = "A2" sheet.auto_filter.ref = sheet.dimensions for column in sheet.columns: width = max(len(str(cell.value or "")) for cell in column) + 2 sheet.column_dimensions[column[0].column_letter].width = min( max(width, 10), 60, ) def main() -> int: parser = argparse.ArgumentParser( description="radioeins-Jurylisten zu einer Top 100 zusammenfassen" ) parser.add_argument("url", nargs="?", default=DEFAULT_URL) parser.add_argument("--limit", type=int, default=None) parser.add_argument("--delay", type=float, default=0.2) parser.add_argument("--output", type=Path, default=None) args = parser.parse_args() output_dir = args.output or derive_output_dir(args.url) session = requests.Session() session.headers.update(HEADERS) pages = discover_jury_pages(session, args.url) if args.limit is not None: pages = pages[: args.limit] print(f"{len(pages)} Juryseiten gefunden.\n") all_votes: list[Vote] = [] errors: list[dict[str, str]] = [] for index, (url, name) in enumerate(pages, start=1): print(f"[{index:>3}/{len(pages)}] {name}") try: all_votes.extend( parse_jury_page( html=get_html(session, url), fallback_name=name, source_url=url, ) ) except Exception as exc: errors.append( { "juror": name, "url": url, "error": str(exc), } ) if args.delay > 0: time.sleep(args.delay) ranking = aggregate(all_votes) write_outputs(output_dir, all_votes, ranking, errors) print("\nFertig:") print(f" gefundene Seiten: {len(pages)}") print(f" gültige Jurylisten: {len(all_votes) // 10}") print(f" Wertungen: {len(all_votes)}") print(f" verschiedene Titel: {len(ranking)}") print(f" Fehler: {len(errors)}") print(f" Ausgabe: {output_dir.resolve()}") return 1 if errors else 0 if __name__ == "__main__": sys.exit(main())