# -*- coding: utf-8 -*- from __future__ import annotations import argparse import re import sys import time import unicodedata from collections import Counter, defaultdict from dataclasses import asdict, dataclass from pathlib import Path from urllib.parse import urljoin, urlparse import pandas as pd import requests from bs4 import BeautifulSoup DEFAULT_URL = "https://www.radioeins.de/musik/top_100/2026/deutscher-hip-hop/" HEADERS = { "User-Agent": ( "Mozilla/5.0 (Windows NT 10.0; Win64; x64) " "AppleWebKit/537.36 Chrome/142 Safari/537.36" ) } @dataclass(frozen=True) class Vote: juror: str rank: int points: int artist: str title: str source_url: str def clean(value: str) -> str: return re.sub(r"\s+", " ", value.replace("\xa0", " ")).strip() def normalize_title(value: str) -> str: """Nur sichere typografische Unterschiede vereinheitlichen.""" value = unicodedata.normalize("NFKC", clean(value)).casefold() value = value.replace("’", "'").replace("`", "'").replace("´", "'") value = re.sub(r"[‐‑‒–—―_-]+", " ", value) return re.sub(r"\s+", " ", value).strip() def get_html(session: requests.Session, url: str, retries: int = 3) -> str: last_error: Exception | None = None for attempt in range(1, retries + 1): try: response = session.get(url, timeout=30) response.raise_for_status() response.encoding = response.apparent_encoding or "utf-8" return response.text except requests.RequestException as exc: last_error = exc if attempt < retries: time.sleep(attempt) raise RuntimeError(f"Seite konnte nicht geladen werden: {url}") from last_error def discover_jury_pages( session: requests.Session, overview_url: str, ) -> list[tuple[str, str]]: soup = BeautifulSoup(get_html(session, overview_url), "html.parser") base_path = urlparse(overview_url).path.rstrip("/") + "/" pages: dict[str, str] = {} for link in soup.find_all("a", href=True): url = urljoin(overview_url, str(link["href"])) parsed = urlparse(url) if not parsed.path.startswith(base_path): continue if not parsed.path.endswith(".html"): continue name = clean(link.get_text(" ", strip=True)).lstrip("- ").strip() if not name: name = Path(parsed.path).stem.replace("-", " ").title() pages[url] = name return sorted(pages.items(), key=lambda item: item[1].casefold()) def parse_jury_page(html: str, fallback_name: str, source_url: str) -> list[Vote]: soup = BeautifulSoup(html, "html.parser") juror = fallback_name page_heading = soup.find("h1") if page_heading: heading_text = clean(page_heading.get_text(" ", strip=True)) if " - " in heading_text: juror = clean(heading_text.rsplit(" - ", 1)[-1]) or fallback_name heading = None for candidate in soup.find_all(["h2", "h3", "h4", "h5"]): if clean(candidate.get_text(" ", strip=True)).casefold() == "meine top 10": heading = candidate break if heading is None: raise RuntimeError("Überschrift 'Meine Top 10' nicht gefunden") table = heading.find_next("table") if table is None: raise RuntimeError("Tabelle nach 'Meine Top 10' nicht gefunden") votes: list[Vote] = [] for row in table.find_all("tr"): cells = row.find_all(["td", "th"]) if len(cells) < 3: continue rank_text = clean(cells[0].get_text(" ", strip=True)) if not rank_text.isdigit(): continue rank = int(rank_text) if not 1 <= rank <= 10: continue artist = clean(cells[1].get_text(" ", strip=True)) title = clean(cells[2].get_text(" ", strip=True)) if not artist or not title: raise RuntimeError(f"Leerer Künstler oder Titel auf Platz {rank}") votes.append( Vote( juror=juror, rank=rank, points=11 - rank, artist=artist, title=title, source_url=source_url, ) ) votes.sort(key=lambda vote: vote.rank) found_ranks = [vote.rank for vote in votes] if found_ranks != list(range(1, 11)): raise RuntimeError( f"Erwartet wurden die Plätze 1 bis 10, gefunden wurden: {found_ranks}" ) return votes def aggregate(votes: list[Vote]) -> pd.DataFrame: groups: dict[str, list[Vote]] = defaultdict(list) for vote in votes: groups[normalize_title(vote.title)].append(vote) rows: list[dict[str, object]] = [] for group in groups.values(): title_counts = Counter(vote.title for vote in group) artist_counts = Counter(vote.artist for vote in group) display_title = sorted( title_counts, key=lambda title: (-title_counts[title], -len(title), title.casefold()), )[0] rows.append( { "Titel": display_title, "Punkte": sum(vote.points for vote in group), "Nennungen": len(group), "Erste Plätze": sum(vote.rank == 1 for vote in group), "Beste Platzierung": min(vote.rank for vote in group), "Durchschnittsplatz": round( sum(vote.rank for vote in group) / len(group), 3 ), "Künstlerangaben": " | ".join( artist for artist, _ in artist_counts.most_common() ), "Titelvarianten": " | ".join( title for title, _ in title_counts.most_common() ), } ) ranking = pd.DataFrame(rows) if ranking.empty: return ranking ranking = ranking.sort_values( by=[ "Punkte", "Nennungen", "Erste Plätze", "Beste Platzierung", "Durchschnittsplatz", "Titel", ], ascending=[False, False, False, True, True, True], kind="stable", ).reset_index(drop=True) ranking.insert(0, "Rang", range(1, len(ranking) + 1)) return ranking def write_outputs( output_dir: Path, votes: list[Vote], ranking: pd.DataFrame, errors: list[dict[str, str]], ) -> None: output_dir.mkdir(parents=True, exist_ok=True) votes_df = pd.DataFrame([asdict(vote) for vote in votes]) errors_df = pd.DataFrame(errors, columns=["juror", "url", "error"]) ranking.head(100).to_csv(output_dir / "top100.csv", index=False, encoding="utf-8-sig") ranking.to_csv(output_dir / "gesamtwertung.csv", index=False, encoding="utf-8-sig") votes_df.to_csv(output_dir / "einzelwertungen.csv", index=False, encoding="utf-8-sig") errors_df.to_csv(output_dir / "fehler.csv", index=False, encoding="utf-8-sig") with pd.ExcelWriter(output_dir / "radioeins_top100.xlsx", engine="openpyxl") as writer: ranking.head(100).to_excel(writer, sheet_name="Top 100", index=False) ranking.to_excel(writer, sheet_name="Gesamtwertung", index=False) votes_df.to_excel(writer, sheet_name="Einzelwertungen", index=False) errors_df.to_excel(writer, sheet_name="Fehler", index=False) for sheet in writer.book.worksheets: sheet.freeze_panes = "A2" sheet.auto_filter.ref = sheet.dimensions for column in sheet.columns: width = max(len(str(cell.value or "")) for cell in column) + 2 sheet.column_dimensions[column[0].column_letter].width = min(max(width, 10), 60) def main() -> int: parser = argparse.ArgumentParser( description="radioeins-Jurylisten zu einer Top 100 zusammenfassen" ) parser.add_argument("url", nargs="?", default=DEFAULT_URL) parser.add_argument("--limit", type=int, default=None) parser.add_argument("--delay", type=float, default=0.2) parser.add_argument("--output", type=Path, default=Path("output")) args = parser.parse_args() session = requests.Session() session.headers.update(HEADERS) pages = discover_jury_pages(session, args.url) if args.limit is not None: pages = pages[: args.limit] print(f"{len(pages)} Juryseiten gefunden.\n") all_votes: list[Vote] = [] errors: list[dict[str, str]] = [] for index, (url, name) in enumerate(pages, start=1): print(f"[{index:>3}/{len(pages)}] {name}") try: all_votes.extend(parse_jury_page(get_html(session, url), name, url)) except Exception as exc: errors.append({"juror": name, "url": url, "error": str(exc)}) if args.delay > 0: time.sleep(args.delay) ranking = aggregate(all_votes) write_outputs(args.output, all_votes, ranking, errors) print("\nFertig:") print(f" Juryseiten: {len(pages)}") print(f" Wertungen: {len(all_votes)}") print(f" verschiedene Titel: {len(ranking)}") print(f" Fehler: {len(errors)}") print(f" Ausgabe: {args.output.resolve()}") return 1 if errors else 0 if __name__ == "__main__": sys.exit(main())