Add working radioeins Top100 scraper

This commit is contained in:
2026-08-09 09:28:47 +02:00
parent 1b5539a80d
commit eb915333f0
31 changed files with 9829 additions and 0 deletions
+275
View File
@@ -0,0 +1,275 @@
# -*- coding: utf-8 -*-
from __future__ import annotations
import argparse
import re
import sys
import time
import unicodedata
from collections import Counter, defaultdict
from dataclasses import asdict, dataclass
from pathlib import Path
from urllib.parse import urljoin, urlparse
import pandas as pd
import requests
from bs4 import BeautifulSoup
DEFAULT_URL = "https://www.radioeins.de/musik/top_100/2026/deutscher-hip-hop/"
HEADERS = {
"User-Agent": (
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) "
"AppleWebKit/537.36 Chrome/142 Safari/537.36"
)
}
@dataclass(frozen=True)
class Vote:
juror: str
rank: int
points: int
artist: str
title: str
source_url: str
def clean(value: str) -> str:
return re.sub(r"\s+", " ", value.replace("\xa0", " ")).strip()
def normalize_title(value: str) -> str:
"""Nur sichere typografische Unterschiede vereinheitlichen."""
value = unicodedata.normalize("NFKC", clean(value)).casefold()
value = value.replace("’", "'").replace("`", "'").replace("´", "'")
value = re.sub(r"[‐‑‒–—―_-]+", " ", value)
return re.sub(r"\s+", " ", value).strip()
def get_html(session: requests.Session, url: str, retries: int = 3) -> str:
last_error: Exception | None = None
for attempt in range(1, retries + 1):
try:
response = session.get(url, timeout=30)
response.raise_for_status()
response.encoding = response.apparent_encoding or "utf-8"
return response.text
except requests.RequestException as exc:
last_error = exc
if attempt < retries:
time.sleep(attempt)
raise RuntimeError(f"Seite konnte nicht geladen werden: {url}") from last_error
def discover_jury_pages(
session: requests.Session,
overview_url: str,
) -> list[tuple[str, str]]:
soup = BeautifulSoup(get_html(session, overview_url), "html.parser")
base_path = urlparse(overview_url).path.rstrip("/") + "/"
pages: dict[str, str] = {}
for link in soup.find_all("a", href=True):
url = urljoin(overview_url, str(link["href"]))
parsed = urlparse(url)
if not parsed.path.startswith(base_path):
continue
if not parsed.path.endswith(".html"):
continue
name = clean(link.get_text(" ", strip=True)).lstrip("- ").strip()
if not name:
name = Path(parsed.path).stem.replace("-", " ").title()
pages[url] = name
return sorted(pages.items(), key=lambda item: item[1].casefold())
def parse_jury_page(html: str, fallback_name: str, source_url: str) -> list[Vote]:
soup = BeautifulSoup(html, "html.parser")
juror = fallback_name
page_heading = soup.find("h1")
if page_heading:
heading_text = clean(page_heading.get_text(" ", strip=True))
if " - " in heading_text:
juror = clean(heading_text.rsplit(" - ", 1)[-1]) or fallback_name
heading = None
for candidate in soup.find_all(["h2", "h3", "h4", "h5"]):
if clean(candidate.get_text(" ", strip=True)).casefold() == "meine top 10":
heading = candidate
break
if heading is None:
raise RuntimeError("Überschrift 'Meine Top 10' nicht gefunden")
table = heading.find_next("table")
if table is None:
raise RuntimeError("Tabelle nach 'Meine Top 10' nicht gefunden")
votes: list[Vote] = []
for row in table.find_all("tr"):
cells = row.find_all(["td", "th"])
if len(cells) < 3:
continue
rank_text = clean(cells[0].get_text(" ", strip=True))
if not rank_text.isdigit():
continue
rank = int(rank_text)
if not 1 <= rank <= 10:
continue
artist = clean(cells[1].get_text(" ", strip=True))
title = clean(cells[2].get_text(" ", strip=True))
if not artist or not title:
raise RuntimeError(f"Leerer Künstler oder Titel auf Platz {rank}")
votes.append(
Vote(
juror=juror,
rank=rank,
points=11 - rank,
artist=artist,
title=title,
source_url=source_url,
)
)
votes.sort(key=lambda vote: vote.rank)
found_ranks = [vote.rank for vote in votes]
if found_ranks != list(range(1, 11)):
raise RuntimeError(
f"Erwartet wurden die Plätze 1 bis 10, gefunden wurden: {found_ranks}"
)
return votes
def aggregate(votes: list[Vote]) -> pd.DataFrame:
groups: dict[str, list[Vote]] = defaultdict(list)
for vote in votes:
groups[normalize_title(vote.title)].append(vote)
rows: list[dict[str, object]] = []
for group in groups.values():
title_counts = Counter(vote.title for vote in group)
artist_counts = Counter(vote.artist for vote in group)
display_title = sorted(
title_counts,
key=lambda title: (-title_counts[title], -len(title), title.casefold()),
)[0]
rows.append(
{
"Titel": display_title,
"Punkte": sum(vote.points for vote in group),
"Nennungen": len(group),
"Erste Plätze": sum(vote.rank == 1 for vote in group),
"Beste Platzierung": min(vote.rank for vote in group),
"Durchschnittsplatz": round(
sum(vote.rank for vote in group) / len(group), 3
),
"Künstlerangaben": " | ".join(
artist for artist, _ in artist_counts.most_common()
),
"Titelvarianten": " | ".join(
title for title, _ in title_counts.most_common()
),
}
)
ranking = pd.DataFrame(rows)
if ranking.empty:
return ranking
ranking = ranking.sort_values(
by=[
"Punkte",
"Nennungen",
"Erste Plätze",
"Beste Platzierung",
"Durchschnittsplatz",
"Titel",
],
ascending=[False, False, False, True, True, True],
kind="stable",
).reset_index(drop=True)
ranking.insert(0, "Rang", range(1, len(ranking) + 1))
return ranking
def write_outputs(
output_dir: Path,
votes: list[Vote],
ranking: pd.DataFrame,
errors: list[dict[str, str]],
) -> None:
output_dir.mkdir(parents=True, exist_ok=True)
votes_df = pd.DataFrame([asdict(vote) for vote in votes])
errors_df = pd.DataFrame(errors, columns=["juror", "url", "error"])
ranking.head(100).to_csv(output_dir / "top100.csv", index=False, encoding="utf-8-sig")
ranking.to_csv(output_dir / "gesamtwertung.csv", index=False, encoding="utf-8-sig")
votes_df.to_csv(output_dir / "einzelwertungen.csv", index=False, encoding="utf-8-sig")
errors_df.to_csv(output_dir / "fehler.csv", index=False, encoding="utf-8-sig")
with pd.ExcelWriter(output_dir / "radioeins_top100.xlsx", engine="openpyxl") as writer:
ranking.head(100).to_excel(writer, sheet_name="Top 100", index=False)
ranking.to_excel(writer, sheet_name="Gesamtwertung", index=False)
votes_df.to_excel(writer, sheet_name="Einzelwertungen", index=False)
errors_df.to_excel(writer, sheet_name="Fehler", index=False)
for sheet in writer.book.worksheets:
sheet.freeze_panes = "A2"
sheet.auto_filter.ref = sheet.dimensions
for column in sheet.columns:
width = max(len(str(cell.value or "")) for cell in column) + 2
sheet.column_dimensions[column[0].column_letter].width = min(max(width, 10), 60)
def main() -> int:
parser = argparse.ArgumentParser(
description="radioeins-Jurylisten zu einer Top 100 zusammenfassen"
)
parser.add_argument("url", nargs="?", default=DEFAULT_URL)
parser.add_argument("--limit", type=int, default=None)
parser.add_argument("--delay", type=float, default=0.2)
parser.add_argument("--output", type=Path, default=Path("output"))
args = parser.parse_args()
session = requests.Session()
session.headers.update(HEADERS)
pages = discover_jury_pages(session, args.url)
if args.limit is not None:
pages = pages[: args.limit]
print(f"{len(pages)} Juryseiten gefunden.\n")
all_votes: list[Vote] = []
errors: list[dict[str, str]] = []
for index, (url, name) in enumerate(pages, start=1):
print(f"[{index:>3}/{len(pages)}] {name}")
try:
all_votes.extend(parse_jury_page(get_html(session, url), name, url))
except Exception as exc:
errors.append({"juror": name, "url": url, "error": str(exc)})
if args.delay > 0:
time.sleep(args.delay)
ranking = aggregate(all_votes)
write_outputs(args.output, all_votes, ranking, errors)
print("\nFertig:")
print(f" Juryseiten: {len(pages)}")
print(f" Wertungen: {len(all_votes)}")
print(f" verschiedene Titel: {len(ranking)}")
print(f" Fehler: {len(errors)}")
print(f" Ausgabe: {args.output.resolve()}")
return 1 if errors else 0
if __name__ == "__main__":
sys.exit(main())