Add working radioeins Top100 scraper
This commit is contained in:
@@ -0,0 +1,275 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import re
|
||||
import sys
|
||||
import time
|
||||
import unicodedata
|
||||
from collections import Counter, defaultdict
|
||||
from dataclasses import asdict, dataclass
|
||||
from pathlib import Path
|
||||
from urllib.parse import urljoin, urlparse
|
||||
|
||||
import pandas as pd
|
||||
import requests
|
||||
from bs4 import BeautifulSoup
|
||||
|
||||
DEFAULT_URL = "https://www.radioeins.de/musik/top_100/2026/deutscher-hip-hop/"
|
||||
HEADERS = {
|
||||
"User-Agent": (
|
||||
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) "
|
||||
"AppleWebKit/537.36 Chrome/142 Safari/537.36"
|
||||
)
|
||||
}
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class Vote:
|
||||
juror: str
|
||||
rank: int
|
||||
points: int
|
||||
artist: str
|
||||
title: str
|
||||
source_url: str
|
||||
|
||||
|
||||
def clean(value: str) -> str:
|
||||
return re.sub(r"\s+", " ", value.replace("\xa0", " ")).strip()
|
||||
|
||||
|
||||
def normalize_title(value: str) -> str:
|
||||
"""Nur sichere typografische Unterschiede vereinheitlichen."""
|
||||
value = unicodedata.normalize("NFKC", clean(value)).casefold()
|
||||
value = value.replace("’", "'").replace("`", "'").replace("´", "'")
|
||||
value = re.sub(r"[‐‑‒–—―_-]+", " ", value)
|
||||
return re.sub(r"\s+", " ", value).strip()
|
||||
|
||||
|
||||
def get_html(session: requests.Session, url: str, retries: int = 3) -> str:
|
||||
last_error: Exception | None = None
|
||||
for attempt in range(1, retries + 1):
|
||||
try:
|
||||
response = session.get(url, timeout=30)
|
||||
response.raise_for_status()
|
||||
response.encoding = response.apparent_encoding or "utf-8"
|
||||
return response.text
|
||||
except requests.RequestException as exc:
|
||||
last_error = exc
|
||||
if attempt < retries:
|
||||
time.sleep(attempt)
|
||||
raise RuntimeError(f"Seite konnte nicht geladen werden: {url}") from last_error
|
||||
|
||||
|
||||
def discover_jury_pages(
|
||||
session: requests.Session,
|
||||
overview_url: str,
|
||||
) -> list[tuple[str, str]]:
|
||||
soup = BeautifulSoup(get_html(session, overview_url), "html.parser")
|
||||
base_path = urlparse(overview_url).path.rstrip("/") + "/"
|
||||
pages: dict[str, str] = {}
|
||||
|
||||
for link in soup.find_all("a", href=True):
|
||||
url = urljoin(overview_url, str(link["href"]))
|
||||
parsed = urlparse(url)
|
||||
if not parsed.path.startswith(base_path):
|
||||
continue
|
||||
if not parsed.path.endswith(".html"):
|
||||
continue
|
||||
|
||||
name = clean(link.get_text(" ", strip=True)).lstrip("- ").strip()
|
||||
if not name:
|
||||
name = Path(parsed.path).stem.replace("-", " ").title()
|
||||
pages[url] = name
|
||||
|
||||
return sorted(pages.items(), key=lambda item: item[1].casefold())
|
||||
|
||||
|
||||
def parse_jury_page(html: str, fallback_name: str, source_url: str) -> list[Vote]:
|
||||
soup = BeautifulSoup(html, "html.parser")
|
||||
|
||||
juror = fallback_name
|
||||
page_heading = soup.find("h1")
|
||||
if page_heading:
|
||||
heading_text = clean(page_heading.get_text(" ", strip=True))
|
||||
if " - " in heading_text:
|
||||
juror = clean(heading_text.rsplit(" - ", 1)[-1]) or fallback_name
|
||||
|
||||
heading = None
|
||||
for candidate in soup.find_all(["h2", "h3", "h4", "h5"]):
|
||||
if clean(candidate.get_text(" ", strip=True)).casefold() == "meine top 10":
|
||||
heading = candidate
|
||||
break
|
||||
if heading is None:
|
||||
raise RuntimeError("Überschrift 'Meine Top 10' nicht gefunden")
|
||||
|
||||
table = heading.find_next("table")
|
||||
if table is None:
|
||||
raise RuntimeError("Tabelle nach 'Meine Top 10' nicht gefunden")
|
||||
|
||||
votes: list[Vote] = []
|
||||
for row in table.find_all("tr"):
|
||||
cells = row.find_all(["td", "th"])
|
||||
if len(cells) < 3:
|
||||
continue
|
||||
|
||||
rank_text = clean(cells[0].get_text(" ", strip=True))
|
||||
if not rank_text.isdigit():
|
||||
continue
|
||||
|
||||
rank = int(rank_text)
|
||||
if not 1 <= rank <= 10:
|
||||
continue
|
||||
|
||||
artist = clean(cells[1].get_text(" ", strip=True))
|
||||
title = clean(cells[2].get_text(" ", strip=True))
|
||||
if not artist or not title:
|
||||
raise RuntimeError(f"Leerer Künstler oder Titel auf Platz {rank}")
|
||||
|
||||
votes.append(
|
||||
Vote(
|
||||
juror=juror,
|
||||
rank=rank,
|
||||
points=11 - rank,
|
||||
artist=artist,
|
||||
title=title,
|
||||
source_url=source_url,
|
||||
)
|
||||
)
|
||||
|
||||
votes.sort(key=lambda vote: vote.rank)
|
||||
found_ranks = [vote.rank for vote in votes]
|
||||
if found_ranks != list(range(1, 11)):
|
||||
raise RuntimeError(
|
||||
f"Erwartet wurden die Plätze 1 bis 10, gefunden wurden: {found_ranks}"
|
||||
)
|
||||
return votes
|
||||
|
||||
|
||||
def aggregate(votes: list[Vote]) -> pd.DataFrame:
|
||||
groups: dict[str, list[Vote]] = defaultdict(list)
|
||||
for vote in votes:
|
||||
groups[normalize_title(vote.title)].append(vote)
|
||||
|
||||
rows: list[dict[str, object]] = []
|
||||
for group in groups.values():
|
||||
title_counts = Counter(vote.title for vote in group)
|
||||
artist_counts = Counter(vote.artist for vote in group)
|
||||
display_title = sorted(
|
||||
title_counts,
|
||||
key=lambda title: (-title_counts[title], -len(title), title.casefold()),
|
||||
)[0]
|
||||
|
||||
rows.append(
|
||||
{
|
||||
"Titel": display_title,
|
||||
"Punkte": sum(vote.points for vote in group),
|
||||
"Nennungen": len(group),
|
||||
"Erste Plätze": sum(vote.rank == 1 for vote in group),
|
||||
"Beste Platzierung": min(vote.rank for vote in group),
|
||||
"Durchschnittsplatz": round(
|
||||
sum(vote.rank for vote in group) / len(group), 3
|
||||
),
|
||||
"Künstlerangaben": " | ".join(
|
||||
artist for artist, _ in artist_counts.most_common()
|
||||
),
|
||||
"Titelvarianten": " | ".join(
|
||||
title for title, _ in title_counts.most_common()
|
||||
),
|
||||
}
|
||||
)
|
||||
|
||||
ranking = pd.DataFrame(rows)
|
||||
if ranking.empty:
|
||||
return ranking
|
||||
|
||||
ranking = ranking.sort_values(
|
||||
by=[
|
||||
"Punkte",
|
||||
"Nennungen",
|
||||
"Erste Plätze",
|
||||
"Beste Platzierung",
|
||||
"Durchschnittsplatz",
|
||||
"Titel",
|
||||
],
|
||||
ascending=[False, False, False, True, True, True],
|
||||
kind="stable",
|
||||
).reset_index(drop=True)
|
||||
ranking.insert(0, "Rang", range(1, len(ranking) + 1))
|
||||
return ranking
|
||||
|
||||
|
||||
def write_outputs(
|
||||
output_dir: Path,
|
||||
votes: list[Vote],
|
||||
ranking: pd.DataFrame,
|
||||
errors: list[dict[str, str]],
|
||||
) -> None:
|
||||
output_dir.mkdir(parents=True, exist_ok=True)
|
||||
votes_df = pd.DataFrame([asdict(vote) for vote in votes])
|
||||
errors_df = pd.DataFrame(errors, columns=["juror", "url", "error"])
|
||||
|
||||
ranking.head(100).to_csv(output_dir / "top100.csv", index=False, encoding="utf-8-sig")
|
||||
ranking.to_csv(output_dir / "gesamtwertung.csv", index=False, encoding="utf-8-sig")
|
||||
votes_df.to_csv(output_dir / "einzelwertungen.csv", index=False, encoding="utf-8-sig")
|
||||
errors_df.to_csv(output_dir / "fehler.csv", index=False, encoding="utf-8-sig")
|
||||
|
||||
with pd.ExcelWriter(output_dir / "radioeins_top100.xlsx", engine="openpyxl") as writer:
|
||||
ranking.head(100).to_excel(writer, sheet_name="Top 100", index=False)
|
||||
ranking.to_excel(writer, sheet_name="Gesamtwertung", index=False)
|
||||
votes_df.to_excel(writer, sheet_name="Einzelwertungen", index=False)
|
||||
errors_df.to_excel(writer, sheet_name="Fehler", index=False)
|
||||
|
||||
for sheet in writer.book.worksheets:
|
||||
sheet.freeze_panes = "A2"
|
||||
sheet.auto_filter.ref = sheet.dimensions
|
||||
for column in sheet.columns:
|
||||
width = max(len(str(cell.value or "")) for cell in column) + 2
|
||||
sheet.column_dimensions[column[0].column_letter].width = min(max(width, 10), 60)
|
||||
|
||||
|
||||
def main() -> int:
|
||||
parser = argparse.ArgumentParser(
|
||||
description="radioeins-Jurylisten zu einer Top 100 zusammenfassen"
|
||||
)
|
||||
parser.add_argument("url", nargs="?", default=DEFAULT_URL)
|
||||
parser.add_argument("--limit", type=int, default=None)
|
||||
parser.add_argument("--delay", type=float, default=0.2)
|
||||
parser.add_argument("--output", type=Path, default=Path("output"))
|
||||
args = parser.parse_args()
|
||||
|
||||
session = requests.Session()
|
||||
session.headers.update(HEADERS)
|
||||
|
||||
pages = discover_jury_pages(session, args.url)
|
||||
if args.limit is not None:
|
||||
pages = pages[: args.limit]
|
||||
|
||||
print(f"{len(pages)} Juryseiten gefunden.\n")
|
||||
|
||||
all_votes: list[Vote] = []
|
||||
errors: list[dict[str, str]] = []
|
||||
|
||||
for index, (url, name) in enumerate(pages, start=1):
|
||||
print(f"[{index:>3}/{len(pages)}] {name}")
|
||||
try:
|
||||
all_votes.extend(parse_jury_page(get_html(session, url), name, url))
|
||||
except Exception as exc:
|
||||
errors.append({"juror": name, "url": url, "error": str(exc)})
|
||||
if args.delay > 0:
|
||||
time.sleep(args.delay)
|
||||
|
||||
ranking = aggregate(all_votes)
|
||||
write_outputs(args.output, all_votes, ranking, errors)
|
||||
|
||||
print("\nFertig:")
|
||||
print(f" Juryseiten: {len(pages)}")
|
||||
print(f" Wertungen: {len(all_votes)}")
|
||||
print(f" verschiedene Titel: {len(ranking)}")
|
||||
print(f" Fehler: {len(errors)}")
|
||||
print(f" Ausgabe: {args.output.resolve()}")
|
||||
return 1 if errors else 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
Reference in New Issue
Block a user