Files
radioeins_charts/radioeins_top100.py
T

347 lines
9.6 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
# -*- coding: utf-8 -*-
from __future__ import annotations
import argparse
import re
import sys
import time
import unicodedata
from collections import Counter, defaultdict
from dataclasses import asdict, dataclass
from pathlib import Path
from urllib.parse import urljoin, urlparse
import pandas as pd
import requests
from bs4 import BeautifulSoup
DEFAULT_URL = "https://www.radioeins.de/musik/top_100/2026/deutscher-hip-hop/"
HEADERS = {
"User-Agent": (
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) "
"AppleWebKit/537.36 Chrome/142 Safari/537.36"
)
}
POINTS = {
1: 12,
2: 10,
3: 8,
4: 7,
5: 6,
6: 5,
7: 4,
8: 3,
9: 2,
10: 1,
}
@dataclass
class Vote:
juror: str
rank: int
points: int
artist: str
title: str
source_url: str
def clean(value: str) -> str:
return re.sub(r"\s+", " ", value.replace("\xa0", " ")).strip()
def normalize_title(value: str) -> str:
value = unicodedata.normalize("NFKC", clean(value)).casefold()
value = value.replace("’", "'").replace("`", "'").replace("´", "'")
value = re.sub(r"[‐‑‒–—―_-]+", " ", value)
value = re.sub(r"\s+", " ", value)
return value.strip()
def get_html(session: requests.Session, url: str, retries: int = 3) -> str:
last_error: Exception | None = None
for attempt in range(1, retries + 1):
try:
response = session.get(url, timeout=30)
response.raise_for_status()
response.encoding = response.apparent_encoding or "utf-8"
return response.text
except requests.RequestException as exc:
last_error = exc
if attempt < retries:
time.sleep(attempt)
raise RuntimeError(f"Seite konnte nicht geladen werden: {url}") from last_error
def discover_jury_pages(
session: requests.Session,
overview_url: str,
) -> list[tuple[str, str]]:
soup = BeautifulSoup(get_html(session, overview_url), "html.parser")
base_path = urlparse(overview_url).path.rstrip("/") + "/"
pages: dict[str, str] = {}
for link in soup.find_all("a", href=True):
url = urljoin(overview_url, str(link["href"]))
parsed = urlparse(url)
if not parsed.path.startswith(base_path):
continue
if not parsed.path.endswith(".html"):
continue
name = clean(link.get_text(" ", strip=True)).lstrip("- ").strip()
if not name:
name = Path(parsed.path).stem.replace("-", " ").title()
pages[url] = name
return sorted(pages.items(), key=lambda item: item[1].casefold())
def juror_name_from_page(soup: BeautifulSoup, fallback_name: str) -> str:
h1 = soup.find("h1")
if h1:
name = clean(h1.get_text(" ", strip=True))
if name:
return name
og_title = soup.find("meta", attrs={"property": "og:title"})
if og_title and og_title.get("content"):
name = clean(str(og_title["content"]))
if name:
return name
return fallback_name
def extract_votes_from_table(
table,
juror: str,
source_url: str,
) -> list[Vote] | None:
rows_by_rank: dict[int, Vote] = {}
for row in table.find_all("tr"):
cells = row.find_all(["td", "th"])
if len(cells) < 3:
continue
rank_text = clean(cells[0].get_text(" ", strip=True))
rank_match = re.fullmatch(r"(10|[1-9])(?:[.)])?", rank_text)
if not rank_match:
continue
rank = int(rank_match.group(1))
artist = clean(cells[1].get_text(" ", strip=True))
title = clean(cells[2].get_text(" ", strip=True))
if not artist or not title:
continue
rows_by_rank[rank] = Vote(
juror=juror,
rank=rank,
points=POINTS[rank],
artist=artist,
title=title,
source_url=source_url,
)
if set(rows_by_rank) != set(range(1, 11)):
return None
return [rows_by_rank[rank] for rank in range(1, 11)]
def parse_jury_page(
html: str,
fallback_name: str,
source_url: str,
) -> list[Vote]:
soup = BeautifulSoup(html, "html.parser")
juror = juror_name_from_page(soup, fallback_name)
for table in soup.find_all("table"):
votes = extract_votes_from_table(table, juror, source_url)
if votes is not None:
return votes
raise RuntimeError("Keine vollständige Top-10-Tabelle gefunden")
def aggregate(votes: list[Vote]) -> pd.DataFrame:
groups: dict[str, list[Vote]] = defaultdict(list)
for vote in votes:
groups[normalize_title(vote.title)].append(vote)
rows: list[dict[str, object]] = []
for group in groups.values():
title_counts = Counter(vote.title for vote in group)
artist_counts = Counter(vote.artist for vote in group)
display_title = sorted(
title_counts,
key=lambda title: (
-title_counts[title],
-len(title),
title.casefold(),
),
)[0]
rows.append(
{
"Titel": display_title,
"Punkte": sum(vote.points for vote in group),
"Nennungen": len(group),
"Erste Plätze": sum(vote.rank == 1 for vote in group),
"Beste Platzierung": min(vote.rank for vote in group),
"Durchschnittsplatz": round(
sum(vote.rank for vote in group) / len(group),
3,
),
"Künstlerangaben": " | ".join(
artist for artist, _ in artist_counts.most_common()
),
"Titelvarianten": " | ".join(
title for title, _ in title_counts.most_common()
),
}
)
ranking = pd.DataFrame(rows)
if ranking.empty:
return ranking
ranking = ranking.sort_values(
by=[
"Punkte",
"Nennungen",
"Erste Plätze",
"Beste Platzierung",
"Durchschnittsplatz",
"Titel",
],
ascending=[False, False, False, True, True, True],
kind="stable",
).reset_index(drop=True)
ranking.insert(0, "Rang", range(1, len(ranking) + 1))
return ranking
def write_outputs(
output_dir: Path,
votes: list[Vote],
ranking: pd.DataFrame,
errors: list[dict[str, str]],
) -> None:
output_dir.mkdir(parents=True, exist_ok=True)
votes_df = pd.DataFrame(asdict(vote) for vote in votes)
errors_df = pd.DataFrame(errors)
ranking.head(100).to_csv(
output_dir / "top100.csv",
index=False,
encoding="utf-8-sig",
)
ranking.to_csv(
output_dir / "gesamtwertung.csv",
index=False,
encoding="utf-8-sig",
)
votes_df.to_csv(
output_dir / "einzelwertungen.csv",
index=False,
encoding="utf-8-sig",
)
errors_df.to_csv(
output_dir / "fehler.csv",
index=False,
encoding="utf-8-sig",
)
with pd.ExcelWriter(
output_dir / "radioeins_top100.xlsx",
engine="openpyxl",
) as writer:
ranking.head(100).to_excel(writer, sheet_name="Top 100", index=False)
ranking.to_excel(writer, sheet_name="Gesamtwertung", index=False)
votes_df.to_excel(writer, sheet_name="Einzelwertungen", index=False)
errors_df.to_excel(writer, sheet_name="Fehler", index=False)
for sheet in writer.book.worksheets:
sheet.freeze_panes = "A2"
sheet.auto_filter.ref = sheet.dimensions
for column in sheet.columns:
width = max(len(str(cell.value or "")) for cell in column) + 2
sheet.column_dimensions[column[0].column_letter].width = min(
max(width, 10),
60,
)
def main() -> int:
parser = argparse.ArgumentParser(
description="radioeins-Jurylisten zu einer Top 100 zusammenfassen"
)
parser.add_argument("url", nargs="?", default=DEFAULT_URL)
parser.add_argument("--limit", type=int, default=None)
parser.add_argument("--delay", type=float, default=0.2)
parser.add_argument("--output", type=Path, default=Path("output"))
args = parser.parse_args()
session = requests.Session()
session.headers.update(HEADERS)
pages = discover_jury_pages(session, args.url)
if args.limit is not None:
pages = pages[: args.limit]
print(f"{len(pages)} Juryseiten gefunden.\n")
all_votes: list[Vote] = []
errors: list[dict[str, str]] = []
for index, (url, name) in enumerate(pages, start=1):
print(f"[{index:>3}/{len(pages)}] {name}")
try:
all_votes.extend(
parse_jury_page(
html=get_html(session, url),
fallback_name=name,
source_url=url,
)
)
except Exception as exc:
errors.append(
{
"juror": name,
"url": url,
"error": str(exc),
}
)
if args.delay > 0:
time.sleep(args.delay)
ranking = aggregate(all_votes)
write_outputs(args.output, all_votes, ranking, errors)
print("\nFertig:")
print(f" gefundene Seiten: {len(pages)}")
print(f" gültige Jurylisten: {len(all_votes) // 10}")
print(f" Wertungen: {len(all_votes)}")
print(f" verschiedene Titel: {len(ranking)}")
print(f" Fehler: {len(errors)}")
print(f" Ausgabe: {args.output.resolve()}")
return 1 if errors else 0
if __name__ == "__main__":
sys.exit(main())