353 lines
9.8 KiB
Python
353 lines
9.8 KiB
Python
# -*- coding: utf-8 -*-
|
||
from __future__ import annotations
|
||
|
||
import argparse
|
||
import re
|
||
import sys
|
||
import time
|
||
import unicodedata
|
||
from collections import Counter, defaultdict
|
||
from dataclasses import asdict, dataclass
|
||
from pathlib import Path
|
||
from urllib.parse import urljoin, urlparse
|
||
|
||
import pandas as pd
|
||
import requests
|
||
from bs4 import BeautifulSoup
|
||
|
||
DEFAULT_URL = "https://www.radioeins.de/musik/top_100/2026/deutscher-hip-hop/"
|
||
|
||
HEADERS = {
|
||
"User-Agent": (
|
||
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) "
|
||
"AppleWebKit/537.36 Chrome/142 Safari/537.36"
|
||
)
|
||
}
|
||
|
||
POINTS = {
|
||
1: 12,
|
||
2: 10,
|
||
3: 8,
|
||
4: 7,
|
||
5: 6,
|
||
6: 5,
|
||
7: 4,
|
||
8: 3,
|
||
9: 2,
|
||
10: 1,
|
||
}
|
||
|
||
|
||
@dataclass
|
||
class Vote:
|
||
juror: str
|
||
rank: int
|
||
points: int
|
||
artist: str
|
||
title: str
|
||
source_url: str
|
||
|
||
|
||
def clean(value: str) -> str:
|
||
return re.sub(r"\s+", " ", value.replace("\xa0", " ")).strip()
|
||
|
||
|
||
def normalize_title(value: str) -> str:
|
||
value = unicodedata.normalize("NFKC", clean(value)).casefold()
|
||
value = value.replace("’", "'").replace("`", "'").replace("´", "'")
|
||
value = re.sub(r"[‐‑‒–—―_-]+", " ", value)
|
||
value = re.sub(r"\s+", " ", value)
|
||
return value.strip()
|
||
|
||
|
||
def derive_output_dir(overview_url: str) -> Path:
|
||
page_name = Path(urlparse(overview_url).path.rstrip("/")).name
|
||
return Path(f"output_{page_name}") if page_name else Path("output")
|
||
|
||
|
||
def get_html(session: requests.Session, url: str, retries: int = 3) -> str:
|
||
last_error: Exception | None = None
|
||
for attempt in range(1, retries + 1):
|
||
try:
|
||
response = session.get(url, timeout=30)
|
||
response.raise_for_status()
|
||
response.encoding = response.apparent_encoding or "utf-8"
|
||
return response.text
|
||
except requests.RequestException as exc:
|
||
last_error = exc
|
||
if attempt < retries:
|
||
time.sleep(attempt)
|
||
raise RuntimeError(f"Seite konnte nicht geladen werden: {url}") from last_error
|
||
|
||
|
||
def discover_jury_pages(
|
||
session: requests.Session,
|
||
overview_url: str,
|
||
) -> list[tuple[str, str]]:
|
||
soup = BeautifulSoup(get_html(session, overview_url), "html.parser")
|
||
base_path = urlparse(overview_url).path.rstrip("/") + "/"
|
||
pages: dict[str, str] = {}
|
||
|
||
for link in soup.find_all("a", href=True):
|
||
url = urljoin(overview_url, str(link["href"]))
|
||
parsed = urlparse(url)
|
||
|
||
if not parsed.path.startswith(base_path):
|
||
continue
|
||
if not parsed.path.endswith(".html"):
|
||
continue
|
||
|
||
name = clean(link.get_text(" ", strip=True)).lstrip("- ").strip()
|
||
if not name:
|
||
name = Path(parsed.path).stem.replace("-", " ").title()
|
||
|
||
pages[url] = name
|
||
|
||
return sorted(pages.items(), key=lambda item: item[1].casefold())
|
||
|
||
|
||
def juror_name_from_page(soup: BeautifulSoup, fallback_name: str) -> str:
|
||
h1 = soup.find("h1")
|
||
if h1:
|
||
name = clean(h1.get_text(" ", strip=True))
|
||
if name:
|
||
return name
|
||
|
||
og_title = soup.find("meta", attrs={"property": "og:title"})
|
||
if og_title and og_title.get("content"):
|
||
name = clean(str(og_title["content"]))
|
||
if name:
|
||
return name
|
||
|
||
return fallback_name
|
||
|
||
|
||
def extract_votes_from_table(
|
||
table,
|
||
juror: str,
|
||
source_url: str,
|
||
) -> list[Vote] | None:
|
||
rows_by_rank: dict[int, Vote] = {}
|
||
|
||
for row in table.find_all("tr"):
|
||
cells = row.find_all(["td", "th"])
|
||
if len(cells) < 3:
|
||
continue
|
||
|
||
rank_text = clean(cells[0].get_text(" ", strip=True))
|
||
rank_match = re.fullmatch(r"(10|[1-9])(?:[.)])?", rank_text)
|
||
if not rank_match:
|
||
continue
|
||
|
||
rank = int(rank_match.group(1))
|
||
artist = clean(cells[1].get_text(" ", strip=True))
|
||
title = clean(cells[2].get_text(" ", strip=True))
|
||
if not artist or not title:
|
||
continue
|
||
|
||
rows_by_rank[rank] = Vote(
|
||
juror=juror,
|
||
rank=rank,
|
||
points=POINTS[rank],
|
||
artist=artist,
|
||
title=title,
|
||
source_url=source_url,
|
||
)
|
||
|
||
if set(rows_by_rank) != set(range(1, 11)):
|
||
return None
|
||
|
||
return [rows_by_rank[rank] for rank in range(1, 11)]
|
||
|
||
|
||
def parse_jury_page(
|
||
html: str,
|
||
fallback_name: str,
|
||
source_url: str,
|
||
) -> list[Vote]:
|
||
soup = BeautifulSoup(html, "html.parser")
|
||
juror = juror_name_from_page(soup, fallback_name)
|
||
|
||
for table in soup.find_all("table"):
|
||
votes = extract_votes_from_table(table, juror, source_url)
|
||
if votes is not None:
|
||
return votes
|
||
|
||
raise RuntimeError("Keine vollständige Top-10-Tabelle gefunden")
|
||
|
||
|
||
def aggregate(votes: list[Vote]) -> pd.DataFrame:
|
||
groups: dict[str, list[Vote]] = defaultdict(list)
|
||
for vote in votes:
|
||
groups[normalize_title(vote.title)].append(vote)
|
||
|
||
rows: list[dict[str, object]] = []
|
||
|
||
for group in groups.values():
|
||
title_counts = Counter(vote.title for vote in group)
|
||
artist_counts = Counter(vote.artist for vote in group)
|
||
|
||
display_title = sorted(
|
||
title_counts,
|
||
key=lambda title: (
|
||
-title_counts[title],
|
||
-len(title),
|
||
title.casefold(),
|
||
),
|
||
)[0]
|
||
|
||
rows.append(
|
||
{
|
||
"Titel": display_title,
|
||
"Punkte": sum(vote.points for vote in group),
|
||
"Nennungen": len(group),
|
||
"Erste Plätze": sum(vote.rank == 1 for vote in group),
|
||
"Beste Platzierung": min(vote.rank for vote in group),
|
||
"Durchschnittsplatz": round(
|
||
sum(vote.rank for vote in group) / len(group),
|
||
3,
|
||
),
|
||
"Künstlerangaben": " | ".join(
|
||
artist for artist, _ in artist_counts.most_common()
|
||
),
|
||
"Titelvarianten": " | ".join(
|
||
title for title, _ in title_counts.most_common()
|
||
),
|
||
}
|
||
)
|
||
|
||
ranking = pd.DataFrame(rows)
|
||
if ranking.empty:
|
||
return ranking
|
||
|
||
ranking = ranking.sort_values(
|
||
by=[
|
||
"Punkte",
|
||
"Nennungen",
|
||
"Erste Plätze",
|
||
"Beste Platzierung",
|
||
"Durchschnittsplatz",
|
||
"Titel",
|
||
],
|
||
ascending=[False, False, False, True, True, True],
|
||
kind="stable",
|
||
).reset_index(drop=True)
|
||
|
||
ranking.insert(0, "Rang", range(1, len(ranking) + 1))
|
||
return ranking
|
||
|
||
|
||
def write_outputs(
|
||
output_dir: Path,
|
||
votes: list[Vote],
|
||
ranking: pd.DataFrame,
|
||
errors: list[dict[str, str]],
|
||
) -> None:
|
||
output_dir.mkdir(parents=True, exist_ok=True)
|
||
|
||
votes_df = pd.DataFrame(asdict(vote) for vote in votes)
|
||
errors_df = pd.DataFrame(errors)
|
||
|
||
ranking.head(100).to_csv(
|
||
output_dir / "top100.csv",
|
||
index=False,
|
||
encoding="utf-8-sig",
|
||
)
|
||
ranking.to_csv(
|
||
output_dir / "gesamtwertung.csv",
|
||
index=False,
|
||
encoding="utf-8-sig",
|
||
)
|
||
votes_df.to_csv(
|
||
output_dir / "einzelwertungen.csv",
|
||
index=False,
|
||
encoding="utf-8-sig",
|
||
)
|
||
errors_df.to_csv(
|
||
output_dir / "fehler.csv",
|
||
index=False,
|
||
encoding="utf-8-sig",
|
||
)
|
||
|
||
with pd.ExcelWriter(
|
||
output_dir / "radioeins_top100.xlsx",
|
||
engine="openpyxl",
|
||
) as writer:
|
||
ranking.head(100).to_excel(writer, sheet_name="Top 100", index=False)
|
||
ranking.to_excel(writer, sheet_name="Gesamtwertung", index=False)
|
||
votes_df.to_excel(writer, sheet_name="Einzelwertungen", index=False)
|
||
errors_df.to_excel(writer, sheet_name="Fehler", index=False)
|
||
|
||
for sheet in writer.book.worksheets:
|
||
sheet.freeze_panes = "A2"
|
||
sheet.auto_filter.ref = sheet.dimensions
|
||
for column in sheet.columns:
|
||
width = max(len(str(cell.value or "")) for cell in column) + 2
|
||
sheet.column_dimensions[column[0].column_letter].width = min(
|
||
max(width, 10),
|
||
60,
|
||
)
|
||
|
||
|
||
def main() -> int:
|
||
parser = argparse.ArgumentParser(
|
||
description="radioeins-Jurylisten zu einer Top 100 zusammenfassen"
|
||
)
|
||
parser.add_argument("url", nargs="?", default=DEFAULT_URL)
|
||
parser.add_argument("--limit", type=int, default=None)
|
||
parser.add_argument("--delay", type=float, default=0.2)
|
||
parser.add_argument("--output", type=Path, default=None)
|
||
args = parser.parse_args()
|
||
output_dir = args.output or derive_output_dir(args.url)
|
||
|
||
session = requests.Session()
|
||
session.headers.update(HEADERS)
|
||
|
||
pages = discover_jury_pages(session, args.url)
|
||
if args.limit is not None:
|
||
pages = pages[: args.limit]
|
||
|
||
print(f"{len(pages)} Juryseiten gefunden.\n")
|
||
|
||
all_votes: list[Vote] = []
|
||
errors: list[dict[str, str]] = []
|
||
|
||
for index, (url, name) in enumerate(pages, start=1):
|
||
print(f"[{index:>3}/{len(pages)}] {name}")
|
||
try:
|
||
all_votes.extend(
|
||
parse_jury_page(
|
||
html=get_html(session, url),
|
||
fallback_name=name,
|
||
source_url=url,
|
||
)
|
||
)
|
||
except Exception as exc:
|
||
errors.append(
|
||
{
|
||
"juror": name,
|
||
"url": url,
|
||
"error": str(exc),
|
||
}
|
||
)
|
||
|
||
if args.delay > 0:
|
||
time.sleep(args.delay)
|
||
|
||
ranking = aggregate(all_votes)
|
||
write_outputs(output_dir, all_votes, ranking, errors)
|
||
|
||
print("\nFertig:")
|
||
print(f" gefundene Seiten: {len(pages)}")
|
||
print(f" gültige Jurylisten: {len(all_votes) // 10}")
|
||
print(f" Wertungen: {len(all_votes)}")
|
||
print(f" verschiedene Titel: {len(ranking)}")
|
||
print(f" Fehler: {len(errors)}")
|
||
print(f" Ausgabe: {output_dir.resolve()}")
|
||
|
||
return 1 if errors else 0
|
||
|
||
|
||
if __name__ == "__main__":
|
||
sys.exit(main())
|