Add working radioeins Top100 scraper
This commit is contained in:
@@ -0,0 +1,346 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import re
|
||||
import sys
|
||||
import time
|
||||
import unicodedata
|
||||
from collections import Counter, defaultdict
|
||||
from dataclasses import asdict, dataclass
|
||||
from pathlib import Path
|
||||
from urllib.parse import urljoin, urlparse
|
||||
|
||||
import pandas as pd
|
||||
import requests
|
||||
from bs4 import BeautifulSoup
|
||||
|
||||
DEFAULT_URL = "https://www.radioeins.de/musik/top_100/2026/deutscher-hip-hop/"
|
||||
|
||||
HEADERS = {
|
||||
"User-Agent": (
|
||||
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) "
|
||||
"AppleWebKit/537.36 Chrome/142 Safari/537.36"
|
||||
)
|
||||
}
|
||||
|
||||
POINTS = {
|
||||
1: 12,
|
||||
2: 10,
|
||||
3: 8,
|
||||
4: 7,
|
||||
5: 6,
|
||||
6: 5,
|
||||
7: 4,
|
||||
8: 3,
|
||||
9: 2,
|
||||
10: 1,
|
||||
}
|
||||
|
||||
|
||||
@dataclass
|
||||
class Vote:
|
||||
juror: str
|
||||
rank: int
|
||||
points: int
|
||||
artist: str
|
||||
title: str
|
||||
source_url: str
|
||||
|
||||
|
||||
def clean(value: str) -> str:
|
||||
return re.sub(r"\s+", " ", value.replace("\xa0", " ")).strip()
|
||||
|
||||
|
||||
def normalize_title(value: str) -> str:
|
||||
value = unicodedata.normalize("NFKC", clean(value)).casefold()
|
||||
value = value.replace("’", "'").replace("`", "'").replace("´", "'")
|
||||
value = re.sub(r"[‐‑‒–—―_-]+", " ", value)
|
||||
value = re.sub(r"\s+", " ", value)
|
||||
return value.strip()
|
||||
|
||||
|
||||
def get_html(session: requests.Session, url: str, retries: int = 3) -> str:
|
||||
last_error: Exception | None = None
|
||||
for attempt in range(1, retries + 1):
|
||||
try:
|
||||
response = session.get(url, timeout=30)
|
||||
response.raise_for_status()
|
||||
response.encoding = response.apparent_encoding or "utf-8"
|
||||
return response.text
|
||||
except requests.RequestException as exc:
|
||||
last_error = exc
|
||||
if attempt < retries:
|
||||
time.sleep(attempt)
|
||||
raise RuntimeError(f"Seite konnte nicht geladen werden: {url}") from last_error
|
||||
|
||||
|
||||
def discover_jury_pages(
|
||||
session: requests.Session,
|
||||
overview_url: str,
|
||||
) -> list[tuple[str, str]]:
|
||||
soup = BeautifulSoup(get_html(session, overview_url), "html.parser")
|
||||
base_path = urlparse(overview_url).path.rstrip("/") + "/"
|
||||
pages: dict[str, str] = {}
|
||||
|
||||
for link in soup.find_all("a", href=True):
|
||||
url = urljoin(overview_url, str(link["href"]))
|
||||
parsed = urlparse(url)
|
||||
|
||||
if not parsed.path.startswith(base_path):
|
||||
continue
|
||||
if not parsed.path.endswith(".html"):
|
||||
continue
|
||||
|
||||
name = clean(link.get_text(" ", strip=True)).lstrip("- ").strip()
|
||||
if not name:
|
||||
name = Path(parsed.path).stem.replace("-", " ").title()
|
||||
|
||||
pages[url] = name
|
||||
|
||||
return sorted(pages.items(), key=lambda item: item[1].casefold())
|
||||
|
||||
|
||||
def juror_name_from_page(soup: BeautifulSoup, fallback_name: str) -> str:
|
||||
h1 = soup.find("h1")
|
||||
if h1:
|
||||
name = clean(h1.get_text(" ", strip=True))
|
||||
if name:
|
||||
return name
|
||||
|
||||
og_title = soup.find("meta", attrs={"property": "og:title"})
|
||||
if og_title and og_title.get("content"):
|
||||
name = clean(str(og_title["content"]))
|
||||
if name:
|
||||
return name
|
||||
|
||||
return fallback_name
|
||||
|
||||
|
||||
def extract_votes_from_table(
|
||||
table,
|
||||
juror: str,
|
||||
source_url: str,
|
||||
) -> list[Vote] | None:
|
||||
rows_by_rank: dict[int, Vote] = {}
|
||||
|
||||
for row in table.find_all("tr"):
|
||||
cells = row.find_all(["td", "th"])
|
||||
if len(cells) < 3:
|
||||
continue
|
||||
|
||||
rank_text = clean(cells[0].get_text(" ", strip=True))
|
||||
rank_match = re.fullmatch(r"(10|[1-9])(?:[.)])?", rank_text)
|
||||
if not rank_match:
|
||||
continue
|
||||
|
||||
rank = int(rank_match.group(1))
|
||||
artist = clean(cells[1].get_text(" ", strip=True))
|
||||
title = clean(cells[2].get_text(" ", strip=True))
|
||||
if not artist or not title:
|
||||
continue
|
||||
|
||||
rows_by_rank[rank] = Vote(
|
||||
juror=juror,
|
||||
rank=rank,
|
||||
points=POINTS[rank],
|
||||
artist=artist,
|
||||
title=title,
|
||||
source_url=source_url,
|
||||
)
|
||||
|
||||
if set(rows_by_rank) != set(range(1, 11)):
|
||||
return None
|
||||
|
||||
return [rows_by_rank[rank] for rank in range(1, 11)]
|
||||
|
||||
|
||||
def parse_jury_page(
|
||||
html: str,
|
||||
fallback_name: str,
|
||||
source_url: str,
|
||||
) -> list[Vote]:
|
||||
soup = BeautifulSoup(html, "html.parser")
|
||||
juror = juror_name_from_page(soup, fallback_name)
|
||||
|
||||
for table in soup.find_all("table"):
|
||||
votes = extract_votes_from_table(table, juror, source_url)
|
||||
if votes is not None:
|
||||
return votes
|
||||
|
||||
raise RuntimeError("Keine vollständige Top-10-Tabelle gefunden")
|
||||
|
||||
|
||||
def aggregate(votes: list[Vote]) -> pd.DataFrame:
|
||||
groups: dict[str, list[Vote]] = defaultdict(list)
|
||||
for vote in votes:
|
||||
groups[normalize_title(vote.title)].append(vote)
|
||||
|
||||
rows: list[dict[str, object]] = []
|
||||
|
||||
for group in groups.values():
|
||||
title_counts = Counter(vote.title for vote in group)
|
||||
artist_counts = Counter(vote.artist for vote in group)
|
||||
|
||||
display_title = sorted(
|
||||
title_counts,
|
||||
key=lambda title: (
|
||||
-title_counts[title],
|
||||
-len(title),
|
||||
title.casefold(),
|
||||
),
|
||||
)[0]
|
||||
|
||||
rows.append(
|
||||
{
|
||||
"Titel": display_title,
|
||||
"Punkte": sum(vote.points for vote in group),
|
||||
"Nennungen": len(group),
|
||||
"Erste Plätze": sum(vote.rank == 1 for vote in group),
|
||||
"Beste Platzierung": min(vote.rank for vote in group),
|
||||
"Durchschnittsplatz": round(
|
||||
sum(vote.rank for vote in group) / len(group),
|
||||
3,
|
||||
),
|
||||
"Künstlerangaben": " | ".join(
|
||||
artist for artist, _ in artist_counts.most_common()
|
||||
),
|
||||
"Titelvarianten": " | ".join(
|
||||
title for title, _ in title_counts.most_common()
|
||||
),
|
||||
}
|
||||
)
|
||||
|
||||
ranking = pd.DataFrame(rows)
|
||||
if ranking.empty:
|
||||
return ranking
|
||||
|
||||
ranking = ranking.sort_values(
|
||||
by=[
|
||||
"Punkte",
|
||||
"Nennungen",
|
||||
"Erste Plätze",
|
||||
"Beste Platzierung",
|
||||
"Durchschnittsplatz",
|
||||
"Titel",
|
||||
],
|
||||
ascending=[False, False, False, True, True, True],
|
||||
kind="stable",
|
||||
).reset_index(drop=True)
|
||||
|
||||
ranking.insert(0, "Rang", range(1, len(ranking) + 1))
|
||||
return ranking
|
||||
|
||||
|
||||
def write_outputs(
|
||||
output_dir: Path,
|
||||
votes: list[Vote],
|
||||
ranking: pd.DataFrame,
|
||||
errors: list[dict[str, str]],
|
||||
) -> None:
|
||||
output_dir.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
votes_df = pd.DataFrame(asdict(vote) for vote in votes)
|
||||
errors_df = pd.DataFrame(errors)
|
||||
|
||||
ranking.head(100).to_csv(
|
||||
output_dir / "top100.csv",
|
||||
index=False,
|
||||
encoding="utf-8-sig",
|
||||
)
|
||||
ranking.to_csv(
|
||||
output_dir / "gesamtwertung.csv",
|
||||
index=False,
|
||||
encoding="utf-8-sig",
|
||||
)
|
||||
votes_df.to_csv(
|
||||
output_dir / "einzelwertungen.csv",
|
||||
index=False,
|
||||
encoding="utf-8-sig",
|
||||
)
|
||||
errors_df.to_csv(
|
||||
output_dir / "fehler.csv",
|
||||
index=False,
|
||||
encoding="utf-8-sig",
|
||||
)
|
||||
|
||||
with pd.ExcelWriter(
|
||||
output_dir / "radioeins_top100.xlsx",
|
||||
engine="openpyxl",
|
||||
) as writer:
|
||||
ranking.head(100).to_excel(writer, sheet_name="Top 100", index=False)
|
||||
ranking.to_excel(writer, sheet_name="Gesamtwertung", index=False)
|
||||
votes_df.to_excel(writer, sheet_name="Einzelwertungen", index=False)
|
||||
errors_df.to_excel(writer, sheet_name="Fehler", index=False)
|
||||
|
||||
for sheet in writer.book.worksheets:
|
||||
sheet.freeze_panes = "A2"
|
||||
sheet.auto_filter.ref = sheet.dimensions
|
||||
for column in sheet.columns:
|
||||
width = max(len(str(cell.value or "")) for cell in column) + 2
|
||||
sheet.column_dimensions[column[0].column_letter].width = min(
|
||||
max(width, 10),
|
||||
60,
|
||||
)
|
||||
|
||||
|
||||
def main() -> int:
|
||||
parser = argparse.ArgumentParser(
|
||||
description="radioeins-Jurylisten zu einer Top 100 zusammenfassen"
|
||||
)
|
||||
parser.add_argument("url", nargs="?", default=DEFAULT_URL)
|
||||
parser.add_argument("--limit", type=int, default=None)
|
||||
parser.add_argument("--delay", type=float, default=0.2)
|
||||
parser.add_argument("--output", type=Path, default=Path("output"))
|
||||
args = parser.parse_args()
|
||||
|
||||
session = requests.Session()
|
||||
session.headers.update(HEADERS)
|
||||
|
||||
pages = discover_jury_pages(session, args.url)
|
||||
if args.limit is not None:
|
||||
pages = pages[: args.limit]
|
||||
|
||||
print(f"{len(pages)} Juryseiten gefunden.\n")
|
||||
|
||||
all_votes: list[Vote] = []
|
||||
errors: list[dict[str, str]] = []
|
||||
|
||||
for index, (url, name) in enumerate(pages, start=1):
|
||||
print(f"[{index:>3}/{len(pages)}] {name}")
|
||||
try:
|
||||
all_votes.extend(
|
||||
parse_jury_page(
|
||||
html=get_html(session, url),
|
||||
fallback_name=name,
|
||||
source_url=url,
|
||||
)
|
||||
)
|
||||
except Exception as exc:
|
||||
errors.append(
|
||||
{
|
||||
"juror": name,
|
||||
"url": url,
|
||||
"error": str(exc),
|
||||
}
|
||||
)
|
||||
|
||||
if args.delay > 0:
|
||||
time.sleep(args.delay)
|
||||
|
||||
ranking = aggregate(all_votes)
|
||||
write_outputs(args.output, all_votes, ranking, errors)
|
||||
|
||||
print("\nFertig:")
|
||||
print(f" gefundene Seiten: {len(pages)}")
|
||||
print(f" gültige Jurylisten: {len(all_votes) // 10}")
|
||||
print(f" Wertungen: {len(all_votes)}")
|
||||
print(f" verschiedene Titel: {len(ranking)}")
|
||||
print(f" Fehler: {len(errors)}")
|
||||
print(f" Ausgabe: {args.output.resolve()}")
|
||||
|
||||
return 1 if errors else 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
Reference in New Issue
Block a user