From 5db0e1816b8a6fef4fe007373a9e541c102d5de4 Mon Sep 17 00:00:00 2001 From: Martin Date: Sun, 9 Aug 2026 12:42:16 +0200 Subject: [PATCH] Handle incomplete historical jury lists --- .gitignore | 1 + README.md | 8 ++ radioeins_top100.py | 116 ++++++++++++++++++--- tests/fixtures/love-aditya-sharma.html | 14 +++ tests/fixtures/love-christiane-falk.html | 14 +++ tests/fixtures/love-dagobert.html | 14 +++ tests/fixtures/love-sven-erik-stephan.html | 14 +++ tests/test_radioeins_top100.py | 78 +++++++++++++- 8 files changed, 244 insertions(+), 15 deletions(-) create mode 100644 tests/fixtures/love-aditya-sharma.html create mode 100644 tests/fixtures/love-christiane-falk.html create mode 100644 tests/fixtures/love-dagobert.html create mode 100644 tests/fixtures/love-sven-erik-stephan.html diff --git a/.gitignore b/.gitignore index 3a3e5b8..20e3c86 100644 --- a/.gitignore +++ b/.gitignore @@ -11,6 +11,7 @@ top100.csv gesamtwertung.csv einzelwertungen.csv fehler.csv +incomplete_jury_lists.csv # macOS metadata .DS_Store diff --git a/README.md b/README.md index bdf0629..84bde18 100644 --- a/README.md +++ b/README.md @@ -47,6 +47,7 @@ Optionen: --limit ANZAHL nur die ersten ANZAHL Juryseiten verarbeiten --delay SEKUNDEN Pause zwischen Abrufen (Standard: 0.2) --output ORDNER abweichenden Ausgabeordner verwenden +--accept-incomplete unvollständige historische Listen automatisch akzeptieren ``` Beispiel für einen kurzen Testlauf: @@ -64,12 +65,19 @@ Beispiel für einen kurzen Testlauf: - `gesamtwertung.csv` - `einzelwertungen.csv` - `fehler.csv` +- `incomplete_jury_lists.csv` Der Ausgabeordner wird bei Bedarf automatisch erstellt. Bereits vorhandene Dateien gleichen Namens werden bei einem neuen Lauf ersetzt. Mit `--output` kann jederzeit ausdrücklich ein anderer Ordner gewählt werden; diese Angabe hat Vorrang vor dem aus der URL abgeleiteten Namen. +Unvollständige, aber strukturell gültige historische Jurylisten werden im +Terminal zur Bestätigung angeboten. Ohne interaktives Terminal werden sie +standardmäßig abgelehnt. `--accept-incomplete` akzeptiert sie automatisch; +fehlende Ränge werden dabei nicht ergänzt oder verschoben. Alle entsprechenden +Entscheidungen stehen in `incomplete_jury_lists.csv`. + ## Tests ```sh diff --git a/radioeins_top100.py b/radioeins_top100.py index c237e52..226eed0 100644 --- a/radioeins_top100.py +++ b/radioeins_top100.py @@ -48,6 +48,14 @@ class Vote: source_url: str +class IncompleteJuryListError(RuntimeError): + def __init__(self, votes: list[Vote]): + self.votes = votes + self.found_ranks = [vote.rank for vote in votes] + self.missing_ranks = sorted(set(range(1, 11)) - set(self.found_ranks)) + super().__init__("Keine vollständige Top-10-Tabelle gefunden") + + def clean(value: str) -> str: return re.sub(r"\s+", " ", value.replace("\xa0", " ")).strip() @@ -169,7 +177,7 @@ def extract_votes_from_historical_text( for element in soup.find_all(["tr", "li", "p"]): text = clean(element.get_text(" ", strip=True)) - match = re.fullmatch(r"(10|[1-9])\.\s*(.+)", text) + match = re.fullmatch(r"(10|[1-9])(?:\.\s*|\s+)(.+)", text) if not match: continue @@ -178,6 +186,8 @@ def extract_votes_from_historical_text( artist, separator, title = entry.partition(":") if not separator: artist, separator, title = entry.partition(" - ") + if not separator: + artist, separator, title = entry.partition(". ") if not separator: continue @@ -195,10 +205,10 @@ def extract_votes_from_historical_text( source_url=source_url, ) - if set(rows_by_rank) != set(range(1, 11)): + if not rows_by_rank: return None - return [rows_by_rank[rank] for rank in range(1, 11)] + return [rows_by_rank[rank] for rank in sorted(rows_by_rank)] def parse_jury_page( @@ -215,12 +225,37 @@ def parse_jury_page( return votes votes = extract_votes_from_historical_text(soup, juror, source_url) - if votes is not None: + if votes and len(votes) == 10: return votes + if votes: + raise IncompleteJuryListError(votes) raise RuntimeError("Keine vollständige Top-10-Tabelle gefunden") +def decide_incomplete_list( + error: IncompleteJuryListError, + *, + accept_incomplete: bool, + interactive: bool, + decision_helper=input, +) -> bool: + vote = error.votes[0] + print("\nUnvollständige Juryliste:") + print(f" Juror: {vote.juror}") + print(f" URL: {vote.source_url}") + print(f" Gefundene Ränge: {', '.join(map(str, error.found_ranks))}") + print(f" Fehlende Ränge: {', '.join(map(str, error.missing_ranks))}") + + if accept_incomplete: + return True + if not interactive: + return False + + answer = decision_helper("Count this incomplete jury list anyway? [y/N]: ") + return answer.strip().casefold() in {"y", "yes"} + + def aggregate(votes: list[Vote]) -> pd.DataFrame: groups: dict[str, list[Vote]] = defaultdict(list) for vote in votes: @@ -287,11 +322,16 @@ def write_outputs( votes: list[Vote], ranking: pd.DataFrame, errors: list[dict[str, str]], + incomplete_lists: list[dict[str, object]], ) -> None: output_dir.mkdir(parents=True, exist_ok=True) votes_df = pd.DataFrame(asdict(vote) for vote in votes) errors_df = pd.DataFrame(errors) + incomplete_df = pd.DataFrame( + incomplete_lists, + columns=["juror", "url", "found_ranks", "missing_ranks", "accepted"], + ) ranking.head(100).to_csv( output_dir / "top100.csv", @@ -313,6 +353,11 @@ def write_outputs( index=False, encoding="utf-8-sig", ) + incomplete_df.to_csv( + output_dir / "incomplete_jury_lists.csv", + index=False, + encoding="utf-8-sig", + ) with pd.ExcelWriter( output_dir / "radioeins_top100.xlsx", @@ -334,7 +379,7 @@ def write_outputs( ) -def main() -> int: +def build_argument_parser() -> argparse.ArgumentParser: parser = argparse.ArgumentParser( description="radioeins-Jurylisten zu einer Top 100 zusammenfassen" ) @@ -342,7 +387,16 @@ def main() -> int: parser.add_argument("--limit", type=int, default=None) parser.add_argument("--delay", type=float, default=0.2) parser.add_argument("--output", type=Path, default=None) - args = parser.parse_args() + parser.add_argument( + "--accept-incomplete", + action="store_true", + help="unvollständige historische Jurylisten ohne Nachfrage akzeptieren", + ) + return parser + + +def main() -> int: + args = build_argument_parser().parse_args() output_dir = args.output or derive_output_dir(args.url) session = requests.Session() @@ -356,17 +410,39 @@ def main() -> int: all_votes: list[Vote] = [] errors: list[dict[str, str]] = [] + incomplete_lists: list[dict[str, object]] = [] + valid_jury_lists = 0 for index, (url, name) in enumerate(pages, start=1): print(f"[{index:>3}/{len(pages)}] {name}") try: - all_votes.extend( - parse_jury_page( - html=get_html(session, url), - fallback_name=name, - source_url=url, - ) + votes = parse_jury_page( + html=get_html(session, url), + fallback_name=name, + source_url=url, ) + except IncompleteJuryListError as exc: + accepted = decide_incomplete_list( + exc, + accept_incomplete=args.accept_incomplete, + interactive=sys.stdin.isatty(), + ) + incomplete_lists.append( + { + "juror": exc.votes[0].juror, + "url": exc.votes[0].source_url, + "found_ranks": ",".join(map(str, exc.found_ranks)), + "missing_ranks": ",".join(map(str, exc.missing_ranks)), + "accepted": accepted, + } + ) + if accepted: + all_votes.extend(exc.votes) + valid_jury_lists += 1 + else: + errors.append( + {"juror": name, "url": url, "error": str(exc)} + ) except Exception as exc: errors.append( { @@ -375,19 +451,31 @@ def main() -> int: "error": str(exc), } ) + else: + all_votes.extend(votes) + valid_jury_lists += 1 if args.delay > 0: time.sleep(args.delay) ranking = aggregate(all_votes) - write_outputs(output_dir, all_votes, ranking, errors) + write_outputs(output_dir, all_votes, ranking, errors, incomplete_lists) print("\nFertig:") print(f" gefundene Seiten: {len(pages)}") - print(f" gültige Jurylisten: {len(all_votes) // 10}") + print(f" gültige Jurylisten: {valid_jury_lists}") print(f" Wertungen: {len(all_votes)}") print(f" verschiedene Titel: {len(ranking)}") print(f" Fehler: {len(errors)}") + print(f" unvollständig gefunden: {len(incomplete_lists)}") + print( + " unvollständig akzeptiert: " + f"{sum(bool(item['accepted']) for item in incomplete_lists)}" + ) + print( + " unvollständig abgelehnt: " + f"{sum(not bool(item['accepted']) for item in incomplete_lists)}" + ) print(f" Ausgabe: {output_dir.resolve()}") return 1 if errors else 0 diff --git a/tests/fixtures/love-aditya-sharma.html b/tests/fixtures/love-aditya-sharma.html new file mode 100644 index 0000000..5c7b71e --- /dev/null +++ b/tests/fixtures/love-aditya-sharma.html @@ -0,0 +1,14 @@ + + + + + + + + + + + + + +
1. The Sundays: Here's Where The Story Ends  
2. The Beatles: Something  
3. The Pogues & Kirsty MacColl: Fairytale of New York  
4. China Crisis: Wishful Thinking  
5  Nick Drake: Northern Sky  
6. Elton John: Your Song  
7. Sam Cooke: Cupid  
8. Carole King: You've Got A Friend  
9. Turbonegro: I Got Erection  
10. Dead Kennedys: Too Drunk To Fuck  
diff --git a/tests/fixtures/love-christiane-falk.html b/tests/fixtures/love-christiane-falk.html new file mode 100644 index 0000000..93ed684 --- /dev/null +++ b/tests/fixtures/love-christiane-falk.html @@ -0,0 +1,14 @@ + + + + + + + + + + + + + +
1. Pearl Jam: Black  
2. Element of Crime: Weißes Papier  
3. R.E.M.: Country Feedback  
4. Jeff Buckley. Hallelujah  
5. Gisbert zu Knyphausen: Dreh dich nicht um  
6. John Grant feat. Midlake: I Wanna Go To Marz  
7. James Blake: Limit To Your Love  
8. The National: I Need My Girl  
9. Blur: No Distance Left To Run  
10. Keane: Bend and Break  
diff --git a/tests/fixtures/love-dagobert.html b/tests/fixtures/love-dagobert.html new file mode 100644 index 0000000..b0b2db9 --- /dev/null +++ b/tests/fixtures/love-dagobert.html @@ -0,0 +1,14 @@ + + + + + + + + + + + + + +
1. Doris Day: The Way I Dreamed It  
2. Telly Savalas: If  
3. Lou Reed: Crazy Feeling  
4. Roy Rogers: Cleanin' My Rifle (And Dreamin' Of You)  
5. Carpenters: I Need To Be In Love  
6. Hank Williams: Cold Cold Heart  
7. Chris Isaak: Wicked Game  
8. Astrud Gilberto: Never My Love  
9. Tiny Tim: Earth Angel  
   
diff --git a/tests/fixtures/love-sven-erik-stephan.html b/tests/fixtures/love-sven-erik-stephan.html new file mode 100644 index 0000000..941df8f --- /dev/null +++ b/tests/fixtures/love-sven-erik-stephan.html @@ -0,0 +1,14 @@ + + + + + + + + + + + + + +
1. The Cure: Lovesong  
2. Joy Division: Love Will Tear Us Apart  
3. Roxy Music: More Than This  
4. Prince: Raspberry Beret  
5. Mazzy Star: Fade Into You  
6. Rufus & Chaka Khan: Ain’t Nobody  
7. Cyndi Lauper. Time After Time  
8. Arthur Russell: That’s Us/ Wild Cimbination  
9. The Pharcyde: Passin’ Me By  
10. Frankie Goes To Hollywood: The Power Of Love  
diff --git a/tests/test_radioeins_top100.py b/tests/test_radioeins_top100.py index baba79e..43bf15c 100644 --- a/tests/test_radioeins_top100.py +++ b/tests/test_radioeins_top100.py @@ -1,7 +1,16 @@ import unittest +from contextlib import redirect_stdout +from io import StringIO from pathlib import Path -from radioeins_top100 import POINTS, derive_output_dir, parse_jury_page +from radioeins_top100 import ( + POINTS, + IncompleteJuryListError, + build_argument_parser, + decide_incomplete_list, + derive_output_dir, + parse_jury_page, +) class RadioeinsTop100Tests(unittest.TestCase): @@ -67,6 +76,73 @@ class RadioeinsTop100Tests(unittest.TestCase): self.assertEqual(votes[7].artist, "Boxhmasters") self.assertEqual(votes[7].title, "Mogli") + def test_historical_rank_without_dot(self): + votes = self.parse_fixture("love-aditya-sharma.html") + + self.assertEqual((votes[4].rank, votes[4].artist, votes[4].title), + (5, "Nick Drake", "Northern Sky")) + + def test_historical_period_separator(self): + christiane = self.parse_fixture("love-christiane-falk.html") + sven = self.parse_fixture("love-sven-erik-stephan.html") + + self.assertEqual((christiane[3].artist, christiane[3].title), + ("Jeff Buckley", "Hallelujah")) + self.assertEqual((sven[6].artist, sven[6].title), + ("Cyndi Lauper", "Time After Time")) + self.assertEqual(christiane[2].artist, "R.E.M.") + + def test_incomplete_historical_list_is_rejected(self): + with self.assertRaises(IncompleteJuryListError) as caught: + self.parse_fixture("love-dagobert.html") + + output = StringIO() + with redirect_stdout(output): + accepted = decide_incomplete_list( + caught.exception, + accept_incomplete=False, + interactive=False, + decision_helper=lambda _: self.fail("must not prompt"), + ) + + self.assertFalse(accepted) + self.assertIn("Dagobert", output.getvalue()) + self.assertEqual(caught.exception.missing_ranks, [10]) + + def test_incomplete_historical_list_is_accepted_interactively(self): + with self.assertRaises(IncompleteJuryListError) as caught: + self.parse_fixture("love-dagobert.html") + + with redirect_stdout(StringIO()): + accepted = decide_incomplete_list( + caught.exception, + accept_incomplete=False, + interactive=True, + decision_helper=lambda prompt: "yes", + ) + + self.assertTrue(accepted) + self.assertEqual( + [(vote.rank, vote.points) for vote in caught.exception.votes], + [(1, 12), (2, 10), (3, 8), (4, 7), (5, 6), + (6, 5), (7, 4), (8, 3), (9, 2)], + ) + + def test_accept_incomplete_option(self): + args = build_argument_parser().parse_args(["--accept-incomplete"]) + self.assertTrue(args.accept_incomplete) + + with self.assertRaises(IncompleteJuryListError) as caught: + self.parse_fixture("love-dagobert.html") + with redirect_stdout(StringIO()): + accepted = decide_incomplete_list( + caught.exception, + accept_incomplete=args.accept_incomplete, + interactive=False, + decision_helper=lambda _: self.fail("must not prompt"), + ) + self.assertTrue(accepted) + if __name__ == "__main__": unittest.main()