Handle incomplete historical jury lists

This commit is contained in:
2026-08-09 12:42:16 +02:00
parent bbf0f1bf0a
commit 5db0e1816b
8 changed files with 244 additions and 15 deletions
+1
View File
@@ -11,6 +11,7 @@ top100.csv
gesamtwertung.csv
einzelwertungen.csv
fehler.csv
incomplete_jury_lists.csv
# macOS metadata
.DS_Store
+8
View File
@@ -47,6 +47,7 @@ Optionen:
--limit ANZAHL nur die ersten ANZAHL Juryseiten verarbeiten
--delay SEKUNDEN Pause zwischen Abrufen (Standard: 0.2)
--output ORDNER abweichenden Ausgabeordner verwenden
--accept-incomplete unvollständige historische Listen automatisch akzeptieren
```
Beispiel für einen kurzen Testlauf:
@@ -64,12 +65,19 @@ Beispiel für einen kurzen Testlauf:
- `gesamtwertung.csv`
- `einzelwertungen.csv`
- `fehler.csv`
- `incomplete_jury_lists.csv`
Der Ausgabeordner wird bei Bedarf automatisch erstellt. Bereits vorhandene
Dateien gleichen Namens werden bei einem neuen Lauf ersetzt. Mit `--output`
kann jederzeit ausdrücklich ein anderer Ordner gewählt werden; diese Angabe
hat Vorrang vor dem aus der URL abgeleiteten Namen.
Unvollständige, aber strukturell gültige historische Jurylisten werden im
Terminal zur Bestätigung angeboten. Ohne interaktives Terminal werden sie
standardmäßig abgelehnt. `--accept-incomplete` akzeptiert sie automatisch;
fehlende Ränge werden dabei nicht ergänzt oder verschoben. Alle entsprechenden
Entscheidungen stehen in `incomplete_jury_lists.csv`.
## Tests
```sh
+98 -10
View File
@@ -48,6 +48,14 @@ class Vote:
source_url: str
class IncompleteJuryListError(RuntimeError):
def __init__(self, votes: list[Vote]):
self.votes = votes
self.found_ranks = [vote.rank for vote in votes]
self.missing_ranks = sorted(set(range(1, 11)) - set(self.found_ranks))
super().__init__("Keine vollständige Top-10-Tabelle gefunden")
def clean(value: str) -> str:
return re.sub(r"\s+", " ", value.replace("\xa0", " ")).strip()
@@ -169,7 +177,7 @@ def extract_votes_from_historical_text(
for element in soup.find_all(["tr", "li", "p"]):
text = clean(element.get_text(" ", strip=True))
match = re.fullmatch(r"(10|[1-9])\.\s*(.+)", text)
match = re.fullmatch(r"(10|[1-9])(?:\.\s*|\s+)(.+)", text)
if not match:
continue
@@ -178,6 +186,8 @@ def extract_votes_from_historical_text(
artist, separator, title = entry.partition(":")
if not separator:
artist, separator, title = entry.partition(" - ")
if not separator:
artist, separator, title = entry.partition(". ")
if not separator:
continue
@@ -195,10 +205,10 @@ def extract_votes_from_historical_text(
source_url=source_url,
)
if set(rows_by_rank) != set(range(1, 11)):
if not rows_by_rank:
return None
return [rows_by_rank[rank] for rank in range(1, 11)]
return [rows_by_rank[rank] for rank in sorted(rows_by_rank)]
def parse_jury_page(
@@ -215,12 +225,37 @@ def parse_jury_page(
return votes
votes = extract_votes_from_historical_text(soup, juror, source_url)
if votes is not None:
if votes and len(votes) == 10:
return votes
if votes:
raise IncompleteJuryListError(votes)
raise RuntimeError("Keine vollständige Top-10-Tabelle gefunden")
def decide_incomplete_list(
error: IncompleteJuryListError,
*,
accept_incomplete: bool,
interactive: bool,
decision_helper=input,
) -> bool:
vote = error.votes[0]
print("\nUnvollständige Juryliste:")
print(f" Juror: {vote.juror}")
print(f" URL: {vote.source_url}")
print(f" Gefundene Ränge: {', '.join(map(str, error.found_ranks))}")
print(f" Fehlende Ränge: {', '.join(map(str, error.missing_ranks))}")
if accept_incomplete:
return True
if not interactive:
return False
answer = decision_helper("Count this incomplete jury list anyway? [y/N]: ")
return answer.strip().casefold() in {"y", "yes"}
def aggregate(votes: list[Vote]) -> pd.DataFrame:
groups: dict[str, list[Vote]] = defaultdict(list)
for vote in votes:
@@ -287,11 +322,16 @@ def write_outputs(
votes: list[Vote],
ranking: pd.DataFrame,
errors: list[dict[str, str]],
incomplete_lists: list[dict[str, object]],
) -> None:
output_dir.mkdir(parents=True, exist_ok=True)
votes_df = pd.DataFrame(asdict(vote) for vote in votes)
errors_df = pd.DataFrame(errors)
incomplete_df = pd.DataFrame(
incomplete_lists,
columns=["juror", "url", "found_ranks", "missing_ranks", "accepted"],
)
ranking.head(100).to_csv(
output_dir / "top100.csv",
@@ -313,6 +353,11 @@ def write_outputs(
index=False,
encoding="utf-8-sig",
)
incomplete_df.to_csv(
output_dir / "incomplete_jury_lists.csv",
index=False,
encoding="utf-8-sig",
)
with pd.ExcelWriter(
output_dir / "radioeins_top100.xlsx",
@@ -334,7 +379,7 @@ def write_outputs(
)
def main() -> int:
def build_argument_parser() -> argparse.ArgumentParser:
parser = argparse.ArgumentParser(
description="radioeins-Jurylisten zu einer Top 100 zusammenfassen"
)
@@ -342,7 +387,16 @@ def main() -> int:
parser.add_argument("--limit", type=int, default=None)
parser.add_argument("--delay", type=float, default=0.2)
parser.add_argument("--output", type=Path, default=None)
args = parser.parse_args()
parser.add_argument(
"--accept-incomplete",
action="store_true",
help="unvollständige historische Jurylisten ohne Nachfrage akzeptieren",
)
return parser
def main() -> int:
args = build_argument_parser().parse_args()
output_dir = args.output or derive_output_dir(args.url)
session = requests.Session()
@@ -356,16 +410,38 @@ def main() -> int:
all_votes: list[Vote] = []
errors: list[dict[str, str]] = []
incomplete_lists: list[dict[str, object]] = []
valid_jury_lists = 0
for index, (url, name) in enumerate(pages, start=1):
print(f"[{index:>3}/{len(pages)}] {name}")
try:
all_votes.extend(
parse_jury_page(
votes = parse_jury_page(
html=get_html(session, url),
fallback_name=name,
source_url=url,
)
except IncompleteJuryListError as exc:
accepted = decide_incomplete_list(
exc,
accept_incomplete=args.accept_incomplete,
interactive=sys.stdin.isatty(),
)
incomplete_lists.append(
{
"juror": exc.votes[0].juror,
"url": exc.votes[0].source_url,
"found_ranks": ",".join(map(str, exc.found_ranks)),
"missing_ranks": ",".join(map(str, exc.missing_ranks)),
"accepted": accepted,
}
)
if accepted:
all_votes.extend(exc.votes)
valid_jury_lists += 1
else:
errors.append(
{"juror": name, "url": url, "error": str(exc)}
)
except Exception as exc:
errors.append(
@@ -375,19 +451,31 @@ def main() -> int:
"error": str(exc),
}
)
else:
all_votes.extend(votes)
valid_jury_lists += 1
if args.delay > 0:
time.sleep(args.delay)
ranking = aggregate(all_votes)
write_outputs(output_dir, all_votes, ranking, errors)
write_outputs(output_dir, all_votes, ranking, errors, incomplete_lists)
print("\nFertig:")
print(f" gefundene Seiten: {len(pages)}")
print(f" gültige Jurylisten: {len(all_votes) // 10}")
print(f" gültige Jurylisten: {valid_jury_lists}")
print(f" Wertungen: {len(all_votes)}")
print(f" verschiedene Titel: {len(ranking)}")
print(f" Fehler: {len(errors)}")
print(f" unvollständig gefunden: {len(incomplete_lists)}")
print(
" unvollständig akzeptiert: "
f"{sum(bool(item['accepted']) for item in incomplete_lists)}"
)
print(
" unvollständig abgelehnt: "
f"{sum(not bool(item['accepted']) for item in incomplete_lists)}"
)
print(f" Ausgabe: {output_dir.resolve()}")
return 1 if errors else 0
+14
View File
@@ -0,0 +1,14 @@
<!doctype html>
<html><head><meta property="og:title" content="Aditya Sharma"></head><body>
<table>
<tr><td>1. The Sundays: Here's Where The Story Ends</td><td>&nbsp;</td><td>&nbsp;</td></tr>
<tr><td>2. The Beatles: Something</td><td>&nbsp;</td><td>&nbsp;</td></tr>
<tr><td>3. The Pogues &amp; Kirsty MacColl: Fairytale of New York</td><td>&nbsp;</td><td>&nbsp;</td></tr>
<tr><td>4. China Crisis: Wishful Thinking</td><td>&nbsp;</td><td>&nbsp;</td></tr>
<tr><td>5&nbsp;&nbsp;Nick Drake: Northern Sky</td><td>&nbsp;</td><td>&nbsp;</td></tr>
<tr><td>6. Elton John: Your Song</td><td>&nbsp;</td><td>&nbsp;</td></tr>
<tr><td>7. Sam Cooke: Cupid</td><td>&nbsp;</td><td>&nbsp;</td></tr>
<tr><td>8. Carole King: You've Got&nbsp;A Friend</td><td>&nbsp;</td><td>&nbsp;</td></tr>
<tr><td>9. Turbonegro: I&nbsp;Got Erection</td><td>&nbsp;</td><td>&nbsp;</td></tr>
<tr><td>10. Dead Kennedys: Too Drunk To Fuck</td><td>&nbsp;</td><td>&nbsp;</td></tr>
</table></body></html>
+14
View File
@@ -0,0 +1,14 @@
<!doctype html>
<html><head><meta property="og:title" content="Christiane Falk"></head><body>
<table>
<tr><td>1. Pearl Jam: Black</td><td>&nbsp;</td><td>&nbsp;</td></tr>
<tr><td>2. Element of Crime: Weißes Papier</td><td>&nbsp;</td><td>&nbsp;</td></tr>
<tr><td>3. R.E.M.: Country Feedback</td><td>&nbsp;</td><td>&nbsp;</td></tr>
<tr><td>4. Jeff Buckley. Hallelujah</td><td>&nbsp;</td><td>&nbsp;</td></tr>
<tr><td>5. Gisbert zu Knyphausen: Dreh dich nicht um</td><td>&nbsp;</td><td>&nbsp;</td></tr>
<tr><td>6. John Grant feat. Midlake: I Wanna Go To Marz</td><td>&nbsp;</td><td>&nbsp;</td></tr>
<tr><td>7. James Blake: Limit To Your Love</td><td>&nbsp;</td><td>&nbsp;</td></tr>
<tr><td>8. The National: I Need My Girl</td><td>&nbsp;</td><td>&nbsp;</td></tr>
<tr><td>9. Blur: No Distance Left To Run</td><td>&nbsp;</td><td>&nbsp;</td></tr>
<tr><td>10. Keane: Bend and Break</td><td>&nbsp;</td><td>&nbsp;</td></tr>
</table></body></html>
+14
View File
@@ -0,0 +1,14 @@
<!doctype html>
<html><head><meta property="og:title" content="Dagobert"></head><body>
<table>
<tr><td>1. Doris Day: The Way I Dreamed It</td><td>&nbsp;</td><td>&nbsp;</td></tr>
<tr><td>2. Telly Savalas: If</td><td>&nbsp;</td><td>&nbsp;</td></tr>
<tr><td>3. Lou Reed: Crazy Feeling</td><td>&nbsp;</td><td>&nbsp;</td></tr>
<tr><td>4. Roy Rogers: Cleanin' My Rifle (And Dreamin' Of You)</td><td>&nbsp;</td><td>&nbsp;</td></tr>
<tr><td>5. Carpenters: I Need To Be In Love</td><td>&nbsp;</td><td>&nbsp;</td></tr>
<tr><td>6. Hank Williams: Cold Cold Heart</td><td>&nbsp;</td><td>&nbsp;</td></tr>
<tr><td>7. Chris Isaak: Wicked Game</td><td>&nbsp;</td><td>&nbsp;</td></tr>
<tr><td>8. Astrud Gilberto: Never My Love</td><td>&nbsp;</td><td>&nbsp;</td></tr>
<tr><td>9. Tiny Tim: Earth Angel</td><td>&nbsp;</td><td>&nbsp;</td></tr>
<tr><td>&nbsp;</td><td>&nbsp;</td><td>&nbsp;</td></tr>
</table></body></html>
+14
View File
@@ -0,0 +1,14 @@
<!doctype html>
<html><head><meta property="og:title" content="Sven-Erik Stephan"></head><body>
<table>
<tr><td>1. The Cure: Lovesong</td><td>&nbsp;</td><td>&nbsp;</td></tr>
<tr><td>2. Joy Division: Love Will Tear Us Apart</td><td>&nbsp;</td><td>&nbsp;</td></tr>
<tr><td>3. Roxy Music: More Than This</td><td>&nbsp;</td><td>&nbsp;</td></tr>
<tr><td>4. Prince: Raspberry Beret</td><td>&nbsp;</td><td>&nbsp;</td></tr>
<tr><td>5. Mazzy Star: Fade Into You</td><td>&nbsp;</td><td>&nbsp;</td></tr>
<tr><td>6. Rufus &amp; Chaka Khan: Ain’t Nobody</td><td>&nbsp;</td><td>&nbsp;</td></tr>
<tr><td>7. Cyndi Lauper. Time After Time</td><td>&nbsp;</td><td>&nbsp;</td></tr>
<tr><td>8. Arthur Russell: That’s Us/ Wild Cimbination</td><td>&nbsp;</td><td>&nbsp;</td></tr>
<tr><td>9. The Pharcyde: Passin’ Me By</td><td>&nbsp;</td><td>&nbsp;</td></tr>
<tr><td>10. Frankie Goes To Hollywood: The Power Of Love</td><td>&nbsp;</td><td>&nbsp;</td></tr>
</table></body></html>
+77 -1
View File
@@ -1,7 +1,16 @@
import unittest
from contextlib import redirect_stdout
from io import StringIO
from pathlib import Path
from radioeins_top100 import POINTS, derive_output_dir, parse_jury_page
from radioeins_top100 import (
POINTS,
IncompleteJuryListError,
build_argument_parser,
decide_incomplete_list,
derive_output_dir,
parse_jury_page,
)
class RadioeinsTop100Tests(unittest.TestCase):
@@ -67,6 +76,73 @@ class RadioeinsTop100Tests(unittest.TestCase):
self.assertEqual(votes[7].artist, "Boxhmasters")
self.assertEqual(votes[7].title, "Mogli")
def test_historical_rank_without_dot(self):
votes = self.parse_fixture("love-aditya-sharma.html")
self.assertEqual((votes[4].rank, votes[4].artist, votes[4].title),
(5, "Nick Drake", "Northern Sky"))
def test_historical_period_separator(self):
christiane = self.parse_fixture("love-christiane-falk.html")
sven = self.parse_fixture("love-sven-erik-stephan.html")
self.assertEqual((christiane[3].artist, christiane[3].title),
("Jeff Buckley", "Hallelujah"))
self.assertEqual((sven[6].artist, sven[6].title),
("Cyndi Lauper", "Time After Time"))
self.assertEqual(christiane[2].artist, "R.E.M.")
def test_incomplete_historical_list_is_rejected(self):
with self.assertRaises(IncompleteJuryListError) as caught:
self.parse_fixture("love-dagobert.html")
output = StringIO()
with redirect_stdout(output):
accepted = decide_incomplete_list(
caught.exception,
accept_incomplete=False,
interactive=False,
decision_helper=lambda _: self.fail("must not prompt"),
)
self.assertFalse(accepted)
self.assertIn("Dagobert", output.getvalue())
self.assertEqual(caught.exception.missing_ranks, [10])
def test_incomplete_historical_list_is_accepted_interactively(self):
with self.assertRaises(IncompleteJuryListError) as caught:
self.parse_fixture("love-dagobert.html")
with redirect_stdout(StringIO()):
accepted = decide_incomplete_list(
caught.exception,
accept_incomplete=False,
interactive=True,
decision_helper=lambda prompt: "yes",
)
self.assertTrue(accepted)
self.assertEqual(
[(vote.rank, vote.points) for vote in caught.exception.votes],
[(1, 12), (2, 10), (3, 8), (4, 7), (5, 6),
(6, 5), (7, 4), (8, 3), (9, 2)],
)
def test_accept_incomplete_option(self):
args = build_argument_parser().parse_args(["--accept-incomplete"])
self.assertTrue(args.accept_incomplete)
with self.assertRaises(IncompleteJuryListError) as caught:
self.parse_fixture("love-dagobert.html")
with redirect_stdout(StringIO()):
accepted = decide_incomplete_list(
caught.exception,
accept_incomplete=args.accept_incomplete,
interactive=False,
decision_helper=lambda _: self.fail("must not prompt"),
)
self.assertTrue(accepted)
if __name__ == "__main__":
unittest.main()