diff --git a/.gitignore b/.gitignore
index 3a3e5b8..20e3c86 100644
--- a/.gitignore
+++ b/.gitignore
@@ -11,6 +11,7 @@ top100.csv
gesamtwertung.csv
einzelwertungen.csv
fehler.csv
+incomplete_jury_lists.csv
# macOS metadata
.DS_Store
diff --git a/README.md b/README.md
index bdf0629..84bde18 100644
--- a/README.md
+++ b/README.md
@@ -47,6 +47,7 @@ Optionen:
--limit ANZAHL nur die ersten ANZAHL Juryseiten verarbeiten
--delay SEKUNDEN Pause zwischen Abrufen (Standard: 0.2)
--output ORDNER abweichenden Ausgabeordner verwenden
+--accept-incomplete unvollständige historische Listen automatisch akzeptieren
```
Beispiel für einen kurzen Testlauf:
@@ -64,12 +65,19 @@ Beispiel für einen kurzen Testlauf:
- `gesamtwertung.csv`
- `einzelwertungen.csv`
- `fehler.csv`
+- `incomplete_jury_lists.csv`
Der Ausgabeordner wird bei Bedarf automatisch erstellt. Bereits vorhandene
Dateien gleichen Namens werden bei einem neuen Lauf ersetzt. Mit `--output`
kann jederzeit ausdrücklich ein anderer Ordner gewählt werden; diese Angabe
hat Vorrang vor dem aus der URL abgeleiteten Namen.
+Unvollständige, aber strukturell gültige historische Jurylisten werden im
+Terminal zur Bestätigung angeboten. Ohne interaktives Terminal werden sie
+standardmäßig abgelehnt. `--accept-incomplete` akzeptiert sie automatisch;
+fehlende Ränge werden dabei nicht ergänzt oder verschoben. Alle entsprechenden
+Entscheidungen stehen in `incomplete_jury_lists.csv`.
+
## Tests
```sh
diff --git a/radioeins_top100.py b/radioeins_top100.py
index c237e52..226eed0 100644
--- a/radioeins_top100.py
+++ b/radioeins_top100.py
@@ -48,6 +48,14 @@ class Vote:
source_url: str
+class IncompleteJuryListError(RuntimeError):
+ def __init__(self, votes: list[Vote]):
+ self.votes = votes
+ self.found_ranks = [vote.rank for vote in votes]
+ self.missing_ranks = sorted(set(range(1, 11)) - set(self.found_ranks))
+ super().__init__("Keine vollständige Top-10-Tabelle gefunden")
+
+
def clean(value: str) -> str:
return re.sub(r"\s+", " ", value.replace("\xa0", " ")).strip()
@@ -169,7 +177,7 @@ def extract_votes_from_historical_text(
for element in soup.find_all(["tr", "li", "p"]):
text = clean(element.get_text(" ", strip=True))
- match = re.fullmatch(r"(10|[1-9])\.\s*(.+)", text)
+ match = re.fullmatch(r"(10|[1-9])(?:\.\s*|\s+)(.+)", text)
if not match:
continue
@@ -178,6 +186,8 @@ def extract_votes_from_historical_text(
artist, separator, title = entry.partition(":")
if not separator:
artist, separator, title = entry.partition(" - ")
+ if not separator:
+ artist, separator, title = entry.partition(". ")
if not separator:
continue
@@ -195,10 +205,10 @@ def extract_votes_from_historical_text(
source_url=source_url,
)
- if set(rows_by_rank) != set(range(1, 11)):
+ if not rows_by_rank:
return None
- return [rows_by_rank[rank] for rank in range(1, 11)]
+ return [rows_by_rank[rank] for rank in sorted(rows_by_rank)]
def parse_jury_page(
@@ -215,12 +225,37 @@ def parse_jury_page(
return votes
votes = extract_votes_from_historical_text(soup, juror, source_url)
- if votes is not None:
+ if votes and len(votes) == 10:
return votes
+ if votes:
+ raise IncompleteJuryListError(votes)
raise RuntimeError("Keine vollständige Top-10-Tabelle gefunden")
+def decide_incomplete_list(
+ error: IncompleteJuryListError,
+ *,
+ accept_incomplete: bool,
+ interactive: bool,
+ decision_helper=input,
+) -> bool:
+ vote = error.votes[0]
+ print("\nUnvollständige Juryliste:")
+ print(f" Juror: {vote.juror}")
+ print(f" URL: {vote.source_url}")
+ print(f" Gefundene Ränge: {', '.join(map(str, error.found_ranks))}")
+ print(f" Fehlende Ränge: {', '.join(map(str, error.missing_ranks))}")
+
+ if accept_incomplete:
+ return True
+ if not interactive:
+ return False
+
+ answer = decision_helper("Count this incomplete jury list anyway? [y/N]: ")
+ return answer.strip().casefold() in {"y", "yes"}
+
+
def aggregate(votes: list[Vote]) -> pd.DataFrame:
groups: dict[str, list[Vote]] = defaultdict(list)
for vote in votes:
@@ -287,11 +322,16 @@ def write_outputs(
votes: list[Vote],
ranking: pd.DataFrame,
errors: list[dict[str, str]],
+ incomplete_lists: list[dict[str, object]],
) -> None:
output_dir.mkdir(parents=True, exist_ok=True)
votes_df = pd.DataFrame(asdict(vote) for vote in votes)
errors_df = pd.DataFrame(errors)
+ incomplete_df = pd.DataFrame(
+ incomplete_lists,
+ columns=["juror", "url", "found_ranks", "missing_ranks", "accepted"],
+ )
ranking.head(100).to_csv(
output_dir / "top100.csv",
@@ -313,6 +353,11 @@ def write_outputs(
index=False,
encoding="utf-8-sig",
)
+ incomplete_df.to_csv(
+ output_dir / "incomplete_jury_lists.csv",
+ index=False,
+ encoding="utf-8-sig",
+ )
with pd.ExcelWriter(
output_dir / "radioeins_top100.xlsx",
@@ -334,7 +379,7 @@ def write_outputs(
)
-def main() -> int:
+def build_argument_parser() -> argparse.ArgumentParser:
parser = argparse.ArgumentParser(
description="radioeins-Jurylisten zu einer Top 100 zusammenfassen"
)
@@ -342,7 +387,16 @@ def main() -> int:
parser.add_argument("--limit", type=int, default=None)
parser.add_argument("--delay", type=float, default=0.2)
parser.add_argument("--output", type=Path, default=None)
- args = parser.parse_args()
+ parser.add_argument(
+ "--accept-incomplete",
+ action="store_true",
+ help="unvollständige historische Jurylisten ohne Nachfrage akzeptieren",
+ )
+ return parser
+
+
+def main() -> int:
+ args = build_argument_parser().parse_args()
output_dir = args.output or derive_output_dir(args.url)
session = requests.Session()
@@ -356,17 +410,39 @@ def main() -> int:
all_votes: list[Vote] = []
errors: list[dict[str, str]] = []
+ incomplete_lists: list[dict[str, object]] = []
+ valid_jury_lists = 0
for index, (url, name) in enumerate(pages, start=1):
print(f"[{index:>3}/{len(pages)}] {name}")
try:
- all_votes.extend(
- parse_jury_page(
- html=get_html(session, url),
- fallback_name=name,
- source_url=url,
- )
+ votes = parse_jury_page(
+ html=get_html(session, url),
+ fallback_name=name,
+ source_url=url,
)
+ except IncompleteJuryListError as exc:
+ accepted = decide_incomplete_list(
+ exc,
+ accept_incomplete=args.accept_incomplete,
+ interactive=sys.stdin.isatty(),
+ )
+ incomplete_lists.append(
+ {
+ "juror": exc.votes[0].juror,
+ "url": exc.votes[0].source_url,
+ "found_ranks": ",".join(map(str, exc.found_ranks)),
+ "missing_ranks": ",".join(map(str, exc.missing_ranks)),
+ "accepted": accepted,
+ }
+ )
+ if accepted:
+ all_votes.extend(exc.votes)
+ valid_jury_lists += 1
+ else:
+ errors.append(
+ {"juror": name, "url": url, "error": str(exc)}
+ )
except Exception as exc:
errors.append(
{
@@ -375,19 +451,31 @@ def main() -> int:
"error": str(exc),
}
)
+ else:
+ all_votes.extend(votes)
+ valid_jury_lists += 1
if args.delay > 0:
time.sleep(args.delay)
ranking = aggregate(all_votes)
- write_outputs(output_dir, all_votes, ranking, errors)
+ write_outputs(output_dir, all_votes, ranking, errors, incomplete_lists)
print("\nFertig:")
print(f" gefundene Seiten: {len(pages)}")
- print(f" gültige Jurylisten: {len(all_votes) // 10}")
+ print(f" gültige Jurylisten: {valid_jury_lists}")
print(f" Wertungen: {len(all_votes)}")
print(f" verschiedene Titel: {len(ranking)}")
print(f" Fehler: {len(errors)}")
+ print(f" unvollständig gefunden: {len(incomplete_lists)}")
+ print(
+ " unvollständig akzeptiert: "
+ f"{sum(bool(item['accepted']) for item in incomplete_lists)}"
+ )
+ print(
+ " unvollständig abgelehnt: "
+ f"{sum(not bool(item['accepted']) for item in incomplete_lists)}"
+ )
print(f" Ausgabe: {output_dir.resolve()}")
return 1 if errors else 0
diff --git a/tests/fixtures/love-aditya-sharma.html b/tests/fixtures/love-aditya-sharma.html
new file mode 100644
index 0000000..5c7b71e
--- /dev/null
+++ b/tests/fixtures/love-aditya-sharma.html
@@ -0,0 +1,14 @@
+
+
+
+| 1. The Sundays: Here's Where The Story Ends | | |
+| 2. The Beatles: Something | | |
+| 3. The Pogues & Kirsty MacColl: Fairytale of New York | | |
+| 4. China Crisis: Wishful Thinking | | |
+| 5 Nick Drake: Northern Sky | | |
+| 6. Elton John: Your Song | | |
+| 7. Sam Cooke: Cupid | | |
+| 8. Carole King: You've Got A Friend | | |
+| 9. Turbonegro: I Got Erection | | |
+| 10. Dead Kennedys: Too Drunk To Fuck | | |
+
diff --git a/tests/fixtures/love-christiane-falk.html b/tests/fixtures/love-christiane-falk.html
new file mode 100644
index 0000000..93ed684
--- /dev/null
+++ b/tests/fixtures/love-christiane-falk.html
@@ -0,0 +1,14 @@
+
+
+
+| 1. Pearl Jam: Black | | |
+| 2. Element of Crime: Weißes Papier | | |
+| 3. R.E.M.: Country Feedback | | |
+| 4. Jeff Buckley. Hallelujah | | |
+| 5. Gisbert zu Knyphausen: Dreh dich nicht um | | |
+| 6. John Grant feat. Midlake: I Wanna Go To Marz | | |
+| 7. James Blake: Limit To Your Love | | |
+| 8. The National: I Need My Girl | | |
+| 9. Blur: No Distance Left To Run | | |
+| 10. Keane: Bend and Break | | |
+
diff --git a/tests/fixtures/love-dagobert.html b/tests/fixtures/love-dagobert.html
new file mode 100644
index 0000000..b0b2db9
--- /dev/null
+++ b/tests/fixtures/love-dagobert.html
@@ -0,0 +1,14 @@
+
+
+
+| 1. Doris Day: The Way I Dreamed It | | |
+| 2. Telly Savalas: If | | |
+| 3. Lou Reed: Crazy Feeling | | |
+| 4. Roy Rogers: Cleanin' My Rifle (And Dreamin' Of You) | | |
+| 5. Carpenters: I Need To Be In Love | | |
+| 6. Hank Williams: Cold Cold Heart | | |
+| 7. Chris Isaak: Wicked Game | | |
+| 8. Astrud Gilberto: Never My Love | | |
+| 9. Tiny Tim: Earth Angel | | |
+| | | |
+
diff --git a/tests/fixtures/love-sven-erik-stephan.html b/tests/fixtures/love-sven-erik-stephan.html
new file mode 100644
index 0000000..941df8f
--- /dev/null
+++ b/tests/fixtures/love-sven-erik-stephan.html
@@ -0,0 +1,14 @@
+
+
+
+| 1. The Cure: Lovesong | | |
+| 2. Joy Division: Love Will Tear Us Apart | | |
+| 3. Roxy Music: More Than This | | |
+| 4. Prince: Raspberry Beret | | |
+| 5. Mazzy Star: Fade Into You | | |
+| 6. Rufus & Chaka Khan: Ain’t Nobody | | |
+| 7. Cyndi Lauper. Time After Time | | |
+| 8. Arthur Russell: That’s Us/ Wild Cimbination | | |
+| 9. The Pharcyde: Passin’ Me By | | |
+| 10. Frankie Goes To Hollywood: The Power Of Love | | |
+
diff --git a/tests/test_radioeins_top100.py b/tests/test_radioeins_top100.py
index baba79e..43bf15c 100644
--- a/tests/test_radioeins_top100.py
+++ b/tests/test_radioeins_top100.py
@@ -1,7 +1,16 @@
import unittest
+from contextlib import redirect_stdout
+from io import StringIO
from pathlib import Path
-from radioeins_top100 import POINTS, derive_output_dir, parse_jury_page
+from radioeins_top100 import (
+ POINTS,
+ IncompleteJuryListError,
+ build_argument_parser,
+ decide_incomplete_list,
+ derive_output_dir,
+ parse_jury_page,
+)
class RadioeinsTop100Tests(unittest.TestCase):
@@ -67,6 +76,73 @@ class RadioeinsTop100Tests(unittest.TestCase):
self.assertEqual(votes[7].artist, "Boxhmasters")
self.assertEqual(votes[7].title, "Mogli")
+ def test_historical_rank_without_dot(self):
+ votes = self.parse_fixture("love-aditya-sharma.html")
+
+ self.assertEqual((votes[4].rank, votes[4].artist, votes[4].title),
+ (5, "Nick Drake", "Northern Sky"))
+
+ def test_historical_period_separator(self):
+ christiane = self.parse_fixture("love-christiane-falk.html")
+ sven = self.parse_fixture("love-sven-erik-stephan.html")
+
+ self.assertEqual((christiane[3].artist, christiane[3].title),
+ ("Jeff Buckley", "Hallelujah"))
+ self.assertEqual((sven[6].artist, sven[6].title),
+ ("Cyndi Lauper", "Time After Time"))
+ self.assertEqual(christiane[2].artist, "R.E.M.")
+
+ def test_incomplete_historical_list_is_rejected(self):
+ with self.assertRaises(IncompleteJuryListError) as caught:
+ self.parse_fixture("love-dagobert.html")
+
+ output = StringIO()
+ with redirect_stdout(output):
+ accepted = decide_incomplete_list(
+ caught.exception,
+ accept_incomplete=False,
+ interactive=False,
+ decision_helper=lambda _: self.fail("must not prompt"),
+ )
+
+ self.assertFalse(accepted)
+ self.assertIn("Dagobert", output.getvalue())
+ self.assertEqual(caught.exception.missing_ranks, [10])
+
+ def test_incomplete_historical_list_is_accepted_interactively(self):
+ with self.assertRaises(IncompleteJuryListError) as caught:
+ self.parse_fixture("love-dagobert.html")
+
+ with redirect_stdout(StringIO()):
+ accepted = decide_incomplete_list(
+ caught.exception,
+ accept_incomplete=False,
+ interactive=True,
+ decision_helper=lambda prompt: "yes",
+ )
+
+ self.assertTrue(accepted)
+ self.assertEqual(
+ [(vote.rank, vote.points) for vote in caught.exception.votes],
+ [(1, 12), (2, 10), (3, 8), (4, 7), (5, 6),
+ (6, 5), (7, 4), (8, 3), (9, 2)],
+ )
+
+ def test_accept_incomplete_option(self):
+ args = build_argument_parser().parse_args(["--accept-incomplete"])
+ self.assertTrue(args.accept_incomplete)
+
+ with self.assertRaises(IncompleteJuryListError) as caught:
+ self.parse_fixture("love-dagobert.html")
+ with redirect_stdout(StringIO()):
+ accepted = decide_incomplete_list(
+ caught.exception,
+ accept_incomplete=args.accept_incomplete,
+ interactive=False,
+ decision_helper=lambda _: self.fail("must not prompt"),
+ )
+ self.assertTrue(accepted)
+
if __name__ == "__main__":
unittest.main()