Handle incomplete historical jury lists

This commit is contained in:
2026-08-09 12:42:16 +02:00
parent bbf0f1bf0a
commit 5db0e1816b
8 changed files with 244 additions and 15 deletions
+102 -14
View File
@@ -48,6 +48,14 @@ class Vote:
source_url: str
class IncompleteJuryListError(RuntimeError):
def __init__(self, votes: list[Vote]):
self.votes = votes
self.found_ranks = [vote.rank for vote in votes]
self.missing_ranks = sorted(set(range(1, 11)) - set(self.found_ranks))
super().__init__("Keine vollständige Top-10-Tabelle gefunden")
def clean(value: str) -> str:
return re.sub(r"\s+", " ", value.replace("\xa0", " ")).strip()
@@ -169,7 +177,7 @@ def extract_votes_from_historical_text(
for element in soup.find_all(["tr", "li", "p"]):
text = clean(element.get_text(" ", strip=True))
match = re.fullmatch(r"(10|[1-9])\.\s*(.+)", text)
match = re.fullmatch(r"(10|[1-9])(?:\.\s*|\s+)(.+)", text)
if not match:
continue
@@ -178,6 +186,8 @@ def extract_votes_from_historical_text(
artist, separator, title = entry.partition(":")
if not separator:
artist, separator, title = entry.partition(" - ")
if not separator:
artist, separator, title = entry.partition(". ")
if not separator:
continue
@@ -195,10 +205,10 @@ def extract_votes_from_historical_text(
source_url=source_url,
)
if set(rows_by_rank) != set(range(1, 11)):
if not rows_by_rank:
return None
return [rows_by_rank[rank] for rank in range(1, 11)]
return [rows_by_rank[rank] for rank in sorted(rows_by_rank)]
def parse_jury_page(
@@ -215,12 +225,37 @@ def parse_jury_page(
return votes
votes = extract_votes_from_historical_text(soup, juror, source_url)
if votes is not None:
if votes and len(votes) == 10:
return votes
if votes:
raise IncompleteJuryListError(votes)
raise RuntimeError("Keine vollständige Top-10-Tabelle gefunden")
def decide_incomplete_list(
error: IncompleteJuryListError,
*,
accept_incomplete: bool,
interactive: bool,
decision_helper=input,
) -> bool:
vote = error.votes[0]
print("\nUnvollständige Juryliste:")
print(f" Juror: {vote.juror}")
print(f" URL: {vote.source_url}")
print(f" Gefundene Ränge: {', '.join(map(str, error.found_ranks))}")
print(f" Fehlende Ränge: {', '.join(map(str, error.missing_ranks))}")
if accept_incomplete:
return True
if not interactive:
return False
answer = decision_helper("Count this incomplete jury list anyway? [y/N]: ")
return answer.strip().casefold() in {"y", "yes"}
def aggregate(votes: list[Vote]) -> pd.DataFrame:
groups: dict[str, list[Vote]] = defaultdict(list)
for vote in votes:
@@ -287,11 +322,16 @@ def write_outputs(
votes: list[Vote],
ranking: pd.DataFrame,
errors: list[dict[str, str]],
incomplete_lists: list[dict[str, object]],
) -> None:
output_dir.mkdir(parents=True, exist_ok=True)
votes_df = pd.DataFrame(asdict(vote) for vote in votes)
errors_df = pd.DataFrame(errors)
incomplete_df = pd.DataFrame(
incomplete_lists,
columns=["juror", "url", "found_ranks", "missing_ranks", "accepted"],
)
ranking.head(100).to_csv(
output_dir / "top100.csv",
@@ -313,6 +353,11 @@ def write_outputs(
index=False,
encoding="utf-8-sig",
)
incomplete_df.to_csv(
output_dir / "incomplete_jury_lists.csv",
index=False,
encoding="utf-8-sig",
)
with pd.ExcelWriter(
output_dir / "radioeins_top100.xlsx",
@@ -334,7 +379,7 @@ def write_outputs(
)
def main() -> int:
def build_argument_parser() -> argparse.ArgumentParser:
parser = argparse.ArgumentParser(
description="radioeins-Jurylisten zu einer Top 100 zusammenfassen"
)
@@ -342,7 +387,16 @@ def main() -> int:
parser.add_argument("--limit", type=int, default=None)
parser.add_argument("--delay", type=float, default=0.2)
parser.add_argument("--output", type=Path, default=None)
args = parser.parse_args()
parser.add_argument(
"--accept-incomplete",
action="store_true",
help="unvollständige historische Jurylisten ohne Nachfrage akzeptieren",
)
return parser
def main() -> int:
args = build_argument_parser().parse_args()
output_dir = args.output or derive_output_dir(args.url)
session = requests.Session()
@@ -356,17 +410,39 @@ def main() -> int:
all_votes: list[Vote] = []
errors: list[dict[str, str]] = []
incomplete_lists: list[dict[str, object]] = []
valid_jury_lists = 0
for index, (url, name) in enumerate(pages, start=1):
print(f"[{index:>3}/{len(pages)}] {name}")
try:
all_votes.extend(
parse_jury_page(
html=get_html(session, url),
fallback_name=name,
source_url=url,
)
votes = parse_jury_page(
html=get_html(session, url),
fallback_name=name,
source_url=url,
)
except IncompleteJuryListError as exc:
accepted = decide_incomplete_list(
exc,
accept_incomplete=args.accept_incomplete,
interactive=sys.stdin.isatty(),
)
incomplete_lists.append(
{
"juror": exc.votes[0].juror,
"url": exc.votes[0].source_url,
"found_ranks": ",".join(map(str, exc.found_ranks)),
"missing_ranks": ",".join(map(str, exc.missing_ranks)),
"accepted": accepted,
}
)
if accepted:
all_votes.extend(exc.votes)
valid_jury_lists += 1
else:
errors.append(
{"juror": name, "url": url, "error": str(exc)}
)
except Exception as exc:
errors.append(
{
@@ -375,19 +451,31 @@ def main() -> int:
"error": str(exc),
}
)
else:
all_votes.extend(votes)
valid_jury_lists += 1
if args.delay > 0:
time.sleep(args.delay)
ranking = aggregate(all_votes)
write_outputs(output_dir, all_votes, ranking, errors)
write_outputs(output_dir, all_votes, ranking, errors, incomplete_lists)
print("\nFertig:")
print(f" gefundene Seiten: {len(pages)}")
print(f" gültige Jurylisten: {len(all_votes) // 10}")
print(f" gültige Jurylisten: {valid_jury_lists}")
print(f" Wertungen: {len(all_votes)}")
print(f" verschiedene Titel: {len(ranking)}")
print(f" Fehler: {len(errors)}")
print(f" unvollständig gefunden: {len(incomplete_lists)}")
print(
" unvollständig akzeptiert: "
f"{sum(bool(item['accepted']) for item in incomplete_lists)}"
)
print(
" unvollständig abgelehnt: "
f"{sum(not bool(item['accepted']) for item in incomplete_lists)}"
)
print(f" Ausgabe: {output_dir.resolve()}")
return 1 if errors else 0