Improve speaker mapping UX

This commit is contained in:
2026-09-12 11:22:09 +02:00
parent e55dd74c19
commit 9840953687
4 changed files with 212 additions and 23 deletions
+69 -23
View File
@@ -21,6 +21,7 @@ from mka.application.meeting_service import (
MeetingProcessingService,
ParticipantInput,
ProcessingOptions,
SpeakerMappingReview,
stable_id,
)
from mka.application.people_yaml import (
@@ -340,6 +341,73 @@ def _progress_callback(
return update
def _speaker_options(
participants: tuple[str, ...], mappings: dict[str, str | None], speaker_label: str
) -> list[str | None]:
"""Reserve other speakers' participants while retaining this speaker's mapping."""
current = mappings.get(speaker_label)
reserved = {value for label, value in mappings.items() if label != speaker_label}
options: list[str | None] = [None]
options.extend(person for person in participants if person == current or person not in reserved)
if current is not None and current not in options:
options.append(current)
return options
def _mapping_counts(
speaker_labels: tuple[str, ...], mappings: dict[str, str | None]
) -> tuple[int, int, int]:
"""Count only detected speakers, including explicit cleared selections."""
detected = len(speaker_labels)
assigned = sum(mappings.get(label) is not None for label in speaker_labels)
return detected, assigned, detected - assigned
def _render_speaker_mapping(review: SpeakerMappingReview, run_name: str) -> dict[str, str]:
st.subheader("Identify diarized speakers")
st.caption(
"Confirm identities explicitly. Unmapped speakers remain anonymous; "
"the diarized source transcript is not modified."
)
participant_names = dict(review.participants)
# Read every widget before rendering so later speakers also reserve their person.
keys = {
speaker.speaker_label: f"speaker_mapping_{run_name}_{speaker.speaker_label}"
for speaker in review.speakers
}
mappings = {
label: st.session_state.get(key, review.current_mappings.get(label))
for label, key in keys.items()
}
detected, assigned, unassigned = _mapping_counts(tuple(keys), mappings)
st.markdown(f"**{detected} speakers detected · {assigned} assigned · {unassigned} unassigned**")
if unassigned:
st.warning(
"Some detected speakers have no confirmed participant mapping. "
"Check whether a participant is missing or speaker assignment is incomplete. "
"You can still generate a protocol with anonymous speakers."
)
selections: dict[str, str] = {}
for speaker in review.speakers:
label = speaker.speaker_label
current = mappings[label]
options = _speaker_options(tuple(participant_names), mappings, label)
st.session_state[keys[label]] = current
selected = st.selectbox(
f"{label} — Unassigned" if current is None else label,
options=options,
format_func=lambda value, names=participant_names: (
"Unmapped / Unknown" if value is None else names.get(value, value)
),
key=keys[label],
)
if selected is not None:
selections[label] = selected
for excerpt in speaker.excerpts:
st.caption(f"“{excerpt}”")
return selections
def _render_result() -> None:
outcome = st.session_state.outcome
if outcome is None:
@@ -383,29 +451,7 @@ def _render_result() -> None:
if review is None or not review.speakers:
return
st.subheader("Identify diarized speakers")
st.caption(
"Confirm identities explicitly. Unmapped speakers remain anonymous; "
"the diarized source transcript is not modified."
)
participant_names = dict(review.participants)
options = [None, *participant_names]
selections: dict[str, str] = {}
for speaker in review.speakers:
current = review.current_mappings.get(speaker.speaker_label)
selected = st.selectbox(
speaker.speaker_label,
options=options,
index=options.index(current) if current in options else 0,
format_func=lambda value, names=participant_names: (
"Unmapped / Unknown" if value is None else names[value]
),
key=f"speaker_mapping_{outcome.run_dir.name}_{speaker.speaker_label}",
)
if selected is not None:
selections[speaker.speaker_label] = selected
for excerpt in speaker.excerpts:
st.caption(f"“{excerpt}”")
selections = _render_speaker_mapping(review, outcome.run_dir.name)
duplicate_assignments = len(selections.values()) != len(set(selections.values()))
if duplicate_assignments: