From 9ce4af69be8a6ee0ff62138c6917c57eff3a2122 Mon Sep 17 00:00:00 2001 From: Martin Date: Thu, 17 Sep 2026 19:42:35 +0200 Subject: [PATCH] Improve speaker-review excerpts --- src/mka/application/meeting_service.py | 223 +++++++++++++++++++++++-- tests/test_meeting_service.py | 95 +++++++++++ 2 files changed, 302 insertions(+), 16 deletions(-) diff --git a/src/mka/application/meeting_service.py b/src/mka/application/meeting_service.py index da82542..e9ead11 100644 --- a/src/mka/application/meeting_service.py +++ b/src/mka/application/meeting_service.py @@ -24,6 +24,37 @@ from mka.application.performance import DEFAULT_PERFORMANCE_PROFILE, resolve_oll from mka.application.progress_timing import ProcessingTimer STAGES = ("preparing", "transcription", "diarization", "protocol_generation") +_EXCERPT_TOKEN_PATTERN = re.compile(r"[^\W\d_]+|\d+", re.UNICODE) +_EXCERPT_STOP_WORDS = frozenset( + { + "a", "an", "and", "are", "das", "der", "die", "ein", "eine", "for", "i", + "ich", "im", "in", "is", "it", "ja", "mit", "of", "or", "the", "to", "und", + "we", "wir", "with", + } +) +_ACKNOWLEDGEMENT_WORDS = frozenset( + { + "absolutely", "alles", "danke", "genau", "gut", "ja", "klar", "okay", "ok", + "perfekt", "right", "sure", "thanks", "verstanden", "yes", + } +) +_FIRST_PERSON_WORDS = frozenset( + {"i", "ich", "me", "mein", "meine", "my", "unser", "unsere", "we", "wir"} +) +_ACTION_WORDS = frozenset( + { + "agree", "approve", "believe", "can", "decide", "habe", "kann", "möchte", "plan", + "recommend", "think", "übernehme", "werde", "will", + } +) +_DATE_WORDS = frozenset( + { + "april", "august", "december", "dienstag", "donnerstag", "february", "freitag", + "friday", "january", "juli", "june", "mai", "march", "monday", "mittwoch", "montag", + "november", "october", "saturday", "september", "sonntag", "sunday", "thursday", + "tuesday", "wednesday", + } +) class MeetingLabPort(Protocol): @@ -109,6 +140,89 @@ class SpeakerMappingReview: current_mappings: dict[str, str] +@dataclass(frozen=True) +class _ExcerptCandidate: + speaker_label: str + text: str + segment_index: int + tokens: frozenset[str] + lexical_tokens: frozenset[str] + previous_other_speaker_tokens: frozenset[str] + + +def _excerpt_tokens(text: str) -> tuple[str, ...]: + return tuple(token.lower() for token in _EXCERPT_TOKEN_PATTERN.findall(text)) + + +def _excerpt_score(candidate: _ExcerptCandidate, distinctive_token_count: int) -> float: + """Score an excerpt using deterministic lexical signals only.""" + word_count = len(candidate.tokens) + lexical_count = len(candidate.lexical_tokens) + score = min(word_count, 24) * 0.3 + min(lexical_count, 16) * 0.6 + if word_count < 5: + score -= 8.0 + elif word_count < 9: + score -= 3.0 + if lexical_count <= 2: + score -= 4.0 + if candidate.tokens and candidate.tokens <= _ACKNOWLEDGEMENT_WORDS: + score -= 14.0 + if candidate.text.rstrip().endswith("?"): + score -= 5.0 + if any(token.isdigit() for token in candidate.tokens): + score += 2.5 + if candidate.tokens & _DATE_WORDS: + score += 2.0 + if candidate.tokens & _FIRST_PERSON_WORDS: + score += 1.5 + if candidate.tokens & _ACTION_WORDS: + score += 2.0 + proper_name_count = len(re.findall(r"\b[A-ZÄÖÜ][a-zäöüß]{2,}\b", candidate.text)) + score += min(proper_name_count, 3) * 0.75 + score += min(sum(len(token) >= 10 for token in candidate.lexical_tokens), 4) * 0.4 + score += min(distinctive_token_count, 6) * 0.65 + if candidate.previous_other_speaker_tokens and candidate.lexical_tokens: + overlap = len(candidate.lexical_tokens & candidate.previous_other_speaker_tokens) + if overlap / len(candidate.lexical_tokens) >= 0.7: + score -= 5.0 + return score + + +def _segment_bucket(segment_index: int, segment_count: int) -> int: + return min(5, segment_index * 6 // max(segment_count, 1)) + + +def _selection_score( + candidate: _ExcerptCandidate, + base_score: float, + selected: list[_ExcerptCandidate], + selected_buckets: set[int], + segment_count: int, +) -> float: + """Favor excerpts from new meeting positions and avoid near duplicates.""" + score = base_score + bucket = _segment_bucket(candidate.segment_index, segment_count) + if bucket not in selected_buckets: + score += 2.5 + if not selected or not candidate.lexical_tokens: + return score + similarities = [ + len(candidate.lexical_tokens & item.lexical_tokens) + / len(candidate.lexical_tokens | item.lexical_tokens) + for item in selected + if item.lexical_tokens + ] + maximum_similarity = max(similarities, default=0.0) + if maximum_similarity >= 0.72: + score -= 18.0 + elif maximum_similarity >= 0.55: + score -= 6.0 + nearest_position = min(abs(candidate.segment_index - item.segment_index) for item in selected) + if nearest_position / max(segment_count - 1, 1) < 0.12: + score -= 1.5 + return score + + def stable_id(value: str) -> str: """Return a schema-safe ID based on user text, with a random fallback.""" normalized = re.sub(r"[^a-z0-9]+", "-", value.lower()).strip("-") @@ -324,7 +438,7 @@ class MeetingProcessingService: self, run_dir: Path, *, - excerpts_per_speaker: int = 3, + excerpts_per_speaker: int = 6, minimum_excerpt_characters: int = 20, ) -> SpeakerMappingReview | None: """Load detected labels and small deterministic identification excerpts.""" @@ -341,34 +455,67 @@ class MeetingProcessingService: if not isinstance(segments, list): raise ValueError("Diarized transcript has no segments list.") - texts_by_speaker: dict[str, list[str]] = {} - short_by_speaker: dict[str, list[str]] = {} - for segment in segments: + candidates_by_speaker: dict[str, list[_ExcerptCandidate]] = {} + short_candidates_by_speaker: dict[str, list[_ExcerptCandidate]] = {} + previous_label: str | None = None + previous_tokens: frozenset[str] = frozenset() + for segment_index, segment in enumerate(segments): if not isinstance(segment, dict): continue label = segment.get("speaker_id") text = segment.get("text") + normalized = " ".join(text.split()) if isinstance(text, str) else "" + tokens = frozenset(_excerpt_tokens(normalized)) if ( not isinstance(label, str) or not label.startswith("SPEAKER_") or label == "SPEAKER_UNASSIGNED" - or not isinstance(text, str) - or not text.strip() + or not normalized ): + if normalized: + previous_label = label if isinstance(label, str) else None + previous_tokens = tokens continue - normalized = " ".join(text.split()) - target = ( - texts_by_speaker - if len(normalized) >= minimum_excerpt_characters - else short_by_speaker - ) - target.setdefault(label, []).append(normalized) - labels = sorted(set(texts_by_speaker) | set(short_by_speaker)) + candidate = _ExcerptCandidate( + speaker_label=label, + text=normalized, + segment_index=segment_index, + tokens=tokens, + lexical_tokens=tokens - _EXCERPT_STOP_WORDS, + previous_other_speaker_tokens=( + previous_tokens + if previous_label is not None and previous_label != label + else frozenset() + ), + ) + target = ( + candidates_by_speaker + if len(normalized) >= minimum_excerpt_characters + else short_candidates_by_speaker + ) + target.setdefault(label, []).append(candidate) + previous_label = label + previous_tokens = tokens + + labels = sorted(set(candidates_by_speaker) | set(short_candidates_by_speaker)) + token_speakers: dict[str, set[str]] = {} + for candidate_group in (*candidates_by_speaker.values(), *short_candidates_by_speaker.values()): + for candidate in candidate_group: + for token in candidate.lexical_tokens: + token_speakers.setdefault(token, set()).add(candidate.speaker_label) speakers = [] for label in labels: - candidates = texts_by_speaker.get(label) or short_by_speaker.get(label, []) - speakers.append(SpeakerReview(label, tuple(candidates[:excerpts_per_speaker]))) + candidates = candidates_by_speaker.get(label) or short_candidates_by_speaker.get( + label, [] + ) + excerpts = self._select_identifying_excerpts( + candidates, + excerpts_per_speaker=excerpts_per_speaker, + segment_count=len(segments), + token_speakers=token_speakers, + ) + speakers.append(SpeakerReview(label, excerpts)) participants = tuple( (participant["participant_id"], participant["display_name"]) for participant in context_data.get("participants", []) @@ -383,6 +530,50 @@ class MeetingProcessingService: current_mappings=dict(mappings) if isinstance(mappings, dict) else {}, ) + @staticmethod + def _select_identifying_excerpts( + candidates: list[_ExcerptCandidate], + *, + excerpts_per_speaker: int, + segment_count: int, + token_speakers: dict[str, set[str]], + ) -> tuple[str, ...]: + """Choose informative, distinct excerpts while retaining transcript order.""" + if len(candidates) <= excerpts_per_speaker: + return tuple(candidate.text for candidate in candidates) + + base_scores = { + candidate: _excerpt_score( + candidate, + sum( + len(token_speakers.get(token, set())) == 1 + for token in candidate.lexical_tokens + ), + ) + for candidate in candidates + } + selected: list[_ExcerptCandidate] = [] + selected_buckets: set[int] = set() + while len(selected) < excerpts_per_speaker: + choice = max( + (candidate for candidate in candidates if candidate not in selected), + key=lambda candidate: ( + _selection_score( + candidate, + base_scores[candidate], + selected, + selected_buckets, + segment_count, + ), + -candidate.segment_index, + ), + ) + selected.append(choice) + selected_buckets.add(_segment_bucket(choice.segment_index, segment_count)) + return tuple( + candidate.text for candidate in sorted(selected, key=lambda item: item.segment_index) + ) + def regenerate_protocol( self, run_dir: Path, diff --git a/tests/test_meeting_service.py b/tests/test_meeting_service.py index 98499c4..eae9f95 100644 --- a/tests/test_meeting_service.py +++ b/tests/test_meeting_service.py @@ -247,6 +247,18 @@ def write_speaker_review_artifacts(run_dir: Path) -> Path: return transcript_path +def write_custom_speaker_review_artifacts( + run_dir: Path, segments: list[dict[str, str | None]] +) -> Path: + """Write custom diarized segments using the standard review context fixture.""" + transcript_path = write_speaker_review_artifacts(run_dir) + transcript_path.write_text( + json.dumps({"speaker_labels_anonymous": True, "segments": segments}), + encoding="utf-8", + ) + return transcript_path + + def test_build_context_uses_actual_v1_shape(tmp_path: Path) -> None: service, gateway = make_service(tmp_path) @@ -534,6 +546,89 @@ def test_speaker_review_lists_detected_labels_participants_and_excerpts( assert "SPEAKER_00" in source.read_text(encoding="utf-8") +def test_speaker_review_prefers_content_rich_excerpts_over_acknowledgements_and_questions( + tmp_path: Path, +) -> None: + service, gateway = make_service(tmp_path) + acknowledgement = "Okay, that sounds good." + question = "Could we continue now?" + rich_excerpts = [ + "I will complete the Atlas deployment in Berlin on 14 September with the database team.", + "I recommend moving the Orion product launch because the supplier test is incomplete.", + "We decided that I will own the Kubernetes migration for the payments service.", + "My experience with the Frankfurt factory rollout suggests a two-week validation period.", + "I will present the security audit results to the Apollo project steering group.", + "We should reserve twelve hours for the API integration and production monitoring.", + ] + write_custom_speaker_review_artifacts( + gateway.run_dir, + [{"speaker_id": "SPEAKER_00", "text": text} for text in [acknowledgement, question, *rich_excerpts]], + ) + + review = service.load_speaker_mapping_review(gateway.run_dir) + + assert review is not None + excerpts = review.speakers[0].excerpts + assert len(excerpts) == 6 + assert acknowledgement not in excerpts + assert question not in excerpts + assert set(excerpts) == set(rich_excerpts) + + +def test_speaker_review_avoids_near_duplicate_excerpts(tmp_path: Path) -> None: + service, gateway = make_service(tmp_path) + primary = "I will lead the Atlas API migration for the Berlin platform team next Monday." + duplicate = "I will lead the Atlas API migration for the Berlin platform group next Monday." + alternatives = [ + "I recommend documenting the supplier quality decision in the project register.", + "My team will complete the production calibration before the factory trial.", + "I will present the customer research findings at the Munich planning workshop.", + "Our database monitoring threshold needs approval from the operations group.", + "I will coordinate the Orion release checklist with the support organization.", + ] + write_custom_speaker_review_artifacts( + gateway.run_dir, + [{"speaker_id": "SPEAKER_00", "text": text} for text in [primary, duplicate, *alternatives]], + ) + + review = service.load_speaker_mapping_review(gateway.run_dir) + + assert review is not None + excerpts = review.speakers[0].excerpts + assert primary in excerpts + assert duplicate not in excerpts + assert set(excerpts) == {primary, *alternatives} + + +def test_speaker_review_spreads_selected_excerpts_across_meeting_positions( + tmp_path: Path, +) -> None: + service, gateway = make_service(tmp_path) + positions = { + 0: "I will lead the Atlas deployment review with the engineering team marker alpha.", + 1: "I will lead the Atlas deployment review with the engineering team marker bravo.", + 6: "I will lead the Atlas deployment review with the engineering team marker charlie.", + 7: "I will lead the Atlas deployment review with the engineering team marker delta.", + 12: "I will lead the Atlas deployment review with the engineering team marker echo.", + 13: "I will lead the Atlas deployment review with the engineering team marker foxtrot.", + 18: "I will lead the Atlas deployment review with the engineering team marker golf.", + 19: "I will lead the Atlas deployment review with the engineering team marker hotel.", + } + segments = [ + { + "speaker_id": "SPEAKER_00" if index in positions else "SPEAKER_01", + "text": positions.get(index, "Acknowledged."), + } + for index in range(20) + ] + write_custom_speaker_review_artifacts(gateway.run_dir, segments) + + review = service.load_speaker_mapping_review(gateway.run_dir) + + assert review is not None + assert review.speakers[0].excerpts == tuple(positions[index] for index in (0, 1, 6, 7, 12, 18)) + + def test_protocol_only_regeneration_persists_mappings_without_rewriting_transcript( tmp_path: Path, ) -> None: