Improve speaker-review excerpts

This commit is contained in:
2026-09-17 19:42:35 +02:00
parent 2215414282
commit 9ce4af69be
2 changed files with 302 additions and 16 deletions
+207 -16
View File
@@ -24,6 +24,37 @@ from mka.application.performance import DEFAULT_PERFORMANCE_PROFILE, resolve_oll
from mka.application.progress_timing import ProcessingTimer
STAGES = ("preparing", "transcription", "diarization", "protocol_generation")
_EXCERPT_TOKEN_PATTERN = re.compile(r"[^\W\d_]+|\d+", re.UNICODE)
_EXCERPT_STOP_WORDS = frozenset(
{
"a", "an", "and", "are", "das", "der", "die", "ein", "eine", "for", "i",
"ich", "im", "in", "is", "it", "ja", "mit", "of", "or", "the", "to", "und",
"we", "wir", "with",
}
)
_ACKNOWLEDGEMENT_WORDS = frozenset(
{
"absolutely", "alles", "danke", "genau", "gut", "ja", "klar", "okay", "ok",
"perfekt", "right", "sure", "thanks", "verstanden", "yes",
}
)
_FIRST_PERSON_WORDS = frozenset(
{"i", "ich", "me", "mein", "meine", "my", "unser", "unsere", "we", "wir"}
)
_ACTION_WORDS = frozenset(
{
"agree", "approve", "believe", "can", "decide", "habe", "kann", "möchte", "plan",
"recommend", "think", "übernehme", "werde", "will",
}
)
_DATE_WORDS = frozenset(
{
"april", "august", "december", "dienstag", "donnerstag", "february", "freitag",
"friday", "january", "juli", "june", "mai", "march", "monday", "mittwoch", "montag",
"november", "october", "saturday", "september", "sonntag", "sunday", "thursday",
"tuesday", "wednesday",
}
)
class MeetingLabPort(Protocol):
@@ -109,6 +140,89 @@ class SpeakerMappingReview:
current_mappings: dict[str, str]
@dataclass(frozen=True)
class _ExcerptCandidate:
speaker_label: str
text: str
segment_index: int
tokens: frozenset[str]
lexical_tokens: frozenset[str]
previous_other_speaker_tokens: frozenset[str]
def _excerpt_tokens(text: str) -> tuple[str, ...]:
return tuple(token.lower() for token in _EXCERPT_TOKEN_PATTERN.findall(text))
def _excerpt_score(candidate: _ExcerptCandidate, distinctive_token_count: int) -> float:
"""Score an excerpt using deterministic lexical signals only."""
word_count = len(candidate.tokens)
lexical_count = len(candidate.lexical_tokens)
score = min(word_count, 24) * 0.3 + min(lexical_count, 16) * 0.6
if word_count < 5:
score -= 8.0
elif word_count < 9:
score -= 3.0
if lexical_count <= 2:
score -= 4.0
if candidate.tokens and candidate.tokens <= _ACKNOWLEDGEMENT_WORDS:
score -= 14.0
if candidate.text.rstrip().endswith("?"):
score -= 5.0
if any(token.isdigit() for token in candidate.tokens):
score += 2.5
if candidate.tokens & _DATE_WORDS:
score += 2.0
if candidate.tokens & _FIRST_PERSON_WORDS:
score += 1.5
if candidate.tokens & _ACTION_WORDS:
score += 2.0
proper_name_count = len(re.findall(r"\b[A-ZÄÖÜ][a-zäöüß]{2,}\b", candidate.text))
score += min(proper_name_count, 3) * 0.75
score += min(sum(len(token) >= 10 for token in candidate.lexical_tokens), 4) * 0.4
score += min(distinctive_token_count, 6) * 0.65
if candidate.previous_other_speaker_tokens and candidate.lexical_tokens:
overlap = len(candidate.lexical_tokens & candidate.previous_other_speaker_tokens)
if overlap / len(candidate.lexical_tokens) >= 0.7:
score -= 5.0
return score
def _segment_bucket(segment_index: int, segment_count: int) -> int:
return min(5, segment_index * 6 // max(segment_count, 1))
def _selection_score(
candidate: _ExcerptCandidate,
base_score: float,
selected: list[_ExcerptCandidate],
selected_buckets: set[int],
segment_count: int,
) -> float:
"""Favor excerpts from new meeting positions and avoid near duplicates."""
score = base_score
bucket = _segment_bucket(candidate.segment_index, segment_count)
if bucket not in selected_buckets:
score += 2.5
if not selected or not candidate.lexical_tokens:
return score
similarities = [
len(candidate.lexical_tokens & item.lexical_tokens)
/ len(candidate.lexical_tokens | item.lexical_tokens)
for item in selected
if item.lexical_tokens
]
maximum_similarity = max(similarities, default=0.0)
if maximum_similarity >= 0.72:
score -= 18.0
elif maximum_similarity >= 0.55:
score -= 6.0
nearest_position = min(abs(candidate.segment_index - item.segment_index) for item in selected)
if nearest_position / max(segment_count - 1, 1) < 0.12:
score -= 1.5
return score
def stable_id(value: str) -> str:
"""Return a schema-safe ID based on user text, with a random fallback."""
normalized = re.sub(r"[^a-z0-9]+", "-", value.lower()).strip("-")
@@ -324,7 +438,7 @@ class MeetingProcessingService:
self,
run_dir: Path,
*,
excerpts_per_speaker: int = 3,
excerpts_per_speaker: int = 6,
minimum_excerpt_characters: int = 20,
) -> SpeakerMappingReview | None:
"""Load detected labels and small deterministic identification excerpts."""
@@ -341,34 +455,67 @@ class MeetingProcessingService:
if not isinstance(segments, list):
raise ValueError("Diarized transcript has no segments list.")
texts_by_speaker: dict[str, list[str]] = {}
short_by_speaker: dict[str, list[str]] = {}
for segment in segments:
candidates_by_speaker: dict[str, list[_ExcerptCandidate]] = {}
short_candidates_by_speaker: dict[str, list[_ExcerptCandidate]] = {}
previous_label: str | None = None
previous_tokens: frozenset[str] = frozenset()
for segment_index, segment in enumerate(segments):
if not isinstance(segment, dict):
continue
label = segment.get("speaker_id")
text = segment.get("text")
normalized = " ".join(text.split()) if isinstance(text, str) else ""
tokens = frozenset(_excerpt_tokens(normalized))
if (
not isinstance(label, str)
or not label.startswith("SPEAKER_")
or label == "SPEAKER_UNASSIGNED"
or not isinstance(text, str)
or not text.strip()
or not normalized
):
if normalized:
previous_label = label if isinstance(label, str) else None
previous_tokens = tokens
continue
normalized = " ".join(text.split())
target = (
texts_by_speaker
if len(normalized) >= minimum_excerpt_characters
else short_by_speaker
)
target.setdefault(label, []).append(normalized)
labels = sorted(set(texts_by_speaker) | set(short_by_speaker))
candidate = _ExcerptCandidate(
speaker_label=label,
text=normalized,
segment_index=segment_index,
tokens=tokens,
lexical_tokens=tokens - _EXCERPT_STOP_WORDS,
previous_other_speaker_tokens=(
previous_tokens
if previous_label is not None and previous_label != label
else frozenset()
),
)
target = (
candidates_by_speaker
if len(normalized) >= minimum_excerpt_characters
else short_candidates_by_speaker
)
target.setdefault(label, []).append(candidate)
previous_label = label
previous_tokens = tokens
labels = sorted(set(candidates_by_speaker) | set(short_candidates_by_speaker))
token_speakers: dict[str, set[str]] = {}
for candidate_group in (*candidates_by_speaker.values(), *short_candidates_by_speaker.values()):
for candidate in candidate_group:
for token in candidate.lexical_tokens:
token_speakers.setdefault(token, set()).add(candidate.speaker_label)
speakers = []
for label in labels:
candidates = texts_by_speaker.get(label) or short_by_speaker.get(label, [])
speakers.append(SpeakerReview(label, tuple(candidates[:excerpts_per_speaker])))
candidates = candidates_by_speaker.get(label) or short_candidates_by_speaker.get(
label, []
)
excerpts = self._select_identifying_excerpts(
candidates,
excerpts_per_speaker=excerpts_per_speaker,
segment_count=len(segments),
token_speakers=token_speakers,
)
speakers.append(SpeakerReview(label, excerpts))
participants = tuple(
(participant["participant_id"], participant["display_name"])
for participant in context_data.get("participants", [])
@@ -383,6 +530,50 @@ class MeetingProcessingService:
current_mappings=dict(mappings) if isinstance(mappings, dict) else {},
)
@staticmethod
def _select_identifying_excerpts(
candidates: list[_ExcerptCandidate],
*,
excerpts_per_speaker: int,
segment_count: int,
token_speakers: dict[str, set[str]],
) -> tuple[str, ...]:
"""Choose informative, distinct excerpts while retaining transcript order."""
if len(candidates) <= excerpts_per_speaker:
return tuple(candidate.text for candidate in candidates)
base_scores = {
candidate: _excerpt_score(
candidate,
sum(
len(token_speakers.get(token, set())) == 1
for token in candidate.lexical_tokens
),
)
for candidate in candidates
}
selected: list[_ExcerptCandidate] = []
selected_buckets: set[int] = set()
while len(selected) < excerpts_per_speaker:
choice = max(
(candidate for candidate in candidates if candidate not in selected),
key=lambda candidate: (
_selection_score(
candidate,
base_scores[candidate],
selected,
selected_buckets,
segment_count,
),
-candidate.segment_index,
),
)
selected.append(choice)
selected_buckets.add(_segment_bucket(choice.segment_index, segment_count))
return tuple(
candidate.text for candidate in sorted(selected, key=lambda item: item.segment_index)
)
def regenerate_protocol(
self,
run_dir: Path,