Author SHA1 Message Date
admin 9ce4af69be Improve speaker-review excerpts 2026-09-17 19:42:35 +02:00
2 changed files with 302 additions and 16 deletions
+207 -16
View File
@@ -24,6 +24,37 @@ from mka.application.performance import DEFAULT_PERFORMANCE_PROFILE, resolve_oll
from mka.application.progress_timing import ProcessingTimer from mka.application.progress_timing import ProcessingTimer
STAGES = ("preparing", "transcription", "diarization", "protocol_generation") STAGES = ("preparing", "transcription", "diarization", "protocol_generation")
_EXCERPT_TOKEN_PATTERN = re.compile(r"[^\W\d_]+|\d+", re.UNICODE)
_EXCERPT_STOP_WORDS = frozenset(
{
"a", "an", "and", "are", "das", "der", "die", "ein", "eine", "for", "i",
"ich", "im", "in", "is", "it", "ja", "mit", "of", "or", "the", "to", "und",
"we", "wir", "with",
}
)
_ACKNOWLEDGEMENT_WORDS = frozenset(
{
"absolutely", "alles", "danke", "genau", "gut", "ja", "klar", "okay", "ok",
"perfekt", "right", "sure", "thanks", "verstanden", "yes",
}
)
_FIRST_PERSON_WORDS = frozenset(
{"i", "ich", "me", "mein", "meine", "my", "unser", "unsere", "we", "wir"}
)
_ACTION_WORDS = frozenset(
{
"agree", "approve", "believe", "can", "decide", "habe", "kann", "möchte", "plan",
"recommend", "think", "übernehme", "werde", "will",
}
)
_DATE_WORDS = frozenset(
{
"april", "august", "december", "dienstag", "donnerstag", "february", "freitag",
"friday", "january", "juli", "june", "mai", "march", "monday", "mittwoch", "montag",
"november", "october", "saturday", "september", "sonntag", "sunday", "thursday",
"tuesday", "wednesday",
}
)
class MeetingLabPort(Protocol): class MeetingLabPort(Protocol):
@@ -109,6 +140,89 @@ class SpeakerMappingReview:
current_mappings: dict[str, str] current_mappings: dict[str, str]
@dataclass(frozen=True)
class _ExcerptCandidate:
speaker_label: str
text: str
segment_index: int
tokens: frozenset[str]
lexical_tokens: frozenset[str]
previous_other_speaker_tokens: frozenset[str]
def _excerpt_tokens(text: str) -> tuple[str, ...]:
return tuple(token.lower() for token in _EXCERPT_TOKEN_PATTERN.findall(text))
def _excerpt_score(candidate: _ExcerptCandidate, distinctive_token_count: int) -> float:
"""Score an excerpt using deterministic lexical signals only."""
word_count = len(candidate.tokens)
lexical_count = len(candidate.lexical_tokens)
score = min(word_count, 24) * 0.3 + min(lexical_count, 16) * 0.6
if word_count < 5:
score -= 8.0
elif word_count < 9:
score -= 3.0
if lexical_count <= 2:
score -= 4.0
if candidate.tokens and candidate.tokens <= _ACKNOWLEDGEMENT_WORDS:
score -= 14.0
if candidate.text.rstrip().endswith("?"):
score -= 5.0
if any(token.isdigit() for token in candidate.tokens):
score += 2.5
if candidate.tokens & _DATE_WORDS:
score += 2.0
if candidate.tokens & _FIRST_PERSON_WORDS:
score += 1.5
if candidate.tokens & _ACTION_WORDS:
score += 2.0
proper_name_count = len(re.findall(r"\b[A-ZÄÖÜ][a-zäöüß]{2,}\b", candidate.text))
score += min(proper_name_count, 3) * 0.75
score += min(sum(len(token) >= 10 for token in candidate.lexical_tokens), 4) * 0.4
score += min(distinctive_token_count, 6) * 0.65
if candidate.previous_other_speaker_tokens and candidate.lexical_tokens:
overlap = len(candidate.lexical_tokens & candidate.previous_other_speaker_tokens)
if overlap / len(candidate.lexical_tokens) >= 0.7:
score -= 5.0
return score
def _segment_bucket(segment_index: int, segment_count: int) -> int:
return min(5, segment_index * 6 // max(segment_count, 1))
def _selection_score(
candidate: _ExcerptCandidate,
base_score: float,
selected: list[_ExcerptCandidate],
selected_buckets: set[int],
segment_count: int,
) -> float:
"""Favor excerpts from new meeting positions and avoid near duplicates."""
score = base_score
bucket = _segment_bucket(candidate.segment_index, segment_count)
if bucket not in selected_buckets:
score += 2.5
if not selected or not candidate.lexical_tokens:
return score
similarities = [
len(candidate.lexical_tokens & item.lexical_tokens)
/ len(candidate.lexical_tokens | item.lexical_tokens)
for item in selected
if item.lexical_tokens
]
maximum_similarity = max(similarities, default=0.0)
if maximum_similarity >= 0.72:
score -= 18.0
elif maximum_similarity >= 0.55:
score -= 6.0
nearest_position = min(abs(candidate.segment_index - item.segment_index) for item in selected)
if nearest_position / max(segment_count - 1, 1) < 0.12:
score -= 1.5
return score
def stable_id(value: str) -> str: def stable_id(value: str) -> str:
"""Return a schema-safe ID based on user text, with a random fallback.""" """Return a schema-safe ID based on user text, with a random fallback."""
normalized = re.sub(r"[^a-z0-9]+", "-", value.lower()).strip("-") normalized = re.sub(r"[^a-z0-9]+", "-", value.lower()).strip("-")
@@ -324,7 +438,7 @@ class MeetingProcessingService:
self, self,
run_dir: Path, run_dir: Path,
*, *,
excerpts_per_speaker: int = 3, excerpts_per_speaker: int = 6,
minimum_excerpt_characters: int = 20, minimum_excerpt_characters: int = 20,
) -> SpeakerMappingReview | None: ) -> SpeakerMappingReview | None:
"""Load detected labels and small deterministic identification excerpts.""" """Load detected labels and small deterministic identification excerpts."""
@@ -341,34 +455,67 @@ class MeetingProcessingService:
if not isinstance(segments, list): if not isinstance(segments, list):
raise ValueError("Diarized transcript has no segments list.") raise ValueError("Diarized transcript has no segments list.")
texts_by_speaker: dict[str, list[str]] = {} candidates_by_speaker: dict[str, list[_ExcerptCandidate]] = {}
short_by_speaker: dict[str, list[str]] = {} short_candidates_by_speaker: dict[str, list[_ExcerptCandidate]] = {}
for segment in segments: previous_label: str | None = None
previous_tokens: frozenset[str] = frozenset()
for segment_index, segment in enumerate(segments):
if not isinstance(segment, dict): if not isinstance(segment, dict):
continue continue
label = segment.get("speaker_id") label = segment.get("speaker_id")
text = segment.get("text") text = segment.get("text")
normalized = " ".join(text.split()) if isinstance(text, str) else ""
tokens = frozenset(_excerpt_tokens(normalized))
if ( if (
not isinstance(label, str) not isinstance(label, str)
or not label.startswith("SPEAKER_") or not label.startswith("SPEAKER_")
or label == "SPEAKER_UNASSIGNED" or label == "SPEAKER_UNASSIGNED"
or not isinstance(text, str) or not normalized
or not text.strip()
): ):
if normalized:
previous_label = label if isinstance(label, str) else None
previous_tokens = tokens
continue continue
normalized = " ".join(text.split())
target = (
texts_by_speaker
if len(normalized) >= minimum_excerpt_characters
else short_by_speaker
)
target.setdefault(label, []).append(normalized)
labels = sorted(set(texts_by_speaker) | set(short_by_speaker)) candidate = _ExcerptCandidate(
speaker_label=label,
text=normalized,
segment_index=segment_index,
tokens=tokens,
lexical_tokens=tokens - _EXCERPT_STOP_WORDS,
previous_other_speaker_tokens=(
previous_tokens
if previous_label is not None and previous_label != label
else frozenset()
),
)
target = (
candidates_by_speaker
if len(normalized) >= minimum_excerpt_characters
else short_candidates_by_speaker
)
target.setdefault(label, []).append(candidate)
previous_label = label
previous_tokens = tokens
labels = sorted(set(candidates_by_speaker) | set(short_candidates_by_speaker))
token_speakers: dict[str, set[str]] = {}
for candidate_group in (*candidates_by_speaker.values(), *short_candidates_by_speaker.values()):
for candidate in candidate_group:
for token in candidate.lexical_tokens:
token_speakers.setdefault(token, set()).add(candidate.speaker_label)
speakers = [] speakers = []
for label in labels: for label in labels:
candidates = texts_by_speaker.get(label) or short_by_speaker.get(label, []) candidates = candidates_by_speaker.get(label) or short_candidates_by_speaker.get(
speakers.append(SpeakerReview(label, tuple(candidates[:excerpts_per_speaker]))) label, []
)
excerpts = self._select_identifying_excerpts(
candidates,
excerpts_per_speaker=excerpts_per_speaker,
segment_count=len(segments),
token_speakers=token_speakers,
)
speakers.append(SpeakerReview(label, excerpts))
participants = tuple( participants = tuple(
(participant["participant_id"], participant["display_name"]) (participant["participant_id"], participant["display_name"])
for participant in context_data.get("participants", []) for participant in context_data.get("participants", [])
@@ -383,6 +530,50 @@ class MeetingProcessingService:
current_mappings=dict(mappings) if isinstance(mappings, dict) else {}, current_mappings=dict(mappings) if isinstance(mappings, dict) else {},
) )
@staticmethod
def _select_identifying_excerpts(
candidates: list[_ExcerptCandidate],
*,
excerpts_per_speaker: int,
segment_count: int,
token_speakers: dict[str, set[str]],
) -> tuple[str, ...]:
"""Choose informative, distinct excerpts while retaining transcript order."""
if len(candidates) <= excerpts_per_speaker:
return tuple(candidate.text for candidate in candidates)
base_scores = {
candidate: _excerpt_score(
candidate,
sum(
len(token_speakers.get(token, set())) == 1
for token in candidate.lexical_tokens
),
)
for candidate in candidates
}
selected: list[_ExcerptCandidate] = []
selected_buckets: set[int] = set()
while len(selected) < excerpts_per_speaker:
choice = max(
(candidate for candidate in candidates if candidate not in selected),
key=lambda candidate: (
_selection_score(
candidate,
base_scores[candidate],
selected,
selected_buckets,
segment_count,
),
-candidate.segment_index,
),
)
selected.append(choice)
selected_buckets.add(_segment_bucket(choice.segment_index, segment_count))
return tuple(
candidate.text for candidate in sorted(selected, key=lambda item: item.segment_index)
)
def regenerate_protocol( def regenerate_protocol(
self, self,
run_dir: Path, run_dir: Path,
+95
View File
@@ -247,6 +247,18 @@ def write_speaker_review_artifacts(run_dir: Path) -> Path:
return transcript_path return transcript_path
def write_custom_speaker_review_artifacts(
run_dir: Path, segments: list[dict[str, str | None]]
) -> Path:
"""Write custom diarized segments using the standard review context fixture."""
transcript_path = write_speaker_review_artifacts(run_dir)
transcript_path.write_text(
json.dumps({"speaker_labels_anonymous": True, "segments": segments}),
encoding="utf-8",
)
return transcript_path
def test_build_context_uses_actual_v1_shape(tmp_path: Path) -> None: def test_build_context_uses_actual_v1_shape(tmp_path: Path) -> None:
service, gateway = make_service(tmp_path) service, gateway = make_service(tmp_path)
@@ -534,6 +546,89 @@ def test_speaker_review_lists_detected_labels_participants_and_excerpts(
assert "SPEAKER_00" in source.read_text(encoding="utf-8") assert "SPEAKER_00" in source.read_text(encoding="utf-8")
def test_speaker_review_prefers_content_rich_excerpts_over_acknowledgements_and_questions(
tmp_path: Path,
) -> None:
service, gateway = make_service(tmp_path)
acknowledgement = "Okay, that sounds good."
question = "Could we continue now?"
rich_excerpts = [
"I will complete the Atlas deployment in Berlin on 14 September with the database team.",
"I recommend moving the Orion product launch because the supplier test is incomplete.",
"We decided that I will own the Kubernetes migration for the payments service.",
"My experience with the Frankfurt factory rollout suggests a two-week validation period.",
"I will present the security audit results to the Apollo project steering group.",
"We should reserve twelve hours for the API integration and production monitoring.",
]
write_custom_speaker_review_artifacts(
gateway.run_dir,
[{"speaker_id": "SPEAKER_00", "text": text} for text in [acknowledgement, question, *rich_excerpts]],
)
review = service.load_speaker_mapping_review(gateway.run_dir)
assert review is not None
excerpts = review.speakers[0].excerpts
assert len(excerpts) == 6
assert acknowledgement not in excerpts
assert question not in excerpts
assert set(excerpts) == set(rich_excerpts)
def test_speaker_review_avoids_near_duplicate_excerpts(tmp_path: Path) -> None:
service, gateway = make_service(tmp_path)
primary = "I will lead the Atlas API migration for the Berlin platform team next Monday."
duplicate = "I will lead the Atlas API migration for the Berlin platform group next Monday."
alternatives = [
"I recommend documenting the supplier quality decision in the project register.",
"My team will complete the production calibration before the factory trial.",
"I will present the customer research findings at the Munich planning workshop.",
"Our database monitoring threshold needs approval from the operations group.",
"I will coordinate the Orion release checklist with the support organization.",
]
write_custom_speaker_review_artifacts(
gateway.run_dir,
[{"speaker_id": "SPEAKER_00", "text": text} for text in [primary, duplicate, *alternatives]],
)
review = service.load_speaker_mapping_review(gateway.run_dir)
assert review is not None
excerpts = review.speakers[0].excerpts
assert primary in excerpts
assert duplicate not in excerpts
assert set(excerpts) == {primary, *alternatives}
def test_speaker_review_spreads_selected_excerpts_across_meeting_positions(
tmp_path: Path,
) -> None:
service, gateway = make_service(tmp_path)
positions = {
0: "I will lead the Atlas deployment review with the engineering team marker alpha.",
1: "I will lead the Atlas deployment review with the engineering team marker bravo.",
6: "I will lead the Atlas deployment review with the engineering team marker charlie.",
7: "I will lead the Atlas deployment review with the engineering team marker delta.",
12: "I will lead the Atlas deployment review with the engineering team marker echo.",
13: "I will lead the Atlas deployment review with the engineering team marker foxtrot.",
18: "I will lead the Atlas deployment review with the engineering team marker golf.",
19: "I will lead the Atlas deployment review with the engineering team marker hotel.",
}
segments = [
{
"speaker_id": "SPEAKER_00" if index in positions else "SPEAKER_01",
"text": positions.get(index, "Acknowledged."),
}
for index in range(20)
]
write_custom_speaker_review_artifacts(gateway.run_dir, segments)
review = service.load_speaker_mapping_review(gateway.run_dir)
assert review is not None
assert review.speakers[0].excerpts == tuple(positions[index] for index in (0, 1, 6, 7, 12, 18))
def test_protocol_only_regeneration_persists_mappings_without_rewriting_transcript( def test_protocol_only_regeneration_persists_mappings_without_rewriting_transcript(
tmp_path: Path, tmp_path: Path,
) -> None: ) -> None: