Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
9ce4af69be |
@@ -24,6 +24,37 @@ from mka.application.performance import DEFAULT_PERFORMANCE_PROFILE, resolve_oll
|
||||
from mka.application.progress_timing import ProcessingTimer
|
||||
|
||||
STAGES = ("preparing", "transcription", "diarization", "protocol_generation")
|
||||
_EXCERPT_TOKEN_PATTERN = re.compile(r"[^\W\d_]+|\d+", re.UNICODE)
|
||||
_EXCERPT_STOP_WORDS = frozenset(
|
||||
{
|
||||
"a", "an", "and", "are", "das", "der", "die", "ein", "eine", "for", "i",
|
||||
"ich", "im", "in", "is", "it", "ja", "mit", "of", "or", "the", "to", "und",
|
||||
"we", "wir", "with",
|
||||
}
|
||||
)
|
||||
_ACKNOWLEDGEMENT_WORDS = frozenset(
|
||||
{
|
||||
"absolutely", "alles", "danke", "genau", "gut", "ja", "klar", "okay", "ok",
|
||||
"perfekt", "right", "sure", "thanks", "verstanden", "yes",
|
||||
}
|
||||
)
|
||||
_FIRST_PERSON_WORDS = frozenset(
|
||||
{"i", "ich", "me", "mein", "meine", "my", "unser", "unsere", "we", "wir"}
|
||||
)
|
||||
_ACTION_WORDS = frozenset(
|
||||
{
|
||||
"agree", "approve", "believe", "can", "decide", "habe", "kann", "möchte", "plan",
|
||||
"recommend", "think", "übernehme", "werde", "will",
|
||||
}
|
||||
)
|
||||
_DATE_WORDS = frozenset(
|
||||
{
|
||||
"april", "august", "december", "dienstag", "donnerstag", "february", "freitag",
|
||||
"friday", "january", "juli", "june", "mai", "march", "monday", "mittwoch", "montag",
|
||||
"november", "october", "saturday", "september", "sonntag", "sunday", "thursday",
|
||||
"tuesday", "wednesday",
|
||||
}
|
||||
)
|
||||
|
||||
|
||||
class MeetingLabPort(Protocol):
|
||||
@@ -109,6 +140,89 @@ class SpeakerMappingReview:
|
||||
current_mappings: dict[str, str]
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class _ExcerptCandidate:
|
||||
speaker_label: str
|
||||
text: str
|
||||
segment_index: int
|
||||
tokens: frozenset[str]
|
||||
lexical_tokens: frozenset[str]
|
||||
previous_other_speaker_tokens: frozenset[str]
|
||||
|
||||
|
||||
def _excerpt_tokens(text: str) -> tuple[str, ...]:
|
||||
return tuple(token.lower() for token in _EXCERPT_TOKEN_PATTERN.findall(text))
|
||||
|
||||
|
||||
def _excerpt_score(candidate: _ExcerptCandidate, distinctive_token_count: int) -> float:
|
||||
"""Score an excerpt using deterministic lexical signals only."""
|
||||
word_count = len(candidate.tokens)
|
||||
lexical_count = len(candidate.lexical_tokens)
|
||||
score = min(word_count, 24) * 0.3 + min(lexical_count, 16) * 0.6
|
||||
if word_count < 5:
|
||||
score -= 8.0
|
||||
elif word_count < 9:
|
||||
score -= 3.0
|
||||
if lexical_count <= 2:
|
||||
score -= 4.0
|
||||
if candidate.tokens and candidate.tokens <= _ACKNOWLEDGEMENT_WORDS:
|
||||
score -= 14.0
|
||||
if candidate.text.rstrip().endswith("?"):
|
||||
score -= 5.0
|
||||
if any(token.isdigit() for token in candidate.tokens):
|
||||
score += 2.5
|
||||
if candidate.tokens & _DATE_WORDS:
|
||||
score += 2.0
|
||||
if candidate.tokens & _FIRST_PERSON_WORDS:
|
||||
score += 1.5
|
||||
if candidate.tokens & _ACTION_WORDS:
|
||||
score += 2.0
|
||||
proper_name_count = len(re.findall(r"\b[A-ZÄÖÜ][a-zäöüß]{2,}\b", candidate.text))
|
||||
score += min(proper_name_count, 3) * 0.75
|
||||
score += min(sum(len(token) >= 10 for token in candidate.lexical_tokens), 4) * 0.4
|
||||
score += min(distinctive_token_count, 6) * 0.65
|
||||
if candidate.previous_other_speaker_tokens and candidate.lexical_tokens:
|
||||
overlap = len(candidate.lexical_tokens & candidate.previous_other_speaker_tokens)
|
||||
if overlap / len(candidate.lexical_tokens) >= 0.7:
|
||||
score -= 5.0
|
||||
return score
|
||||
|
||||
|
||||
def _segment_bucket(segment_index: int, segment_count: int) -> int:
|
||||
return min(5, segment_index * 6 // max(segment_count, 1))
|
||||
|
||||
|
||||
def _selection_score(
|
||||
candidate: _ExcerptCandidate,
|
||||
base_score: float,
|
||||
selected: list[_ExcerptCandidate],
|
||||
selected_buckets: set[int],
|
||||
segment_count: int,
|
||||
) -> float:
|
||||
"""Favor excerpts from new meeting positions and avoid near duplicates."""
|
||||
score = base_score
|
||||
bucket = _segment_bucket(candidate.segment_index, segment_count)
|
||||
if bucket not in selected_buckets:
|
||||
score += 2.5
|
||||
if not selected or not candidate.lexical_tokens:
|
||||
return score
|
||||
similarities = [
|
||||
len(candidate.lexical_tokens & item.lexical_tokens)
|
||||
/ len(candidate.lexical_tokens | item.lexical_tokens)
|
||||
for item in selected
|
||||
if item.lexical_tokens
|
||||
]
|
||||
maximum_similarity = max(similarities, default=0.0)
|
||||
if maximum_similarity >= 0.72:
|
||||
score -= 18.0
|
||||
elif maximum_similarity >= 0.55:
|
||||
score -= 6.0
|
||||
nearest_position = min(abs(candidate.segment_index - item.segment_index) for item in selected)
|
||||
if nearest_position / max(segment_count - 1, 1) < 0.12:
|
||||
score -= 1.5
|
||||
return score
|
||||
|
||||
|
||||
def stable_id(value: str) -> str:
|
||||
"""Return a schema-safe ID based on user text, with a random fallback."""
|
||||
normalized = re.sub(r"[^a-z0-9]+", "-", value.lower()).strip("-")
|
||||
@@ -324,7 +438,7 @@ class MeetingProcessingService:
|
||||
self,
|
||||
run_dir: Path,
|
||||
*,
|
||||
excerpts_per_speaker: int = 3,
|
||||
excerpts_per_speaker: int = 6,
|
||||
minimum_excerpt_characters: int = 20,
|
||||
) -> SpeakerMappingReview | None:
|
||||
"""Load detected labels and small deterministic identification excerpts."""
|
||||
@@ -341,34 +455,67 @@ class MeetingProcessingService:
|
||||
if not isinstance(segments, list):
|
||||
raise ValueError("Diarized transcript has no segments list.")
|
||||
|
||||
texts_by_speaker: dict[str, list[str]] = {}
|
||||
short_by_speaker: dict[str, list[str]] = {}
|
||||
for segment in segments:
|
||||
candidates_by_speaker: dict[str, list[_ExcerptCandidate]] = {}
|
||||
short_candidates_by_speaker: dict[str, list[_ExcerptCandidate]] = {}
|
||||
previous_label: str | None = None
|
||||
previous_tokens: frozenset[str] = frozenset()
|
||||
for segment_index, segment in enumerate(segments):
|
||||
if not isinstance(segment, dict):
|
||||
continue
|
||||
label = segment.get("speaker_id")
|
||||
text = segment.get("text")
|
||||
normalized = " ".join(text.split()) if isinstance(text, str) else ""
|
||||
tokens = frozenset(_excerpt_tokens(normalized))
|
||||
if (
|
||||
not isinstance(label, str)
|
||||
or not label.startswith("SPEAKER_")
|
||||
or label == "SPEAKER_UNASSIGNED"
|
||||
or not isinstance(text, str)
|
||||
or not text.strip()
|
||||
or not normalized
|
||||
):
|
||||
if normalized:
|
||||
previous_label = label if isinstance(label, str) else None
|
||||
previous_tokens = tokens
|
||||
continue
|
||||
normalized = " ".join(text.split())
|
||||
target = (
|
||||
texts_by_speaker
|
||||
if len(normalized) >= minimum_excerpt_characters
|
||||
else short_by_speaker
|
||||
)
|
||||
target.setdefault(label, []).append(normalized)
|
||||
|
||||
labels = sorted(set(texts_by_speaker) | set(short_by_speaker))
|
||||
candidate = _ExcerptCandidate(
|
||||
speaker_label=label,
|
||||
text=normalized,
|
||||
segment_index=segment_index,
|
||||
tokens=tokens,
|
||||
lexical_tokens=tokens - _EXCERPT_STOP_WORDS,
|
||||
previous_other_speaker_tokens=(
|
||||
previous_tokens
|
||||
if previous_label is not None and previous_label != label
|
||||
else frozenset()
|
||||
),
|
||||
)
|
||||
target = (
|
||||
candidates_by_speaker
|
||||
if len(normalized) >= minimum_excerpt_characters
|
||||
else short_candidates_by_speaker
|
||||
)
|
||||
target.setdefault(label, []).append(candidate)
|
||||
previous_label = label
|
||||
previous_tokens = tokens
|
||||
|
||||
labels = sorted(set(candidates_by_speaker) | set(short_candidates_by_speaker))
|
||||
token_speakers: dict[str, set[str]] = {}
|
||||
for candidate_group in (*candidates_by_speaker.values(), *short_candidates_by_speaker.values()):
|
||||
for candidate in candidate_group:
|
||||
for token in candidate.lexical_tokens:
|
||||
token_speakers.setdefault(token, set()).add(candidate.speaker_label)
|
||||
speakers = []
|
||||
for label in labels:
|
||||
candidates = texts_by_speaker.get(label) or short_by_speaker.get(label, [])
|
||||
speakers.append(SpeakerReview(label, tuple(candidates[:excerpts_per_speaker])))
|
||||
candidates = candidates_by_speaker.get(label) or short_candidates_by_speaker.get(
|
||||
label, []
|
||||
)
|
||||
excerpts = self._select_identifying_excerpts(
|
||||
candidates,
|
||||
excerpts_per_speaker=excerpts_per_speaker,
|
||||
segment_count=len(segments),
|
||||
token_speakers=token_speakers,
|
||||
)
|
||||
speakers.append(SpeakerReview(label, excerpts))
|
||||
participants = tuple(
|
||||
(participant["participant_id"], participant["display_name"])
|
||||
for participant in context_data.get("participants", [])
|
||||
@@ -383,6 +530,50 @@ class MeetingProcessingService:
|
||||
current_mappings=dict(mappings) if isinstance(mappings, dict) else {},
|
||||
)
|
||||
|
||||
@staticmethod
|
||||
def _select_identifying_excerpts(
|
||||
candidates: list[_ExcerptCandidate],
|
||||
*,
|
||||
excerpts_per_speaker: int,
|
||||
segment_count: int,
|
||||
token_speakers: dict[str, set[str]],
|
||||
) -> tuple[str, ...]:
|
||||
"""Choose informative, distinct excerpts while retaining transcript order."""
|
||||
if len(candidates) <= excerpts_per_speaker:
|
||||
return tuple(candidate.text for candidate in candidates)
|
||||
|
||||
base_scores = {
|
||||
candidate: _excerpt_score(
|
||||
candidate,
|
||||
sum(
|
||||
len(token_speakers.get(token, set())) == 1
|
||||
for token in candidate.lexical_tokens
|
||||
),
|
||||
)
|
||||
for candidate in candidates
|
||||
}
|
||||
selected: list[_ExcerptCandidate] = []
|
||||
selected_buckets: set[int] = set()
|
||||
while len(selected) < excerpts_per_speaker:
|
||||
choice = max(
|
||||
(candidate for candidate in candidates if candidate not in selected),
|
||||
key=lambda candidate: (
|
||||
_selection_score(
|
||||
candidate,
|
||||
base_scores[candidate],
|
||||
selected,
|
||||
selected_buckets,
|
||||
segment_count,
|
||||
),
|
||||
-candidate.segment_index,
|
||||
),
|
||||
)
|
||||
selected.append(choice)
|
||||
selected_buckets.add(_segment_bucket(choice.segment_index, segment_count))
|
||||
return tuple(
|
||||
candidate.text for candidate in sorted(selected, key=lambda item: item.segment_index)
|
||||
)
|
||||
|
||||
def regenerate_protocol(
|
||||
self,
|
||||
run_dir: Path,
|
||||
|
||||
@@ -247,6 +247,18 @@ def write_speaker_review_artifacts(run_dir: Path) -> Path:
|
||||
return transcript_path
|
||||
|
||||
|
||||
def write_custom_speaker_review_artifacts(
|
||||
run_dir: Path, segments: list[dict[str, str | None]]
|
||||
) -> Path:
|
||||
"""Write custom diarized segments using the standard review context fixture."""
|
||||
transcript_path = write_speaker_review_artifacts(run_dir)
|
||||
transcript_path.write_text(
|
||||
json.dumps({"speaker_labels_anonymous": True, "segments": segments}),
|
||||
encoding="utf-8",
|
||||
)
|
||||
return transcript_path
|
||||
|
||||
|
||||
def test_build_context_uses_actual_v1_shape(tmp_path: Path) -> None:
|
||||
service, gateway = make_service(tmp_path)
|
||||
|
||||
@@ -534,6 +546,89 @@ def test_speaker_review_lists_detected_labels_participants_and_excerpts(
|
||||
assert "SPEAKER_00" in source.read_text(encoding="utf-8")
|
||||
|
||||
|
||||
def test_speaker_review_prefers_content_rich_excerpts_over_acknowledgements_and_questions(
|
||||
tmp_path: Path,
|
||||
) -> None:
|
||||
service, gateway = make_service(tmp_path)
|
||||
acknowledgement = "Okay, that sounds good."
|
||||
question = "Could we continue now?"
|
||||
rich_excerpts = [
|
||||
"I will complete the Atlas deployment in Berlin on 14 September with the database team.",
|
||||
"I recommend moving the Orion product launch because the supplier test is incomplete.",
|
||||
"We decided that I will own the Kubernetes migration for the payments service.",
|
||||
"My experience with the Frankfurt factory rollout suggests a two-week validation period.",
|
||||
"I will present the security audit results to the Apollo project steering group.",
|
||||
"We should reserve twelve hours for the API integration and production monitoring.",
|
||||
]
|
||||
write_custom_speaker_review_artifacts(
|
||||
gateway.run_dir,
|
||||
[{"speaker_id": "SPEAKER_00", "text": text} for text in [acknowledgement, question, *rich_excerpts]],
|
||||
)
|
||||
|
||||
review = service.load_speaker_mapping_review(gateway.run_dir)
|
||||
|
||||
assert review is not None
|
||||
excerpts = review.speakers[0].excerpts
|
||||
assert len(excerpts) == 6
|
||||
assert acknowledgement not in excerpts
|
||||
assert question not in excerpts
|
||||
assert set(excerpts) == set(rich_excerpts)
|
||||
|
||||
|
||||
def test_speaker_review_avoids_near_duplicate_excerpts(tmp_path: Path) -> None:
|
||||
service, gateway = make_service(tmp_path)
|
||||
primary = "I will lead the Atlas API migration for the Berlin platform team next Monday."
|
||||
duplicate = "I will lead the Atlas API migration for the Berlin platform group next Monday."
|
||||
alternatives = [
|
||||
"I recommend documenting the supplier quality decision in the project register.",
|
||||
"My team will complete the production calibration before the factory trial.",
|
||||
"I will present the customer research findings at the Munich planning workshop.",
|
||||
"Our database monitoring threshold needs approval from the operations group.",
|
||||
"I will coordinate the Orion release checklist with the support organization.",
|
||||
]
|
||||
write_custom_speaker_review_artifacts(
|
||||
gateway.run_dir,
|
||||
[{"speaker_id": "SPEAKER_00", "text": text} for text in [primary, duplicate, *alternatives]],
|
||||
)
|
||||
|
||||
review = service.load_speaker_mapping_review(gateway.run_dir)
|
||||
|
||||
assert review is not None
|
||||
excerpts = review.speakers[0].excerpts
|
||||
assert primary in excerpts
|
||||
assert duplicate not in excerpts
|
||||
assert set(excerpts) == {primary, *alternatives}
|
||||
|
||||
|
||||
def test_speaker_review_spreads_selected_excerpts_across_meeting_positions(
|
||||
tmp_path: Path,
|
||||
) -> None:
|
||||
service, gateway = make_service(tmp_path)
|
||||
positions = {
|
||||
0: "I will lead the Atlas deployment review with the engineering team marker alpha.",
|
||||
1: "I will lead the Atlas deployment review with the engineering team marker bravo.",
|
||||
6: "I will lead the Atlas deployment review with the engineering team marker charlie.",
|
||||
7: "I will lead the Atlas deployment review with the engineering team marker delta.",
|
||||
12: "I will lead the Atlas deployment review with the engineering team marker echo.",
|
||||
13: "I will lead the Atlas deployment review with the engineering team marker foxtrot.",
|
||||
18: "I will lead the Atlas deployment review with the engineering team marker golf.",
|
||||
19: "I will lead the Atlas deployment review with the engineering team marker hotel.",
|
||||
}
|
||||
segments = [
|
||||
{
|
||||
"speaker_id": "SPEAKER_00" if index in positions else "SPEAKER_01",
|
||||
"text": positions.get(index, "Acknowledged."),
|
||||
}
|
||||
for index in range(20)
|
||||
]
|
||||
write_custom_speaker_review_artifacts(gateway.run_dir, segments)
|
||||
|
||||
review = service.load_speaker_mapping_review(gateway.run_dir)
|
||||
|
||||
assert review is not None
|
||||
assert review.speakers[0].excerpts == tuple(positions[index] for index in (0, 1, 6, 7, 12, 18))
|
||||
|
||||
|
||||
def test_protocol_only_regeneration_persists_mappings_without_rewriting_transcript(
|
||||
tmp_path: Path,
|
||||
) -> None:
|
||||
|
||||
Reference in New Issue
Block a user