feat: add post-diarization speaker mapping workflow
This commit is contained in:
@@ -15,6 +15,8 @@ Beginne mit # Meeting Protocol. Gliedere themenorientiert mit ## <Thema> und syn
|
||||
|
||||
Entferne nur Wiederholungen, Füllwörter und Gesprächsrauschen. Erfinde keine Fakten oder Identitäten. Gib kein JSON, keine Sprecherlabels und kein Denkprotokoll aus. Eine belegte themenübergreifende Maßnahmenliste am Ende ist optional."""
|
||||
|
||||
MAPPED_SPEAKER_ATTRIBUTION_INSTRUCTION = """Nutze die autoritativen SPEAKER_XX-zu-Teilnehmer-Zuordnungen im Meeting-Kontext, um ausdrücklich belegte Aussagen, Positionen, Entscheidungen, Zuweisungen und angenommene persönliche Verpflichtungen namentlich zuzuordnen. Eine ausdrückliche Ich-Zusage eines zugeordneten Sprechers belegt persönliche Verantwortung. Unterscheide stets den Sprecher einer Aussage von darin nur erwähnten Personen. Leite für nicht zugeordnete Sprecher keine Identität ab und erfinde keine persönliche Verantwortung. Gib die technischen SPEAKER_XX-Bezeichnungen nicht im nutzerseitigen Protokoll aus."""
|
||||
|
||||
|
||||
def build_direct_protocol_prompt(
|
||||
transcript: str,
|
||||
|
||||
@@ -20,6 +20,7 @@ from src.meeting_lab.models.meeting_context import (
|
||||
)
|
||||
from src.meeting_lab.protocol.direct_protocol_prompt import (
|
||||
COMPACT_DIARIZED_PROTOCOL_INSTRUCTION,
|
||||
MAPPED_SPEAKER_ATTRIBUTION_INSTRUCTION,
|
||||
build_direct_protocol_prompt,
|
||||
)
|
||||
from src.meeting_lab.protocol.transcript_input import (
|
||||
@@ -106,10 +107,16 @@ def select_transcript_input(
|
||||
plain_text = plain_segment_transcript(transcript.get("segments"))
|
||||
except TranscriptInputError as exc:
|
||||
raise DirectProtocolError(str(exc)) from exc
|
||||
instruction = COMPACT_DIARIZED_PROTOCOL_INSTRUCTION
|
||||
if (
|
||||
rendered_context
|
||||
and "Confirmed diarization speaker mappings" in rendered_context
|
||||
):
|
||||
instruction = f"{instruction}\n\n{MAPPED_SPEAKER_ATTRIBUTION_INSTRUCTION}"
|
||||
compact_prompt = build_direct_protocol_prompt(
|
||||
compact.text,
|
||||
rendered_context,
|
||||
instruction=COMPACT_DIARIZED_PROTOCOL_INSTRUCTION,
|
||||
instruction=instruction,
|
||||
)
|
||||
compact_estimate = estimate_input_tokens(compact_prompt)
|
||||
if compact_estimate <= safe_input_token_budget:
|
||||
@@ -205,6 +212,19 @@ def generate_direct_protocol(
|
||||
"input_token_estimation_method": "utf8_bytes_divided_by_4.4",
|
||||
"fallback_used": selected.fallback_used,
|
||||
"diarization_enabled": selected.diarization_enabled,
|
||||
"speaker_attribution_available": (
|
||||
True
|
||||
if selected.representation == "diarized_compact"
|
||||
else False
|
||||
if selected.representation == "plain_transcript_fallback"
|
||||
else None
|
||||
),
|
||||
"speaker_attribution_loss_reason": (
|
||||
"plain_transcript_fallback"
|
||||
if selected.representation == "plain_transcript_fallback"
|
||||
else None
|
||||
),
|
||||
"speaker_mapping_count": len(context.speaker_mappings) if context else 0,
|
||||
}
|
||||
return DirectProtocolResult(
|
||||
protocol_text=generation.text,
|
||||
|
||||
Reference in New Issue
Block a user