Add direct protocol MVP core

This commit is contained in:
2026-08-20 21:42:36 +02:00
parent 70007ea8f2
commit a9dab7c81a
5 changed files with 662 additions and 0 deletions
+95
View File
@@ -0,0 +1,95 @@
"""Minimal Ollama client behavior used by the direct protocol MVP."""
from __future__ import annotations
import time
from dataclasses import dataclass
from typing import Any
import requests
DEFAULT_ENDPOINT = "http://127.0.0.1:11434"
class OllamaError(RuntimeError):
"""Raised when Ollama cannot safely complete the requested operation."""
@dataclass(frozen=True)
class OllamaGeneration:
raw_response: dict[str, Any]
text: str
client_wall_time_seconds: float
def ollama_base_url(endpoint: str) -> str:
endpoint = endpoint.rstrip("/")
return endpoint.rsplit("/api/", 1)[0] if "/api/" in endpoint else endpoint
def generate_url(endpoint: str) -> str:
return f"{ollama_base_url(endpoint)}/api/generate"
def require_model(endpoint: str, model: str, timeout: int = 10) -> dict[str, Any]:
base_url = ollama_base_url(endpoint)
try:
response = requests.get(f"{base_url}/api/tags", timeout=timeout)
response.raise_for_status()
data = response.json()
except (requests.RequestException, ValueError) as exc:
raise OllamaError(f"Ollama endpoint is not reachable at {base_url}: {exc}") from exc
models = data.get("models") if isinstance(data, dict) else None
if not isinstance(models, list):
raise OllamaError("Ollama /api/tags returned a malformed response.")
installed = {
item.get("name")
for item in models
if isinstance(item, dict) and isinstance(item.get("name"), str)
}
if model not in installed:
raise OllamaError(f"Requested model is not installed in Ollama: {model}")
return {"base_url": base_url, "model": model, "installed": True}
def generate_once(
endpoint: str,
model: str,
prompt: str,
*,
timeout: int,
num_ctx: int,
num_predict: int,
) -> OllamaGeneration:
payload = {
"model": model,
"prompt": prompt,
"think": False,
"stream": False,
"options": {
"temperature": 0.0,
"num_ctx": num_ctx,
"num_predict": num_predict,
},
}
started = time.perf_counter()
try:
response = requests.post(generate_url(endpoint), json=payload, timeout=timeout)
response.raise_for_status()
data = response.json()
except requests.RequestException as exc:
raise OllamaError(f"Ollama generation request failed: {exc}") from exc
except ValueError as exc:
raise OllamaError("Ollama generation response is not valid JSON.") from exc
wall_time = time.perf_counter() - started
if not isinstance(data, dict):
raise OllamaError("Ollama generation response must be a JSON object.")
text = data.get("response")
if not isinstance(text, str):
raise OllamaError("Ollama generation response has no string 'response' field.")
if not text.strip():
raise OllamaError("Ollama returned an empty protocol.")
return OllamaGeneration(data, text, wall_time)
@@ -0,0 +1,19 @@
"""Prompt construction for the direct transcript-to-protocol MVP."""
from __future__ import annotations
DIRECT_PROTOCOL_INSTRUCTION = """Erstelle aus dem vollständigen Transkript und dem Meeting-Kontext ein prägnantes, professionelles internes Besprechungsprotokoll in deutscher Sprache.
Das Protokoll muss themenorientiert sein, nicht chronologisch und nicht nach technischen Kategorien gegliedert. Beginne mit # Meeting Protocol. Verwende für jedes kohärente Thema eine Überschrift ## <Thema> und darunter eine knappe Synthese der Diskussion. Nenne Entscheidungen oder abgestimmte Positionen nur, wenn sie tatsächlich belegt sind. Führe Maßnahmen nur auf, wenn eine konkrete zukünftige Handlung gestützt ist; nenne verantwortliche Personen und Fristen ausschließlich bei expliziter Zuweisung, Annahme oder Bestätigung im Transkript. Vorschläge, Einwände, Möglichkeiten und vorläufige Ideen sind keine Entscheidungen oder Verpflichtungen. Bewahre relevante Einschränkungen und ungelöste Meinungsverschiedenheiten. Nenne offene Punkte nur, wenn sie wirklich offen bleiben. Nicht jedes Thema benötigt Entscheidungen, Maßnahmen oder offene Punkte.
Synthetisiere zusammengehörige Aussagen, entferne Füllwörter, Wiederholungen und Gesprächsrauschen und erfinde keine Fakten, Verantwortlichen oder Fristen. Gib kein JSON, keine internen Labels und keine Analyse oder Denkprotokolle aus. Das Ergebnis soll als Markdown-Protokoll nach geringfügiger menschlicher Redaktion intern versendbar sein. Eine kompakte themenübergreifende Maßnahmenliste am Ende ist optional, wenn sie nützlich und vollständig belegt ist."""
def build_direct_protocol_prompt(transcript: str, meeting_context: str | None = None) -> str:
context = meeting_context.strip() if meeting_context else "Kein Meeting-Kontext bereitgestellt."
return (
f"{DIRECT_PROTOCOL_INSTRUCTION}\n\n"
f"MEETING-KONTEXT:\n{context}\n\n"
f"VOLLSTAENDIGES TRANSKRIPT:\n{transcript.strip()}\n"
)
@@ -0,0 +1,111 @@
"""One-call direct protocol generation from a compact Whisper transcript."""
from __future__ import annotations
import json
from dataclasses import dataclass
from pathlib import Path
from typing import Any, Callable
from src.meeting_lab.llm.ollama import (
DEFAULT_ENDPOINT,
OllamaGeneration,
generate_once,
require_model,
)
from src.meeting_lab.models.meeting_context import (
MeetingContext,
load_meeting_context,
render_meeting_context_for_prompt,
)
from src.meeting_lab.protocol.direct_protocol_prompt import build_direct_protocol_prompt
DEFAULT_MODEL = "qwen3.6:35B-A3B"
DEFAULT_NUM_CTX = 32768
DEFAULT_NUM_PREDICT = 8192
DEFAULT_TIMEOUT = 1800
class DirectProtocolError(ValueError):
"""Raised for invalid direct-protocol inputs or model output."""
@dataclass(frozen=True)
class DirectProtocolResult:
protocol_text: str
exact_prompt: str
model_metadata: dict[str, Any]
runtime_metadata: dict[str, Any]
raw_response: dict[str, Any]
def load_compact_transcript(path: Path) -> str:
if not path.is_file():
raise DirectProtocolError(f"Transcript file does not exist: {path}")
try:
data = json.loads(path.read_text(encoding="utf-8-sig"))
except json.JSONDecodeError as exc:
raise DirectProtocolError(f"Transcript is not valid JSON: {path}: {exc}") from exc
if not isinstance(data, dict):
raise DirectProtocolError("Transcript JSON must contain a top-level object.")
if "text" not in data:
raise DirectProtocolError("Transcript JSON must contain top-level 'text'.")
text = data["text"]
if not isinstance(text, str) or not text.strip():
raise DirectProtocolError("Transcript top-level 'text' must be a non-empty string.")
return text
def generate_direct_protocol(
transcript_path: Path,
context_path: Path | None = None,
*,
model: str = DEFAULT_MODEL,
endpoint: str = DEFAULT_ENDPOINT,
timeout: int = DEFAULT_TIMEOUT,
num_ctx: int = DEFAULT_NUM_CTX,
num_predict: int = DEFAULT_NUM_PREDICT,
model_check: Callable[[str, str, int], dict[str, Any]] = require_model,
generation_call: Callable[..., OllamaGeneration] = generate_once,
) -> DirectProtocolResult:
transcript = load_compact_transcript(transcript_path)
context: MeetingContext | None = (
load_meeting_context(context_path) if context_path is not None else None
)
rendered_context = render_meeting_context_for_prompt(context) if context else None
prompt = build_direct_protocol_prompt(transcript, rendered_context)
model_metadata = model_check(endpoint, model, 10)
generation = generation_call(
endpoint,
model,
prompt,
timeout=timeout,
num_ctx=num_ctx,
num_predict=num_predict,
)
data = generation.raw_response
runtime_metadata = {
"model": model,
"prompt_token_count": data.get("prompt_eval_count"),
"output_token_count": data.get("eval_count"),
"prompt_evaluation_duration_ns": data.get("prompt_eval_duration"),
"generation_duration_ns": data.get("eval_duration"),
"total_ollama_duration_ns": data.get("total_duration"),
"client_wall_time_seconds": generation.client_wall_time_seconds,
"completion_reason": data.get("done_reason"),
"done": data.get("done"),
"request_count": 1,
"temperature": 0.0,
"think": False,
"num_ctx": num_ctx,
"num_predict": num_predict,
}
return DirectProtocolResult(
protocol_text=generation.text,
exact_prompt=prompt,
model_metadata=model_metadata,
runtime_metadata=runtime_metadata,
raw_response=data,
)