Add direct protocol MVP core
This commit is contained in:
@@ -0,0 +1,95 @@
|
||||
"""Minimal Ollama client behavior used by the direct protocol MVP."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import time
|
||||
from dataclasses import dataclass
|
||||
from typing import Any
|
||||
|
||||
import requests
|
||||
|
||||
|
||||
DEFAULT_ENDPOINT = "http://127.0.0.1:11434"
|
||||
|
||||
|
||||
class OllamaError(RuntimeError):
|
||||
"""Raised when Ollama cannot safely complete the requested operation."""
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class OllamaGeneration:
|
||||
raw_response: dict[str, Any]
|
||||
text: str
|
||||
client_wall_time_seconds: float
|
||||
|
||||
|
||||
def ollama_base_url(endpoint: str) -> str:
|
||||
endpoint = endpoint.rstrip("/")
|
||||
return endpoint.rsplit("/api/", 1)[0] if "/api/" in endpoint else endpoint
|
||||
|
||||
|
||||
def generate_url(endpoint: str) -> str:
|
||||
return f"{ollama_base_url(endpoint)}/api/generate"
|
||||
|
||||
|
||||
def require_model(endpoint: str, model: str, timeout: int = 10) -> dict[str, Any]:
|
||||
base_url = ollama_base_url(endpoint)
|
||||
try:
|
||||
response = requests.get(f"{base_url}/api/tags", timeout=timeout)
|
||||
response.raise_for_status()
|
||||
data = response.json()
|
||||
except (requests.RequestException, ValueError) as exc:
|
||||
raise OllamaError(f"Ollama endpoint is not reachable at {base_url}: {exc}") from exc
|
||||
|
||||
models = data.get("models") if isinstance(data, dict) else None
|
||||
if not isinstance(models, list):
|
||||
raise OllamaError("Ollama /api/tags returned a malformed response.")
|
||||
installed = {
|
||||
item.get("name")
|
||||
for item in models
|
||||
if isinstance(item, dict) and isinstance(item.get("name"), str)
|
||||
}
|
||||
if model not in installed:
|
||||
raise OllamaError(f"Requested model is not installed in Ollama: {model}")
|
||||
return {"base_url": base_url, "model": model, "installed": True}
|
||||
|
||||
|
||||
def generate_once(
|
||||
endpoint: str,
|
||||
model: str,
|
||||
prompt: str,
|
||||
*,
|
||||
timeout: int,
|
||||
num_ctx: int,
|
||||
num_predict: int,
|
||||
) -> OllamaGeneration:
|
||||
payload = {
|
||||
"model": model,
|
||||
"prompt": prompt,
|
||||
"think": False,
|
||||
"stream": False,
|
||||
"options": {
|
||||
"temperature": 0.0,
|
||||
"num_ctx": num_ctx,
|
||||
"num_predict": num_predict,
|
||||
},
|
||||
}
|
||||
started = time.perf_counter()
|
||||
try:
|
||||
response = requests.post(generate_url(endpoint), json=payload, timeout=timeout)
|
||||
response.raise_for_status()
|
||||
data = response.json()
|
||||
except requests.RequestException as exc:
|
||||
raise OllamaError(f"Ollama generation request failed: {exc}") from exc
|
||||
except ValueError as exc:
|
||||
raise OllamaError("Ollama generation response is not valid JSON.") from exc
|
||||
wall_time = time.perf_counter() - started
|
||||
|
||||
if not isinstance(data, dict):
|
||||
raise OllamaError("Ollama generation response must be a JSON object.")
|
||||
text = data.get("response")
|
||||
if not isinstance(text, str):
|
||||
raise OllamaError("Ollama generation response has no string 'response' field.")
|
||||
if not text.strip():
|
||||
raise OllamaError("Ollama returned an empty protocol.")
|
||||
return OllamaGeneration(data, text, wall_time)
|
||||
|
||||
@@ -0,0 +1,19 @@
|
||||
"""Prompt construction for the direct transcript-to-protocol MVP."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
|
||||
DIRECT_PROTOCOL_INSTRUCTION = """Erstelle aus dem vollständigen Transkript und dem Meeting-Kontext ein prägnantes, professionelles internes Besprechungsprotokoll in deutscher Sprache.
|
||||
|
||||
Das Protokoll muss themenorientiert sein, nicht chronologisch und nicht nach technischen Kategorien gegliedert. Beginne mit # Meeting Protocol. Verwende für jedes kohärente Thema eine Überschrift ## <Thema> und darunter eine knappe Synthese der Diskussion. Nenne Entscheidungen oder abgestimmte Positionen nur, wenn sie tatsächlich belegt sind. Führe Maßnahmen nur auf, wenn eine konkrete zukünftige Handlung gestützt ist; nenne verantwortliche Personen und Fristen ausschließlich bei expliziter Zuweisung, Annahme oder Bestätigung im Transkript. Vorschläge, Einwände, Möglichkeiten und vorläufige Ideen sind keine Entscheidungen oder Verpflichtungen. Bewahre relevante Einschränkungen und ungelöste Meinungsverschiedenheiten. Nenne offene Punkte nur, wenn sie wirklich offen bleiben. Nicht jedes Thema benötigt Entscheidungen, Maßnahmen oder offene Punkte.
|
||||
|
||||
Synthetisiere zusammengehörige Aussagen, entferne Füllwörter, Wiederholungen und Gesprächsrauschen und erfinde keine Fakten, Verantwortlichen oder Fristen. Gib kein JSON, keine internen Labels und keine Analyse oder Denkprotokolle aus. Das Ergebnis soll als Markdown-Protokoll nach geringfügiger menschlicher Redaktion intern versendbar sein. Eine kompakte themenübergreifende Maßnahmenliste am Ende ist optional, wenn sie nützlich und vollständig belegt ist."""
|
||||
|
||||
|
||||
def build_direct_protocol_prompt(transcript: str, meeting_context: str | None = None) -> str:
|
||||
context = meeting_context.strip() if meeting_context else "Kein Meeting-Kontext bereitgestellt."
|
||||
return (
|
||||
f"{DIRECT_PROTOCOL_INSTRUCTION}\n\n"
|
||||
f"MEETING-KONTEXT:\n{context}\n\n"
|
||||
f"VOLLSTAENDIGES TRANSKRIPT:\n{transcript.strip()}\n"
|
||||
)
|
||||
@@ -0,0 +1,111 @@
|
||||
"""One-call direct protocol generation from a compact Whisper transcript."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from dataclasses import dataclass
|
||||
from pathlib import Path
|
||||
from typing import Any, Callable
|
||||
|
||||
from src.meeting_lab.llm.ollama import (
|
||||
DEFAULT_ENDPOINT,
|
||||
OllamaGeneration,
|
||||
generate_once,
|
||||
require_model,
|
||||
)
|
||||
from src.meeting_lab.models.meeting_context import (
|
||||
MeetingContext,
|
||||
load_meeting_context,
|
||||
render_meeting_context_for_prompt,
|
||||
)
|
||||
from src.meeting_lab.protocol.direct_protocol_prompt import build_direct_protocol_prompt
|
||||
|
||||
|
||||
DEFAULT_MODEL = "qwen3.6:35B-A3B"
|
||||
DEFAULT_NUM_CTX = 32768
|
||||
DEFAULT_NUM_PREDICT = 8192
|
||||
DEFAULT_TIMEOUT = 1800
|
||||
|
||||
|
||||
class DirectProtocolError(ValueError):
|
||||
"""Raised for invalid direct-protocol inputs or model output."""
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class DirectProtocolResult:
|
||||
protocol_text: str
|
||||
exact_prompt: str
|
||||
model_metadata: dict[str, Any]
|
||||
runtime_metadata: dict[str, Any]
|
||||
raw_response: dict[str, Any]
|
||||
|
||||
|
||||
def load_compact_transcript(path: Path) -> str:
|
||||
if not path.is_file():
|
||||
raise DirectProtocolError(f"Transcript file does not exist: {path}")
|
||||
try:
|
||||
data = json.loads(path.read_text(encoding="utf-8-sig"))
|
||||
except json.JSONDecodeError as exc:
|
||||
raise DirectProtocolError(f"Transcript is not valid JSON: {path}: {exc}") from exc
|
||||
if not isinstance(data, dict):
|
||||
raise DirectProtocolError("Transcript JSON must contain a top-level object.")
|
||||
if "text" not in data:
|
||||
raise DirectProtocolError("Transcript JSON must contain top-level 'text'.")
|
||||
text = data["text"]
|
||||
if not isinstance(text, str) or not text.strip():
|
||||
raise DirectProtocolError("Transcript top-level 'text' must be a non-empty string.")
|
||||
return text
|
||||
|
||||
|
||||
def generate_direct_protocol(
|
||||
transcript_path: Path,
|
||||
context_path: Path | None = None,
|
||||
*,
|
||||
model: str = DEFAULT_MODEL,
|
||||
endpoint: str = DEFAULT_ENDPOINT,
|
||||
timeout: int = DEFAULT_TIMEOUT,
|
||||
num_ctx: int = DEFAULT_NUM_CTX,
|
||||
num_predict: int = DEFAULT_NUM_PREDICT,
|
||||
model_check: Callable[[str, str, int], dict[str, Any]] = require_model,
|
||||
generation_call: Callable[..., OllamaGeneration] = generate_once,
|
||||
) -> DirectProtocolResult:
|
||||
transcript = load_compact_transcript(transcript_path)
|
||||
context: MeetingContext | None = (
|
||||
load_meeting_context(context_path) if context_path is not None else None
|
||||
)
|
||||
rendered_context = render_meeting_context_for_prompt(context) if context else None
|
||||
prompt = build_direct_protocol_prompt(transcript, rendered_context)
|
||||
|
||||
model_metadata = model_check(endpoint, model, 10)
|
||||
generation = generation_call(
|
||||
endpoint,
|
||||
model,
|
||||
prompt,
|
||||
timeout=timeout,
|
||||
num_ctx=num_ctx,
|
||||
num_predict=num_predict,
|
||||
)
|
||||
data = generation.raw_response
|
||||
runtime_metadata = {
|
||||
"model": model,
|
||||
"prompt_token_count": data.get("prompt_eval_count"),
|
||||
"output_token_count": data.get("eval_count"),
|
||||
"prompt_evaluation_duration_ns": data.get("prompt_eval_duration"),
|
||||
"generation_duration_ns": data.get("eval_duration"),
|
||||
"total_ollama_duration_ns": data.get("total_duration"),
|
||||
"client_wall_time_seconds": generation.client_wall_time_seconds,
|
||||
"completion_reason": data.get("done_reason"),
|
||||
"done": data.get("done"),
|
||||
"request_count": 1,
|
||||
"temperature": 0.0,
|
||||
"think": False,
|
||||
"num_ctx": num_ctx,
|
||||
"num_predict": num_predict,
|
||||
}
|
||||
return DirectProtocolResult(
|
||||
protocol_text=generation.text,
|
||||
exact_prompt=prompt,
|
||||
model_metadata=model_metadata,
|
||||
runtime_metadata=runtime_metadata,
|
||||
raw_response=data,
|
||||
)
|
||||
Reference in New Issue
Block a user