diff --git a/.env.example b/.env.example index e69de29..b01eaca 100644 --- a/.env.example +++ b/.env.example @@ -0,0 +1,10 @@ +MKA_WHISPER_MODEL=/path/to/ggml-large-v3-turbo.bin +MKA_WHISPER_EXECUTABLE=whisper-cli +MKA_FFMPEG_EXECUTABLE=ffmpeg +MKA_PROTOCOL_MODEL=qwen3.8:27B +MKA_OLLAMA_ENDPOINT=http://127.0.0.1:11434 +MKA_DATA_ROOT=data/meetings +MKA_WHISPER_THREADS=auto +MKA_DIARIZATION_MODE=auto +MKA_DIARIZATION_RUNTIME=native +# MKA_DIARIZATION_CONTAINER_IMAGE= diff --git a/.gitignore b/.gitignore index ba39f76..a5f5ef6 100644 --- a/.gitignore +++ b/.gitignore @@ -34,6 +34,7 @@ dist/ *.sqlite3 # Meeting data +data/meetings/* data/recordings/* data/transcripts/* data/database/* @@ -42,3 +43,4 @@ data/database/* !data/recordings/.gitkeep !data/transcripts/.gitkeep !data/database/.gitkeep +!data/meetings/.gitkeep diff --git a/CHANGELOG.md b/CHANGELOG.md index 953f176..4bf65e4 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -2,6 +2,17 @@ ## [Unreleased] +### Added + +- First Streamlit MVP for audio upload, Meeting Context entry, participant + management, processing progress and protocol editing. +- Application service and Meeting Lab adapter with environment-based runtime + configuration. +- Per-meeting upload and edited-protocol persistence. +- Unit tests for context construction, configuration translation, progress, + failure and result handling. +- ADR 0011 for the Meeting Lab MVP backend architecture. + ### Changed - Reconciled the documented MVP with the validated Meeting Lab backend. @@ -10,10 +21,6 @@ boundaries. - Set the user-facing Meeting Assistant GUI as the next milestone. -### Added - -- ADR 0011 for the Meeting Lab MVP backend architecture. - ## [0.1.0] - 2026-07-14 ### Added diff --git a/PROJECT_KNOWLEDGE.md b/PROJECT_KNOWLEDGE.md index 36738ca..dacecdb 100644 --- a/PROJECT_KNOWLEDGE.md +++ b/PROJECT_KNOWLEDGE.md @@ -15,7 +15,7 @@ are derived, reproducible artifacts and require human review. Meeting Lab owns reusable and experimental processing: -- FFmpeg audio preparation +- FFmpeg audio preparation, with optional normalization enabled by default - `whisper.cpp` transcription - optional `pyannote.audio` diarization - protocol generation @@ -47,7 +47,8 @@ Processing logic must not be duplicated in the application. ```text Imported audio - -> prepare as mono, 16 kHz PCM WAV with FFmpeg + -> always prepare as mono, 16 kHz PCM WAV with FFmpeg + (optionally normalize loudness; default on) -> transcribe with whisper.cpp and large-v3-turbo -> optionally diarize with pyannote.audio Community-1 -> generate a protocol directly from the full transcript @@ -66,11 +67,14 @@ requirements. CPU execution remains supported and may be substantially slower. ## Meeting Context and Speakers -`MeetingContext` is a structured domain object containing meeting metadata and -participants. It can also contain optional explicit mappings from +`MeetingContext` is a structured domain object containing meeting metadata, +present participants, and people who were mentioned without attending. The GUI +records exactly `present` or `mentioned_only`; missing legacy participant status +defaults to `present` in Meeting Lab. It can also contain optional explicit mappings from `SPEAKER_XX` labels to `participant_id` values. -A speaker mapping is authoritative only when a user explicitly confirms it. +A speaker mapping is authoritative only when a user explicitly confirms it and +may reference only a present participant. Speakers otherwise remain anonymous. Automatic speaker-name inference is not allowed. The GUI must create and edit Meeting Context; hand-written YAML is not a product requirement. @@ -101,6 +105,11 @@ The product should ultimately offer both a detailed contextual protocol and a shorter participant/distribution version. The short version remains follow-up work if it is not available for the first GUI milestone. +Confirmed meeting-specific corrections should ultimately be reusable by both +protocol views. The detailed correction and selective-regeneration decision is +recorded in [ADR 0012](docs/adr/0012-post-run-corrections.md); it is later +product work, not a current MVP requirement. + ## Design Principles - offline-first where practical diff --git a/README.md b/README.md index 04c09ba..965b6dc 100644 --- a/README.md +++ b/README.md @@ -8,8 +8,9 @@ local processing backend. The first product milestone is a desktop GUI that lets a user: -- select an existing audio recording -- create and edit structured meeting metadata and participants +- select an existing WAV, FLAC, or M4A recording +- create and edit structured meeting metadata and relevant people, including whether + they were present or only mentioned - optionally map anonymous `SPEAKER_XX` labels to known participants - start the Meeting Lab processing pipeline - follow stage-based progress @@ -23,7 +24,7 @@ reference recorder, but imported audio is not tied to OBS-specific behavior. ```text Audio file - -> FFmpeg preparation (mono, 16 kHz PCM WAV) + -> FFmpeg preparation (mono, 16 kHz PCM WAV; normalization optional) -> whisper.cpp transcription (large-v3-turbo) -> optional pyannote.audio Community-1 diarization -> direct full-transcript protocol generation @@ -58,8 +59,67 @@ orchestration and protocol-generation logic. Meeting Assistant owns the GUI, context and participant editing, explicit speaker mapping, progress display, protocol editing and export. -The code currently contains the initial Meeting Assistant project and domain -foundation. The GUI has not yet been implemented. +Audio normalization is enabled by default and can be disabled in the processing +options. This controls loudness normalization only: Meeting Lab still prepares +every WAV, FLAC or M4A source as canonical audio before transcription. + +## Run the Streamlit MVP + +The development setup expects `meeting-assistant` and `meeting-lab` to be +sibling repositories. Meeting Lab currently imports its API through the +`src.meeting_lab` package path, so its repository root must be supplied on +`PYTHONPATH`. Meeting Assistant does not modify `sys.path` at runtime. + +Create a virtual environment and install Meeting Assistant, including +Streamlit: + +```bash +python3 -m venv .venv +.venv/bin/pip install -e '.[dev]' +.venv/bin/pip install -e ../meeting-lab +``` + +Configure at least the local whisper.cpp model. All supported values are shown +in `.env.example`; export them in the shell because the application does not +load `.env` files implicitly: + +```bash +export MKA_WHISPER_MODEL=/path/to/ggml-large-v3-turbo.bin +export MKA_WHISPER_EXECUTABLE=whisper-cli +export MKA_FFMPEG_EXECUTABLE=ffmpeg +export MKA_PROTOCOL_MODEL=qwen3.8:27B +``` + +Optional machine-specific settings include: + +- `MKA_DATA_ROOT` (default: `data/meetings`) +- `MKA_OLLAMA_ENDPOINT` (default: `http://127.0.0.1:11434`) +- `MKA_WHISPER_THREADS` (default: `auto`) +- `MKA_FFMPEG_EXECUTABLE` (default: `ffmpeg` found on `PATH`) +- `MKA_DIARIZATION_MODE` (`auto`, `cpu`, or `gpu`; default: `auto`) +- `MKA_DIARIZATION_RUNTIME` (`native` or `container`; default: `native`) +- `MKA_DIARIZATION_CONTAINER_IMAGE` (required for container diarization) + +From the Meeting Assistant repository, start the UI with: + +```bash +PYTHONPATH=src:../meeting-lab .venv/bin/streamlit run src/mka/ui/streamlit_app.py +``` + +Streamlit opens `http://localhost:8501` by default. Installing the sibling +Meeting Lab project supplies its runtime requirements such as PyYAML and +Requests. Optional diarization dependencies are needed only when diarization +is enabled. + +Uploaded source files are stored under +`data/meetings//uploads/`. Meeting Lab run artifacts are stored +under `data/meetings//runs/`. The reviewed protocol is saved as +`protocol_edited.md` inside its run directory; the generated `protocol.md` +remains unchanged. + +The upload remains in its original format and is passed unchanged to Meeting Lab. +Meeting Lab creates the canonical mono 16 kHz signed PCM16 WAV run artifact used by +transcription; Meeting Assistant does not duplicate audio conversion. See [Architecture](docs/architecture.md), [Project Knowledge](PROJECT_KNOWLEDGE.md), [Roadmap](ROADMAP.md) and [ADR 0011](docs/adr/0011-use-meeting-lab-mvp-backend.md). diff --git a/ROADMAP.md b/ROADMAP.md index da31cad..613ca1d 100644 --- a/ROADMAP.md +++ b/ROADMAP.md @@ -5,8 +5,10 @@ Build the user-facing application over the validated Meeting Lab Python API. - select an existing audio file +- reliably import WAV, FLAC and M4A through canonical FFmpeg preparation +- offer optional audio normalization, enabled by default - create and edit Meeting Context through structured fields -- add and manage participants +- add and manage present and `mentioned_only` people - optionally map `SPEAKER_XX` labels to participants with explicit confirmation - configure and start `run_mvp_meeting(...)` - display stage-based status and real progress when available @@ -18,6 +20,9 @@ The milestone must keep diarization, GPU acceleration and fixed-duration audio chunking optional. It must not launch the Meeting Lab CLI as a subprocess or duplicate Meeting Lab processing logic. +Complete real-meeting validation of this vertical slice before expanding the +post-run workflow. + ## Following Milestone: Distribution Protocol - derive or generate a shorter participant-facing version @@ -28,7 +33,12 @@ duplicate Meeting Lab processing logic. ## Later Product Work -- transcript viewing and correction workflows +- speaker/name review with explicit human confirmation +- deterministic correction of suitable derived artifacts, beginning with a + simple search-and-replace path +- protocol-only regeneration from existing transcription and diarization plus + corrected mappings and Meeting Context +- transcript viewing and broader correction workflows - recording and artifact lifecycle management - search, tags, projects and meeting history - richer Markdown, PDF and DOCX export @@ -42,6 +52,8 @@ duplicate Meeting Lab processing logic. - live transcription and real-time summaries - company glossary and custom vocabulary - voice identification with explicit consent and confirmation +- automatic name and speaker suggestions only as non-authoritative candidates + for later human review - audio cleanup - OCR for shared screens - RAG and knowledge-graph integration diff --git a/data/meetings/.gitkeep b/data/meetings/.gitkeep new file mode 100644 index 0000000..e69de29 diff --git a/docs/adr/0012-post-run-corrections.md b/docs/adr/0012-post-run-corrections.md new file mode 100644 index 0000000..e5e02b0 --- /dev/null +++ b/docs/adr/0012-post-run-corrections.md @@ -0,0 +1,67 @@ +# ADR 0012: Preserve Corrections as Post-Run Knowledge + +## Status + +Accepted as a future product direction; not required for the current MVP. + +## Context + +Real-meeting review exposes corrections that should not require repeating +expensive processing. These include misspelled or recurring name variants, +people mentioned but omitted from the initial Meeting Context, and confirmed +`SPEAKER_XX -> participant_id` mappings. A destructive edit would lose useful +provenance, while rerunning Whisper or diarization would add cost without +improving a deterministic correction. + +The product is also expected to provide both a detailed contextual protocol and +a short participant/distribution protocol. Both should eventually use the same +confirmed context and corrections. + +## Decision + +Original machine-generated artifacts remain immutable. Corrected or reviewed +artifacts are separate derivatives. Human corrections should evolve into +structured, meeting-specific knowledge rather than opaque destructive edits; +the correction schema is deliberately not defined by this ADR. + +An early correction tool may be simple deterministic search and replace, for +example `Grossman` to `Herr Grossmann`. Applying such a confirmed correction to +a suitable editable artifact requires neither Whisper, diarization nor an LLM +call. + +A later review stage may run after diarization or the complete initial run. It +may propose likely person/name matches, allow addition of previously omitted +mentioned people, and present anonymous speaker mappings for review. Every +suggestion is non-authoritative: anonymous labels remain anonymous until the +user explicitly confirms or corrects them, and `mentioned_only` people cannot +be mapped as speakers. + +After confirmation, the product should offer two paths: + +1. A fast path applies confirmed speaker, name or text corrections + deterministically where that is semantically safe. +2. A quality path regenerates protocol output using the existing transcription, + existing diarization, corrected Meeting Context and confirmed mappings or + corrections. It reruns protocol generation only. Whisper and diarization run + again only when separately requested or technically necessary. + +The likely later workflow is therefore: + +```text +Audio -> transcription -> optional diarization -> initial protocol + -> name/person/speaker review -> human confirmation + -> deterministic correction or protocol-only regeneration + -> final reviewed detailed and/or distribution protocol +``` + +The exact ordering may evolve, and this review stage is not mandatory for the +current Streamlit MVP. + +## Consequences + +- Human knowledge can improve outputs without unnecessary upstream work. +- Provenance is retained because originals and corrected derivatives coexist. +- Correction data can later be reused consistently across detailed and short + protocol views. +- Search/replace, correction storage, review UI, identity suggestions and + protocol-only rerun controls remain future implementation work. diff --git a/docs/architecture.md b/docs/architecture.md index a835dc3..9a9b48d 100644 --- a/docs/architecture.md +++ b/docs/architecture.md @@ -27,7 +27,7 @@ subprocess. ```text Source audio - -> FFmpeg normalization/preparation + -> FFmpeg preparation (normalization optional, default on) -> mono, 16 kHz PCM WAV -> whisper.cpp transcription with large-v3-turbo -> optional pyannote.audio Community-1 diarization @@ -85,7 +85,11 @@ this context and its mappings. Meeting Lab uses FFmpeg to prepare a consistent local-processing input. The current practical target is mono, 16 kHz PCM WAV. The imported source remains a -separate source artifact. +separate source artifact. Preparation always runs for WAV, FLAC and M4A, +regardless of the normalization switch. When enabled, Meeting Lab currently +uses `loudnorm=I=-16:LRA=11:TP=-1.5`, an isolated conservative default for +speech recordings that may be revisited after empirical comparison. Meeting +Assistant passes only an on/off choice and does not own filter parameters. ### Transcription @@ -137,6 +141,15 @@ and reviewed protocols are derived versions. Generated artifacts should retain their input version, backend/model configuration, prompt version and timestamp where practical. +## Later Post-Run Correction Flow + +Post-run corrections are planned as a separate, non-mandatory workflow after +initial protocol generation. As detailed in [ADR 0012](adr/0012-post-run-corrections.md), +confirmed name, person and anonymous-speaker corrections should become +meeting-specific knowledge. They may then be applied deterministically to safe +derived artifacts or used for protocol-only regeneration without needlessly +rerunning transcription or diarization. + ## Future Extensions - shorter distribution protocols diff --git a/pyproject.toml b/pyproject.toml index 3640b79..d3f68bb 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -12,7 +12,8 @@ authors = [ { name = "Martin Tazl" } ] dependencies = [ - "pydantic>=2,<3" + "pydantic>=2,<3", + "streamlit>=1.40,<2", ] [project.optional-dependencies] diff --git a/src/mka/application/__init__.py b/src/mka/application/__init__.py new file mode 100644 index 0000000..a744c48 --- /dev/null +++ b/src/mka/application/__init__.py @@ -0,0 +1,17 @@ +"""Application services for Meeting Assistant workflows.""" + +from mka.application.meeting_service import ( + MeetingDetails, + MeetingProcessingService, + ParticipantInput, + ProcessingOptions, + ProcessingOutcome, +) + +__all__ = [ + "MeetingDetails", + "MeetingProcessingService", + "ParticipantInput", + "ProcessingOptions", + "ProcessingOutcome", +] diff --git a/src/mka/application/config.py b/src/mka/application/config.py new file mode 100644 index 0000000..ce4bd9a --- /dev/null +++ b/src/mka/application/config.py @@ -0,0 +1,69 @@ +"""Environment-backed configuration for the Meeting Assistant application.""" + +from __future__ import annotations + +import os +from dataclasses import dataclass +from pathlib import Path + + +class ConfigurationError(ValueError): + """Raised when required processing configuration is unavailable.""" + + +@dataclass(frozen=True) +class AppSettings: + """Machine-specific Meeting Lab settings kept outside the UI.""" + + data_root: Path + whisper_model: Path | None + whisper_executable: str = "whisper-cli" + ffmpeg_executable: str = "ffmpeg" + protocol_model: str = "qwen3.8:27B" + ollama_endpoint: str = "http://127.0.0.1:11434" + language: str = "de" + threads: str | int = "auto" + diarization_mode: str = "auto" + diarization_runtime: str = "native" + diarization_container_image: str | None = None + + @classmethod + def from_environment(cls) -> AppSettings: + """Load settings from MKA_* environment variables.""" + whisper_model = os.getenv("MKA_WHISPER_MODEL") + threads = os.getenv("MKA_WHISPER_THREADS", "auto") + parsed_threads: str | int = int(threads) if threads.isdigit() else threads + return cls( + data_root=Path(os.getenv("MKA_DATA_ROOT", "data/meetings")), + whisper_model=Path(whisper_model) if whisper_model else None, + whisper_executable=os.getenv("MKA_WHISPER_EXECUTABLE", "whisper-cli"), + ffmpeg_executable=os.getenv("MKA_FFMPEG_EXECUTABLE", "ffmpeg"), + protocol_model=os.getenv("MKA_PROTOCOL_MODEL", "qwen3.8:27B"), + ollama_endpoint=os.getenv("MKA_OLLAMA_ENDPOINT", "http://127.0.0.1:11434"), + language=os.getenv("MKA_LANGUAGE", "de"), + threads=parsed_threads, + diarization_mode=os.getenv("MKA_DIARIZATION_MODE", "auto"), + diarization_runtime=os.getenv("MKA_DIARIZATION_RUNTIME", "native"), + diarization_container_image=os.getenv("MKA_DIARIZATION_CONTAINER_IMAGE"), + ) + + def validate_for_processing(self) -> None: + """Validate settings needed before a Meeting Lab run starts.""" + if self.whisper_model is None: + raise ConfigurationError("MKA_WHISPER_MODEL is not configured.") + if not self.whisper_model.is_file(): + raise ConfigurationError( + f"Configured Whisper model does not exist: {self.whisper_model}" + ) + if not self.whisper_executable.strip(): + raise ConfigurationError("MKA_WHISPER_EXECUTABLE must not be empty.") + if not self.ffmpeg_executable.strip(): + raise ConfigurationError("MKA_FFMPEG_EXECUTABLE must not be empty.") + if self.diarization_mode not in {"auto", "cpu", "gpu"}: + raise ConfigurationError("MKA_DIARIZATION_MODE must be one of: auto, cpu, gpu.") + if self.diarization_runtime not in {"native", "container"}: + raise ConfigurationError("MKA_DIARIZATION_RUNTIME must be native or container.") + if self.diarization_runtime == "container" and not self.diarization_container_image: + raise ConfigurationError( + "MKA_DIARIZATION_CONTAINER_IMAGE is required for container runtime." + ) diff --git a/src/mka/application/meeting_service.py b/src/mka/application/meeting_service.py new file mode 100644 index 0000000..4309807 --- /dev/null +++ b/src/mka/application/meeting_service.py @@ -0,0 +1,267 @@ +"""Use-case service for processing one uploaded meeting recording.""" + +from __future__ import annotations + +import json +import re +import shutil +from collections.abc import Callable +from dataclasses import dataclass +from datetime import date +from pathlib import Path +from typing import Any, Protocol +from uuid import uuid4 + +from mka.application.config import AppSettings + +STAGES = ("preparing", "transcription", "diarization", "protocol_generation") + + +class MeetingLabPort(Protocol): + """Operations the application requires from Meeting Lab.""" + + def create_context(self, data: dict[str, Any]) -> Any: ... + + def create_config(self, values: dict[str, Any]) -> Any: ... + + def run( + self, + config: Any, + meeting_context: Any, + progress_sink: Callable[[Any], None], + ) -> Any: ... + + +@dataclass(frozen=True) +class MeetingDetails: + title: str + language: str = "de" + meeting_date: date | None = None + description: str = "" + meeting_id: str | None = None + + +@dataclass(frozen=True) +class ParticipantInput: + participant_id: str + display_name: str + role: str = "" + organization: str = "" + attendance_status: str = "present" + + +@dataclass(frozen=True) +class ProcessingOptions: + diarization_enabled: bool = False + audio_normalization: bool = True + + +@dataclass(frozen=True) +class AppProgressEvent: + stage: str + status: str + elapsed_seconds: float + progress: float | None = None + message: str | None = None + + +@dataclass(frozen=True) +class ProcessingOutcome: + succeeded: bool + run_dir: Path | None + original_protocol: str | None + protocol_path: Path | None + failed_stage: str | None = None + error_message: str | None = None + + +def stable_id(value: str) -> str: + """Return a schema-safe ID based on user text, with a random fallback.""" + normalized = re.sub(r"[^a-z0-9]+", "-", value.lower()).strip("-") + return normalized or f"participant-{uuid4().hex[:8]}" + + +class MeetingProcessingService: + """Translate UI input into Meeting Lab calls and persisted artifacts.""" + + def __init__(self, settings: AppSettings, meeting_lab: MeetingLabPort) -> None: + self.settings = settings + self.meeting_lab = meeting_lab + + def build_context( + self, + meeting: MeetingDetails, + participants: list[ParticipantInput], + speaker_mappings: dict[str, str] | None = None, + ) -> Any: + """Build and validate Meeting Context V1 through Meeting Lab.""" + meeting_id = meeting.meeting_id or stable_id(meeting.title) + organizations = sorted( + {item.organization.strip() for item in participants if item.organization.strip()} + ) + departments = [ + {"id": stable_id(name), "name": name, "aliases": []} for name in organizations + ] + department_ids = {item["name"]: item["id"] for item in departments} + data = { + "schema_version": "1", + "meeting": { + "meeting_id": meeting_id, + "title": meeting.title.strip(), + "language": meeting.language, + "date": meeting.meeting_date.isoformat() if meeting.meeting_date else None, + "objective": "", + "notes": meeting.description.strip(), + }, + "participants": [ + { + "participant_id": item.participant_id.strip(), + "display_name": item.display_name.strip(), + "aliases": [], + "role": item.role.strip() or None, + "department": department_ids.get(item.organization.strip()), + "attendance_status": "present", + "notes": None, + } + for item in participants + if item.attendance_status == "present" + ], + "speaker_mappings": dict(speaker_mappings or {}), + "mentioned_people": [ + { + "person_id": item.participant_id.strip(), + "display_name": item.display_name.strip(), + "aliases": [], + "role": item.role.strip() or None, + "department": department_ids.get(item.organization.strip()), + "attendance_status": "mentioned_only", + "notes": None, + } + for item in participants + if item.attendance_status == "mentioned_only" + ], + "organization": { + "name": None, + "departments": departments, + "abbreviations": {}, + }, + "known_entities": {}, + "context_rules": { + "participant_list_is_authoritative": True, + "do_not_infer_roles": True, + "do_not_infer_departments": True, + "do_not_infer_responsibilities": True, + "mentioned_people_are_not_participants": True, + }, + } + return self.meeting_lab.create_context(data) + + def preserve_upload(self, meeting_id: str, filename: str, source: Any) -> Path: + """Persist an uploaded source before processing starts.""" + safe_filename = Path(filename).name + upload_dir = self.settings.data_root / stable_id(meeting_id) / "uploads" + upload_dir.mkdir(parents=True, exist_ok=True) + destination = upload_dir / f"{uuid4().hex[:8]}_{safe_filename}" + with destination.open("wb") as target: + if hasattr(source, "getbuffer"): + target.write(source.getbuffer()) + else: + shutil.copyfileobj(source, target) + return destination + + def process( + self, + audio_file: Path, + meeting: MeetingDetails, + participants: list[ParticipantInput], + options: ProcessingOptions, + progress_sink: Callable[[AppProgressEvent], None] | None = None, + speaker_mappings: dict[str, str] | None = None, + ) -> ProcessingOutcome: + """Validate input, run Meeting Lab and expose a UI-oriented result.""" + self.settings.validate_for_processing() + context = self.build_context(meeting, participants, speaker_mappings) + meeting_id = context.meeting_id + output_root = self.settings.data_root / stable_id(meeting_id) / "runs" + config = self.meeting_lab.create_config( + { + "audio_file": audio_file, + "whisper_model": self.settings.whisper_model, + "whisper_executable": self.settings.whisper_executable, + "ffmpeg_executable": self.settings.ffmpeg_executable, + "audio_normalization": options.audio_normalization, + "output_root": output_root, + "language": meeting.language or self.settings.language, + "threads": self.settings.threads, + "model": self.settings.protocol_model, + "ollama_endpoint": self.settings.ollama_endpoint, + "diarization": ( + self.settings.diarization_mode if options.diarization_enabled else "off" + ), + "diarization_runtime": self.settings.diarization_runtime, + "diarization_container_image": (self.settings.diarization_container_image), + } + ) + current_stage: str | None = None + + def relay(event: Any) -> None: + nonlocal current_stage + if event.stage in STAGES and event.status == "started": + current_stage = event.stage + app_event = AppProgressEvent( + stage=event.stage, + status=event.status, + elapsed_seconds=event.elapsed_seconds, + progress=event.progress, + message=event.message, + ) + if progress_sink is not None: + progress_sink(app_event) + + result = self.meeting_lab.run(config, context, relay) + if result.exit_code != 0: + failure = self._read_failure(result.run_dir) + backend_stage = failure.get("stage") + failed_stage = { + "validation": "preparing", + "setup": "preparing", + "audio_preparation": "preparing", + "whisper": "transcription", + "transcript_validation": "transcription", + "protocol": "protocol_generation", + }.get(backend_stage, backend_stage) + return ProcessingOutcome( + succeeded=False, + run_dir=result.run_dir, + original_protocol=None, + protocol_path=None, + failed_stage=failed_stage or current_stage or "preparing", + error_message=failure.get("message") or "Meeting Lab processing failed.", + ) + protocol_path = Path(result.protocol_path) + return ProcessingOutcome( + succeeded=True, + run_dir=result.run_dir, + original_protocol=protocol_path.read_text(encoding="utf-8"), + protocol_path=protocol_path, + ) + + @staticmethod + def _read_failure(run_dir: Path | None) -> dict[str, str]: + if run_dir is None: + return {} + metadata_path = Path(run_dir) / "run_metadata.json" + if not metadata_path.is_file(): + return {} + try: + failure = json.loads(metadata_path.read_text(encoding="utf-8")).get("failure") + except (OSError, json.JSONDecodeError): + return {} + return failure if isinstance(failure, dict) else {} + + @staticmethod + def save_edited_protocol(run_dir: Path, text: str) -> Path: + """Persist user edits beside, never over, the generated protocol.""" + destination = Path(run_dir) / "protocol_edited.md" + destination.write_text(text, encoding="utf-8") + return destination diff --git a/src/mka/integrations/__init__.py b/src/mka/integrations/__init__.py new file mode 100644 index 0000000..7d57ce8 --- /dev/null +++ b/src/mka/integrations/__init__.py @@ -0,0 +1 @@ +"""Adapters for external processing systems.""" diff --git a/src/mka/integrations/meeting_lab.py b/src/mka/integrations/meeting_lab.py new file mode 100644 index 0000000..498f10c --- /dev/null +++ b/src/mka/integrations/meeting_lab.py @@ -0,0 +1,51 @@ +"""Narrow adapter around the reusable Meeting Lab Python API.""" + +from __future__ import annotations + +from collections.abc import Callable, Mapping +from typing import Any + + +class MeetingLabUnavailableError(RuntimeError): + """Raised when the Meeting Lab package cannot be imported.""" + + +class MeetingLabGateway: + """Load and delegate to Meeting Lab without coupling Streamlit to it.""" + + def __init__(self) -> None: + try: + from src.meeting_lab.models.meeting_context import create_meeting_context + from src.meeting_lab.orchestration.mvp import ( + MvpMeetingConfig, + run_mvp_meeting, + ) + except ImportError as exc: + raise MeetingLabUnavailableError( + "Meeting Lab is unavailable. Add the Meeting Lab repository root " + "to PYTHONPATH as described in README.md." + ) from exc + self._config_type = MvpMeetingConfig + self._create_context = create_meeting_context + self._run_mvp_meeting = run_mvp_meeting + + def create_context(self, data: dict[str, Any]) -> Any: + """Validate structured context through Meeting Lab's domain boundary.""" + return self._create_context(data) + + def create_config(self, values: Mapping[str, Any]) -> Any: + """Create the concrete Meeting Lab run configuration.""" + return self._config_type(**values) + + def run( + self, + config: Any, + meeting_context: Any, + progress_sink: Callable[[Any], None], + ) -> Any: + """Run Meeting Lab synchronously and relay progress callbacks.""" + return self._run_mvp_meeting( + config, + meeting_context=meeting_context, + progress_sink=progress_sink, + ) diff --git a/src/mka/models/base.py b/src/mka/models/base.py index 32d1710..057870b 100644 --- a/src/mka/models/base.py +++ b/src/mka/models/base.py @@ -22,4 +22,4 @@ class DomainModel(BaseModel): id: UUID = Field(default_factory=uuid4) created_at: datetime = Field(default_factory=utc_now) - updated_at: datetime = Field(default_factory=utc_now) \ No newline at end of file + updated_at: datetime = Field(default_factory=utc_now) diff --git a/src/mka/ui/streamlit_app.py b/src/mka/ui/streamlit_app.py new file mode 100644 index 0000000..3d21830 --- /dev/null +++ b/src/mka/ui/streamlit_app.py @@ -0,0 +1,253 @@ +"""Streamlit presentation layer for the first Meeting Assistant MVP.""" + +from __future__ import annotations + +from datetime import date +from typing import Any +from uuid import uuid4 + +import streamlit as st + +from mka.application.config import AppSettings, ConfigurationError +from mka.application.meeting_service import ( + STAGES, + AppProgressEvent, + MeetingDetails, + MeetingProcessingService, + ParticipantInput, + ProcessingOptions, + stable_id, +) +from mka.integrations.meeting_lab import ( + MeetingLabGateway, + MeetingLabUnavailableError, +) + +STAGE_LABELS = { + "preparing": "Preparation", + "transcription": "Transcription", + "diarization": "Diarization", + "protocol_generation": "Protocol generation", +} + + +def _new_participant() -> dict[str, str]: + return { + "row_id": uuid4().hex, + "participant_id": "", + "display_name": "", + "role": "", + "organization": "", + "attendance_status": "present", + } + + +def _initialize_state() -> None: + st.session_state.setdefault("participants", [_new_participant()]) + st.session_state.setdefault("outcome", None) + st.session_state.setdefault("edited_protocol", "") + + +def _render_participants() -> list[ParticipantInput]: + st.subheader("People") + st.caption("Record whether each relevant person attended or was only mentioned.") + rows = st.session_state.participants + remove_index: int | None = None + for index, row in enumerate(rows): + row_id = row["row_id"] + columns = st.columns([2, 2, 2, 2, 2, 0.6]) + row["display_name"] = columns[0].text_input( + "Name", value=row["display_name"], key=f"name_{row_id}" + ) + suggested_id = row["participant_id"] or stable_id(row["display_name"]) + row["participant_id"] = columns[1].text_input( + "Participant ID", value=suggested_id, key=f"id_{row_id}" + ) + row["role"] = columns[2].text_input("Role", value=row["role"], key=f"role_{row_id}") + row["organization"] = columns[3].text_input( + "Organization / department", + value=row["organization"], + key=f"organization_{row_id}", + ) + row["attendance_status"] = columns[4].selectbox( + "Attendance", + options=["present", "mentioned_only"], + format_func=lambda value: { + "present": "Present / participated", + "mentioned_only": "Mentioned, but not present", + }[value], + index=0 if row["attendance_status"] == "present" else 1, + key=f"attendance_{row_id}", + ) + if columns[5].button("Remove", key=f"remove_{row_id}"): + remove_index = index + if remove_index is not None: + rows.pop(remove_index) + st.rerun() + if st.button("Add person"): + rows.append(_new_participant()) + st.rerun() + return [ + ParticipantInput( + participant_id=row["participant_id"], + display_name=row["display_name"], + role=row["role"], + organization=row["organization"], + attendance_status=row["attendance_status"], + ) + for row in rows + ] + + +def _progress_callback( + status_box: Any, + stage_table: Any, + progress_slot: Any, + states: dict[str, str], +) -> Any: + progress_bar = None + + def update(event: AppProgressEvent) -> None: + nonlocal progress_bar + if event.stage in states: + states[event.stage] = "running" if event.status == "started" else event.status + if event.stage == "failed": + running = next( + (stage for stage, status in states.items() if status == "running"), + None, + ) + if running: + states[running] = "failed" + elapsed = f"{event.elapsed_seconds:.1f} s" + message = event.message or STAGE_LABELS.get(event.stage, event.stage) + status_box.info(f"{message} — elapsed {elapsed}") + stage_table.table( + [{"Stage": STAGE_LABELS[stage], "Status": states[stage]} for stage in STAGES] + ) + if event.progress is not None: + if progress_bar is None: + progress_bar = progress_slot.progress(0.0) + progress_bar.progress( + min(max(event.progress, 0.0), 1.0), + text=f"{STAGE_LABELS.get(event.stage, event.stage)}: {event.progress:.0%}", + ) + + return update + + +def _render_result() -> None: + outcome = st.session_state.outcome + if outcome is None: + return + st.divider() + st.header("Protocol result") + if not outcome.succeeded: + st.error(f"Processing failed during {outcome.failed_stage}: {outcome.error_message}") + if outcome.run_dir: + st.code(str(outcome.run_dir)) + st.caption("Intermediate artifacts and run metadata were preserved here.") + return + + st.success("Processing completed. Review the generated protocol before use.") + st.caption(f"Run artifacts: {outcome.run_dir}") + with st.expander("Original generated protocol", expanded=False): + st.markdown(outcome.original_protocol or "") + edited = st.text_area( + "Editable protocol", + key="edited_protocol", + height=500, + help="The original protocol.md remains unchanged.", + ) + if st.button("Save edited protocol", type="primary"): + path = MeetingProcessingService.save_edited_protocol(outcome.run_dir, edited) + st.success(f"Saved edited protocol to {path}") + + +def main() -> None: + st.set_page_config(page_title="Meeting Assistant", layout="wide") + _initialize_state() + st.title("Meeting Assistant") + st.caption("Create meeting context, run Meeting Lab, and review the protocol.") + + st.header("Meeting and audio") + audio = st.file_uploader("Audio recording", type=["wav", "flac", "m4a"]) + title = st.text_input("Meeting title") + description = st.text_area("Description / context", height=100) + metadata_columns = st.columns(3) + language = metadata_columns[0].selectbox("Meeting language", options=["de", "en"]) + has_date = metadata_columns[1].checkbox("Meeting date is known", value=True) + selected_date: date | None = ( + metadata_columns[2].date_input("Meeting date") if has_date else None + ) + + participants = _render_participants() + + st.header("Processing options") + audio_normalization = st.checkbox( + "Audio normalization", + value=True, + help=( + "Normalize speech loudness during preparation. WAV, FLAC, and M4A " + "are always converted to the canonical Meeting Lab audio format, " + "regardless of this setting." + ), + ) + diarization_enabled = st.checkbox( + "Enable speaker diarization", + help="Speaker labels remain anonymous; identities are never inferred.", + ) + + if st.button("Start processing", type="primary", disabled=audio is None): + if not title.strip(): + st.error("Meeting title is required.") + elif any( + not item.display_name.strip() or not item.participant_id.strip() + for item in participants + ): + st.error("Every person row needs a name and participant ID.") + else: + settings = AppSettings.from_environment() + try: + service = MeetingProcessingService(settings, MeetingLabGateway()) + meeting = MeetingDetails( + title=title, + language=language, + meeting_date=selected_date, + description=description, + ) + meeting_id = stable_id(title) + audio_path = service.preserve_upload(meeting_id, audio.name, audio) + st.header("Processing status") + status_box = st.empty() + stage_table = st.empty() + progress_slot = st.empty() + states = {stage: "pending" for stage in STAGES} + if not diarization_enabled: + states["diarization"] = "skipped" + callback = _progress_callback(status_box, stage_table, progress_slot, states) + outcome = service.process( + audio_path, + meeting, + participants, + ProcessingOptions( + diarization_enabled=diarization_enabled, + audio_normalization=audio_normalization, + ), + progress_sink=callback, + ) + st.session_state.outcome = outcome + st.session_state.edited_protocol = outcome.original_protocol or "" + if outcome.succeeded: + status_box.success("Processing completed.") + else: + status_box.error("Processing failed. Artifacts were preserved.") + except (ConfigurationError, MeetingLabUnavailableError, ValueError) as exc: + st.error(str(exc)) + except Exception as exc: + st.error(f"Unable to process meeting: {exc}") + + _render_result() + + +if __name__ == "__main__": + main() diff --git a/tests/test_application_config.py b/tests/test_application_config.py new file mode 100644 index 0000000..bc41b9f --- /dev/null +++ b/tests/test_application_config.py @@ -0,0 +1,30 @@ +from pathlib import Path + +import pytest + +from mka.application.config import AppSettings, ConfigurationError + + +def test_settings_require_whisper_model() -> None: + settings = AppSettings(data_root=Path("runs"), whisper_model=None) + + with pytest.raises(ConfigurationError, match="MKA_WHISPER_MODEL"): + settings.validate_for_processing() + + +def test_settings_reject_missing_whisper_model(tmp_path: Path) -> None: + settings = AppSettings( + data_root=tmp_path / "runs", + whisper_model=tmp_path / "missing.bin", + ) + + with pytest.raises(ConfigurationError, match="does not exist"): + settings.validate_for_processing() + + +def test_settings_accept_valid_native_configuration(tmp_path: Path) -> None: + model = tmp_path / "model.bin" + model.write_bytes(b"model") + settings = AppSettings(data_root=tmp_path / "runs", whisper_model=model) + + settings.validate_for_processing() diff --git a/tests/test_meeting_service.py b/tests/test_meeting_service.py new file mode 100644 index 0000000..17eac59 --- /dev/null +++ b/tests/test_meeting_service.py @@ -0,0 +1,299 @@ +import json +from dataclasses import dataclass +from datetime import date +from pathlib import Path +from types import SimpleNamespace +from typing import Any + +from mka.application.config import AppSettings +from mka.application.meeting_service import ( + MeetingDetails, + MeetingProcessingService, + ParticipantInput, + ProcessingOptions, +) + + +@dataclass +class FakeContext: + data: dict[str, Any] + + @property + def meeting_id(self) -> str: + return self.data["meeting"]["meeting_id"] + + +class FakeMeetingLab: + def __init__(self, run_dir: Path) -> None: + self.run_dir = run_dir + self.context_data: dict[str, Any] | None = None + self.config_values: dict[str, Any] | None = None + self.fail = False + + def create_context(self, data: dict[str, Any]) -> FakeContext: + self.context_data = data + if not data["meeting"]["title"]: + raise ValueError("meeting.title must be present and non-empty") + participant_ids = [item["participant_id"] for item in data["participants"]] + if len(participant_ids) != len(set(participant_ids)): + raise ValueError("Duplicate participant_id") + return FakeContext(data) + + def create_config(self, values: dict[str, Any]) -> dict[str, Any]: + self.config_values = values + return values + + def run(self, config: Any, meeting_context: Any, progress_sink: Any) -> Any: + self.run_dir.mkdir(parents=True, exist_ok=True) + progress_sink( + SimpleNamespace( + stage="preparing", + status="started", + elapsed_seconds=0.1, + progress=None, + message=None, + ) + ) + progress_sink( + SimpleNamespace( + stage="preparing", + status="completed", + elapsed_seconds=0.2, + progress=None, + message=None, + ) + ) + progress_sink( + SimpleNamespace( + stage="transcription", + status="started", + elapsed_seconds=0.2, + progress=0.25, + message="transcribing", + ) + ) + if self.fail: + (self.run_dir / "run_metadata.json").write_text( + json.dumps({"failure": {"stage": "whisper", "message": "model failed"}}), + encoding="utf-8", + ) + progress_sink( + SimpleNamespace( + stage="failed", + status="failed", + elapsed_seconds=0.3, + progress=None, + message="transcription: model failed", + ) + ) + return SimpleNamespace(exit_code=2, run_dir=self.run_dir, protocol_path=None) + protocol = self.run_dir / "protocol.md" + protocol.write_text("# Generated protocol\n", encoding="utf-8") + return SimpleNamespace(exit_code=0, run_dir=self.run_dir, protocol_path=protocol) + + +def make_service(tmp_path: Path) -> tuple[MeetingProcessingService, FakeMeetingLab]: + model = tmp_path / "model.bin" + model.write_bytes(b"model") + gateway = FakeMeetingLab(tmp_path / "backend-run") + settings = AppSettings( + data_root=tmp_path / "meetings", + whisper_model=model, + whisper_executable="/opt/whisper-cli", + protocol_model="test:model", + diarization_mode="gpu", + ) + return MeetingProcessingService(settings, gateway), gateway + + +def meeting() -> MeetingDetails: + return MeetingDetails( + title="Architecture Review", + language="de", + meeting_date=date(2026, 8, 23), + description="Review the MVP.", + ) + + +def participants() -> list[ParticipantInput]: + return [ + ParticipantInput( + participant_id="martin", + display_name="Martin", + role="Project lead", + organization="Engineering", + ), + ParticipantInput( + participant_id="alex", + display_name="Alex", + organization="Engineering", + ), + ] + + +def test_build_context_uses_actual_v1_shape(tmp_path: Path) -> None: + service, gateway = make_service(tmp_path) + + context = service.build_context(meeting(), participants()) + + assert context.meeting_id == "architecture-review" + assert gateway.context_data is not None + assert gateway.context_data["meeting"]["date"] == "2026-08-23" + assert gateway.context_data["meeting"]["notes"] == "Review the MVP." + assert gateway.context_data["participants"][0] == { + "participant_id": "martin", + "display_name": "Martin", + "aliases": [], + "role": "Project lead", + "department": "engineering", + "attendance_status": "present", + "notes": None, + } + assert gateway.context_data["organization"]["departments"] == [ + {"id": "engineering", "name": "Engineering", "aliases": []} + ] + + +def test_build_context_preserves_explicit_speaker_mapping(tmp_path: Path) -> None: + service, gateway = make_service(tmp_path) + + service.build_context(meeting(), participants(), {"SPEAKER_00": "martin"}) + + assert gateway.context_data is not None + assert gateway.context_data["speaker_mappings"] == {"SPEAKER_00": "martin"} + + +def test_build_context_translates_mentioned_only_person(tmp_path: Path) -> None: + service, gateway = make_service(tmp_path) + people = participants() + [ + ParticipantInput( + participant_id="sam", + display_name="Sam", + attendance_status="mentioned_only", + ) + ] + + service.build_context(meeting(), people) + + assert gateway.context_data is not None + assert [item["participant_id"] for item in gateway.context_data["participants"]] == [ + "martin", + "alex", + ] + assert gateway.context_data["mentioned_people"] == [ + { + "person_id": "sam", + "display_name": "Sam", + "aliases": [], + "role": None, + "department": None, + "attendance_status": "mentioned_only", + "notes": None, + } + ] + + +def test_process_translates_configuration_and_disables_diarization( + tmp_path: Path, +) -> None: + service, gateway = make_service(tmp_path) + audio = tmp_path / "meeting.wav" + audio.write_bytes(b"audio") + + outcome = service.process( + audio, meeting(), participants(), ProcessingOptions(diarization_enabled=False) + ) + + assert outcome.succeeded + assert gateway.config_values is not None + assert gateway.config_values["diarization"] == "off" + assert gateway.config_values["model"] == "test:model" + assert gateway.config_values["whisper_executable"] == "/opt/whisper-cli" + assert gateway.config_values["ffmpeg_executable"] == "ffmpeg" + assert gateway.config_values["audio_normalization"] is True + assert gateway.config_values["output_root"] == ( + tmp_path / "meetings" / "architecture-review" / "runs" + ) + + +def test_process_propagates_disabled_audio_normalization(tmp_path: Path) -> None: + service, gateway = make_service(tmp_path) + audio = tmp_path / "meeting.m4a" + audio.write_bytes(b"audio") + + outcome = service.process( + audio, + meeting(), + participants(), + ProcessingOptions(audio_normalization=False), + ) + + assert outcome.succeeded + assert gateway.config_values is not None + assert gateway.config_values["audio_normalization"] is False + + +def test_processing_options_default_to_audio_normalization_on() -> None: + assert ProcessingOptions().audio_normalization is True + + +def test_process_propagates_enabled_diarization_and_progress(tmp_path: Path) -> None: + service, gateway = make_service(tmp_path) + audio = tmp_path / "meeting.flac" + audio.write_bytes(b"audio") + events = [] + + service.process( + audio, + meeting(), + participants(), + ProcessingOptions(diarization_enabled=True), + progress_sink=events.append, + ) + + assert gateway.config_values is not None + assert gateway.config_values["diarization"] == "gpu" + assert [(event.stage, event.status) for event in events[:2]] == [ + ("preparing", "started"), + ("preparing", "completed"), + ] + assert events[2].progress == 0.25 + + +def test_result_and_user_edit_are_preserved_separately(tmp_path: Path) -> None: + service, _ = make_service(tmp_path) + audio = tmp_path / "meeting.wav" + audio.write_bytes(b"audio") + + outcome = service.process(audio, meeting(), participants(), ProcessingOptions()) + edited_path = service.save_edited_protocol(outcome.run_dir, "# Reviewed\n") + + assert outcome.original_protocol == "# Generated protocol\n" + assert outcome.protocol_path.read_text(encoding="utf-8") == "# Generated protocol\n" + assert edited_path.read_text(encoding="utf-8") == "# Reviewed\n" + + +def test_failure_reports_stage_and_preserves_run_dir(tmp_path: Path) -> None: + service, gateway = make_service(tmp_path) + gateway.fail = True + audio = tmp_path / "meeting.wav" + audio.write_bytes(b"audio") + + outcome = service.process(audio, meeting(), participants(), ProcessingOptions()) + + assert not outcome.succeeded + assert outcome.failed_stage == "transcription" + assert outcome.error_message == "model failed" + assert outcome.run_dir == gateway.run_dir + assert (gateway.run_dir / "run_metadata.json").is_file() + + +def test_uploaded_source_is_preserved_in_meeting_directory(tmp_path: Path) -> None: + service, _ = make_service(tmp_path) + source = SimpleNamespace(getbuffer=lambda: b"source audio") + + destination = service.preserve_upload("meeting-1", "../unsafe.wav", source) + + assert destination.parent == tmp_path / "meetings" / "meeting-1" / "uploads" + assert destination.name.endswith("_unsafe.wav") + assert destination.read_bytes() == b"source audio"