Files
meeting-assistant/src/mka/application/run_inputs.py
T

219 lines
8.2 KiB
Python

"""Versioned JSON import/export for the user-facing run input form."""
from __future__ import annotations
import json
from collections.abc import Sequence
from dataclasses import dataclass
from datetime import date
from pathlib import Path
from typing import Any
from mka.application.meeting_service import ParticipantInput, stable_id
from mka.application.people_yaml import ATTENDANCE_VALUES, PARTICIPANT_ID_PATTERN
RUN_INPUT_SCHEMA_VERSION = 1
SUPPORTED_LANGUAGES = frozenset({"de", "en"})
class RunInputJsonError(ValueError):
"""Raised when a run-input JSON document is malformed or unsupported."""
@dataclass(frozen=True)
class RunInputState:
"""All user-configurable values needed before starting a processing run."""
title: str
description: str
language: str
meeting_date: date | None
participants: tuple[ParticipantInput, ...]
audio_normalization: bool
diarization_enabled: bool
source_file_name: str | None = None
@classmethod
def defaults(cls) -> RunInputState:
return cls(
title="",
description="",
language="de",
meeting_date=date.today(),
participants=(ParticipantInput(participant_id="", display_name=""),),
audio_normalization=True,
diarization_enabled=False,
)
def export_run_inputs(state: RunInputState) -> str:
"""Serialize form state as deterministic, human-readable JSON."""
source_file_name = _source_file_name(state.source_file_name)
document = {
"schema_version": RUN_INPUT_SCHEMA_VERSION,
"meeting": {
"title": state.title,
"description": state.description,
"language": state.language,
"date": state.meeting_date.isoformat() if state.meeting_date else None,
"participants": [
{
"participant_id": person.participant_id,
"display_name": person.display_name,
"role": person.role,
"organization": person.organization,
"attendance_status": person.attendance_status,
}
for person in state.participants
],
},
"processing": {
"audio_normalization": state.audio_normalization,
"diarization_enabled": state.diarization_enabled,
},
"source_file_name": source_file_name,
}
return json.dumps(document, ensure_ascii=False, indent=2) + "\n"
def import_run_inputs(content: str | bytes) -> RunInputState:
"""Parse supported fields, applying current defaults to omitted optional fields."""
try:
if isinstance(content, bytes):
content = content.decode("utf-8")
document = json.loads(content)
except (UnicodeDecodeError, json.JSONDecodeError) as exc:
raise RunInputJsonError(f"Malformed input JSON: {exc}") from exc
if not isinstance(document, dict):
raise RunInputJsonError("Input JSON must contain a top-level object.")
version = document.get("schema_version")
if type(version) is not int or version != RUN_INPUT_SCHEMA_VERSION:
raise RunInputJsonError(
f"Unsupported input schema_version {version!r}; expected {RUN_INPUT_SCHEMA_VERSION}."
)
defaults = RunInputState.defaults()
meeting = _optional_mapping(document, "meeting")
processing = _optional_mapping(document, "processing")
title = _optional_text(meeting, "title", defaults.title)
description = _optional_text(meeting, "description", defaults.description)
language = _optional_text(meeting, "language", defaults.language)
if language not in SUPPORTED_LANGUAGES:
raise RunInputJsonError(f"Unsupported meeting language: {language!r}.")
meeting_date = _meeting_date(meeting, defaults.meeting_date)
participants = _participants(meeting, defaults.participants)
audio_normalization = _optional_bool(
processing, "audio_normalization", defaults.audio_normalization
)
diarization_enabled = _optional_bool(
processing, "diarization_enabled", defaults.diarization_enabled
)
source_file_name = _source_file_name(document.get("source_file_name"))
return RunInputState(
title=title,
description=description,
language=language,
meeting_date=meeting_date,
participants=participants,
audio_normalization=audio_normalization,
diarization_enabled=diarization_enabled,
source_file_name=source_file_name,
)
def run_input_filename(title: str) -> str:
"""Build a stable download filename without filesystem-specific characters."""
suffix = stable_id(title) if title.strip() else "untitled"
return f"meeting-inputs-{suffix}.json"
def _optional_mapping(document: dict[str, Any], field: str) -> dict[str, Any]:
value = document.get(field, {})
if not isinstance(value, dict):
raise RunInputJsonError(f"Field {field!r} must be an object.")
return value
def _optional_text(document: dict[str, Any], field: str, default: str) -> str:
value = document.get(field, default)
if not isinstance(value, str):
raise RunInputJsonError(f"Field {field!r} must be text.")
return value
def _optional_bool(document: dict[str, Any], field: str, default: bool) -> bool:
value = document.get(field, default)
if type(value) is not bool:
raise RunInputJsonError(f"Field {field!r} must be true or false.")
return value
def _meeting_date(document: dict[str, Any], default: date | None) -> date | None:
if "date" not in document:
return default
value = document["date"]
if value is None:
return None
if not isinstance(value, str):
raise RunInputJsonError("Field 'date' must be an ISO date or null.")
try:
return date.fromisoformat(value)
except ValueError as exc:
raise RunInputJsonError("Field 'date' must be a valid ISO date or null.") from exc
def _participants(
document: dict[str, Any], default: Sequence[ParticipantInput]
) -> tuple[ParticipantInput, ...]:
if "participants" not in document:
return tuple(default)
entries = document["participants"]
if not isinstance(entries, list):
raise RunInputJsonError("Field 'participants' must be a list.")
people: list[ParticipantInput] = []
seen_ids: set[str] = set()
for index, entry in enumerate(entries, start=1):
if not isinstance(entry, dict):
raise RunInputJsonError(f"Participant {index} must be an object.")
participant_id = _optional_text(entry, "participant_id", "")
display_name = _optional_text(entry, "display_name", "")
role = _optional_text(entry, "role", "")
organization = _optional_text(entry, "organization", "")
attendance = _optional_text(entry, "attendance_status", "present")
if bool(participant_id) != bool(display_name):
raise RunInputJsonError(
f"Participant {index} must provide both participant_id and display_name."
)
if participant_id and PARTICIPANT_ID_PATTERN.fullmatch(participant_id) is None:
raise RunInputJsonError(
f"Participant {index} has invalid participant_id {participant_id!r}."
)
if participant_id in seen_ids:
raise RunInputJsonError(f"Duplicate participant_id: {participant_id!r}.")
if participant_id:
seen_ids.add(participant_id)
if attendance not in ATTENDANCE_VALUES:
raise RunInputJsonError(
f"Participant {index} has invalid attendance_status {attendance!r}."
)
people.append(
ParticipantInput(
participant_id=participant_id,
display_name=display_name,
role=role,
organization=organization,
attendance_status=attendance,
)
)
return tuple(people)
def _source_file_name(value: Any) -> str | None:
if value is None:
return None
if not isinstance(value, str) or not value.strip():
raise RunInputJsonError("Field 'source_file_name' must be non-empty text or null.")
if Path(value).name != value:
raise RunInputJsonError("Field 'source_file_name' must be a filename, not a path.")
return value