152 lines
5.4 KiB
Python
152 lines
5.4 KiB
Python
#!/usr/bin/env python3
|
|
"""CLI adapter for the reusable Meeting Lab MVP orchestration API."""
|
|
|
|
from __future__ import annotations
|
|
|
|
import argparse
|
|
import json
|
|
import sys
|
|
from pathlib import Path
|
|
from typing import Any
|
|
|
|
REPO_ROOT = Path(__file__).resolve().parents[1]
|
|
if str(REPO_ROOT) not in sys.path:
|
|
sys.path.insert(0, str(REPO_ROOT))
|
|
|
|
from src.meeting_lab.llm.ollama import DEFAULT_ENDPOINT # noqa: E402
|
|
from src.meeting_lab.models.meeting_context import MeetingContext # noqa: E402
|
|
from src.meeting_lab.orchestration.mvp import ( # noqa: E402
|
|
DEFAULT_DIARIZATION_MODEL,
|
|
DEFAULT_MODEL,
|
|
DEFAULT_OUTPUT_ROOT,
|
|
DEFAULT_SAFE_INPUT_TOKEN_BUDGET,
|
|
MvpMeetingConfig,
|
|
create_unique_run_dir,
|
|
run_mvp_meeting,
|
|
)
|
|
from src.meeting_lab.progress import ProgressSink # noqa: E402
|
|
|
|
|
|
def parse_args(argv: list[str] | None = None) -> argparse.Namespace:
|
|
parser = argparse.ArgumentParser(
|
|
description="Transcribe one meeting and generate one direct protocol."
|
|
)
|
|
parser.add_argument("audio_file", type=Path)
|
|
parser.add_argument("--whisper-model", type=Path, required=True)
|
|
parser.add_argument("--whisper-executable", default="whisper-cli")
|
|
parser.add_argument("--ffmpeg-executable", default="ffmpeg")
|
|
parser.add_argument(
|
|
"--audio-normalization",
|
|
action=argparse.BooleanOptionalAction,
|
|
default=True,
|
|
help=(
|
|
"Enable FFmpeg loudness normalization during canonical audio preparation "
|
|
"(default: enabled)."
|
|
),
|
|
)
|
|
parser.add_argument("--context", type=Path)
|
|
parser.add_argument("--output-root", type=Path, default=DEFAULT_OUTPUT_ROOT)
|
|
parser.add_argument("--language", default="de")
|
|
parser.add_argument(
|
|
"--threads",
|
|
default="auto",
|
|
help="Thread count or 'auto' for physical CPU cores (default: auto).",
|
|
)
|
|
parser.add_argument("--model", default=DEFAULT_MODEL)
|
|
parser.add_argument("--ollama-endpoint", default=DEFAULT_ENDPOINT)
|
|
parser.add_argument(
|
|
"--protocol-safe-input-token-budget",
|
|
type=int,
|
|
default=DEFAULT_SAFE_INPUT_TOKEN_BUDGET,
|
|
help="Conservative estimated prompt-token limit before any Ollama request.",
|
|
)
|
|
parser.add_argument(
|
|
"--diarization",
|
|
choices=("auto", "gpu", "cpu", "off"),
|
|
default="off",
|
|
help="Optional Community-1 diarization device mode (default: off).",
|
|
)
|
|
parser.add_argument(
|
|
"--diarization-runtime",
|
|
choices=("native", "container"),
|
|
default="native",
|
|
help="Run pyannote in this Python environment or an explicit container.",
|
|
)
|
|
parser.add_argument(
|
|
"--diarization-container-image",
|
|
help="Container image required with --diarization-runtime container.",
|
|
)
|
|
parser.add_argument(
|
|
"--diarization-container-arg",
|
|
action="append",
|
|
default=[],
|
|
help="Additional docker argument; repeat and use = for values beginning with --.",
|
|
)
|
|
return parser.parse_args(argv)
|
|
|
|
|
|
def config_from_args(args: argparse.Namespace) -> MvpMeetingConfig:
|
|
return MvpMeetingConfig(
|
|
audio_file=args.audio_file,
|
|
whisper_model=args.whisper_model,
|
|
whisper_executable=args.whisper_executable,
|
|
ffmpeg_executable=args.ffmpeg_executable,
|
|
audio_normalization=args.audio_normalization,
|
|
context_file=args.context,
|
|
output_root=args.output_root,
|
|
language=args.language,
|
|
threads=args.threads,
|
|
model=args.model,
|
|
ollama_endpoint=args.ollama_endpoint,
|
|
protocol_safe_input_token_budget=args.protocol_safe_input_token_budget,
|
|
diarization=args.diarization,
|
|
diarization_runtime=args.diarization_runtime,
|
|
diarization_container_image=args.diarization_container_image,
|
|
diarization_container_args=tuple(args.diarization_container_arg),
|
|
)
|
|
|
|
|
|
def run(
|
|
args: argparse.Namespace,
|
|
*,
|
|
context_override: MeetingContext | dict[str, Any] | None = None,
|
|
progress_sink: ProgressSink | None = None,
|
|
) -> tuple[int, Path | None, Path | None]:
|
|
"""Compatibility wrapper for existing Python callers of the CLI module."""
|
|
result = run_mvp_meeting(
|
|
config_from_args(args),
|
|
meeting_context=context_override,
|
|
progress_sink=progress_sink,
|
|
)
|
|
return result.exit_code, result.run_dir, result.protocol_path
|
|
|
|
|
|
def main(argv: list[str] | None = None) -> int:
|
|
args = parse_args(argv)
|
|
if args.diarization == "off":
|
|
print("Diarization: disabled")
|
|
else:
|
|
print(
|
|
f"Diarization: enabled; backend=pyannote.audio; "
|
|
f"model={DEFAULT_DIARIZATION_MODEL}; requested_device={args.diarization}; "
|
|
f"runtime={args.diarization_runtime}"
|
|
)
|
|
code, run_dir, protocol_path = run(args)
|
|
if run_dir is not None and args.diarization != "off":
|
|
metadata_path = run_dir / "diarization" / "metadata.json"
|
|
if metadata_path.is_file():
|
|
details = json.loads(metadata_path.read_text(encoding="utf-8"))
|
|
print(
|
|
f"Diarization result: device={details.get('actual_device')}; "
|
|
f"device_name={details.get('device_name') or 'n/a'}; "
|
|
f"runtime={details.get('runtime_seconds'):.3f}s; "
|
|
f"speakers={details.get('speaker_count')}; artifacts={metadata_path.parent}"
|
|
)
|
|
if protocol_path is not None:
|
|
print(protocol_path)
|
|
return code
|
|
|
|
|
|
if __name__ == "__main__":
|
|
raise SystemExit(main())
|