Use physical CPU cores for Whisper runtime defaults
This commit is contained in:
@@ -0,0 +1,211 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Run the minimal Meeting Lab MVP: audio -> Whisper -> direct protocol."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import re
|
||||
import shutil
|
||||
import sys
|
||||
import time
|
||||
from datetime import datetime
|
||||
from pathlib import Path
|
||||
from typing import Any, Callable
|
||||
|
||||
REPO_ROOT = Path(__file__).resolve().parents[1]
|
||||
if str(REPO_ROOT) not in sys.path:
|
||||
sys.path.insert(0, str(REPO_ROOT))
|
||||
|
||||
from src.meeting_lab.llm.ollama import DEFAULT_ENDPOINT # noqa: E402
|
||||
from src.meeting_lab.models.meeting_context import load_meeting_context # noqa: E402
|
||||
from src.meeting_lab.protocol.generate_direct_protocol import ( # noqa: E402
|
||||
DEFAULT_MODEL,
|
||||
DirectProtocolResult,
|
||||
generate_direct_protocol,
|
||||
load_compact_transcript,
|
||||
)
|
||||
from src.meeting_lab.transcription.whisper import transcribe_audio # noqa: E402
|
||||
|
||||
|
||||
DEFAULT_OUTPUT_ROOT = Path("meeting_data/runs")
|
||||
|
||||
|
||||
def parse_args(argv: list[str] | None = None) -> argparse.Namespace:
|
||||
parser = argparse.ArgumentParser(
|
||||
description="Transcribe one meeting and generate one direct protocol."
|
||||
)
|
||||
parser.add_argument("audio_file", type=Path)
|
||||
parser.add_argument("--whisper-model", type=Path, required=True)
|
||||
parser.add_argument("--whisper-executable", default="whisper-cli")
|
||||
parser.add_argument("--context", type=Path)
|
||||
parser.add_argument("--output-root", type=Path, default=DEFAULT_OUTPUT_ROOT)
|
||||
parser.add_argument("--language", default="de")
|
||||
parser.add_argument("--threads", default="auto", help="Thread count or 'auto' for physical CPU cores (default: auto).")
|
||||
parser.add_argument("--model", default=DEFAULT_MODEL)
|
||||
parser.add_argument("--ollama-endpoint", default=DEFAULT_ENDPOINT)
|
||||
return parser.parse_args(argv)
|
||||
|
||||
|
||||
def create_unique_run_dir(
|
||||
output_root: Path,
|
||||
meeting_name: str,
|
||||
now: Callable[[], datetime] = datetime.now,
|
||||
) -> Path:
|
||||
safe_name = re.sub(r"[^A-Za-z0-9_.-]+", "_", meeting_name).strip("._-") or "meeting"
|
||||
base = output_root / f"{safe_name}_{now().strftime('%Y%m%d_%H%M%S')}"
|
||||
candidate = base
|
||||
suffix = 1
|
||||
while candidate.exists():
|
||||
candidate = output_root / f"{base.name}_{suffix:02d}"
|
||||
suffix += 1
|
||||
candidate.mkdir(parents=True)
|
||||
return candidate
|
||||
|
||||
|
||||
def write_json(path: Path, value: Any) -> None:
|
||||
path.write_text(json.dumps(value, ensure_ascii=False, indent=2) + "\n", encoding="utf-8")
|
||||
|
||||
|
||||
def validate_inputs(args: argparse.Namespace) -> None:
|
||||
if not args.audio_file.is_file():
|
||||
raise FileNotFoundError(f"Audio file does not exist: {args.audio_file}")
|
||||
if not args.whisper_model.is_file():
|
||||
raise FileNotFoundError(f"Whisper model does not exist: {args.whisper_model}")
|
||||
if args.context is not None:
|
||||
if not args.context.is_file():
|
||||
raise FileNotFoundError(f"Meeting Context file does not exist: {args.context}")
|
||||
load_meeting_context(args.context)
|
||||
|
||||
|
||||
def persist_protocol(run_dir: Path, result: DirectProtocolResult) -> Path:
|
||||
protocol_dir = run_dir / "protocol"
|
||||
protocol_dir.mkdir(exist_ok=True)
|
||||
(protocol_dir / "exact_prompt.txt").write_text(result.exact_prompt, encoding="utf-8")
|
||||
write_json(protocol_dir / "raw_response.json", result.raw_response)
|
||||
write_json(protocol_dir / "runtime_metadata.json", result.runtime_metadata)
|
||||
protocol_path = run_dir / "protocol.md"
|
||||
protocol_path.write_text(result.protocol_text, encoding="utf-8")
|
||||
return protocol_path
|
||||
|
||||
|
||||
def run(args: argparse.Namespace) -> tuple[int, Path | None, Path | None]:
|
||||
overall_started = time.perf_counter()
|
||||
validation_started = time.perf_counter()
|
||||
try:
|
||||
validate_inputs(args)
|
||||
except Exception as exc:
|
||||
print(f"Error: {type(exc).__name__}: {exc}", file=sys.stderr)
|
||||
return 2, None, None
|
||||
|
||||
validation_runtime = time.perf_counter() - validation_started
|
||||
run_dir = create_unique_run_dir(args.output_root, args.audio_file.stem)
|
||||
timestamp = datetime.now().astimezone().isoformat(timespec="seconds")
|
||||
transcript_path = run_dir / "transcript" / "transcript.json"
|
||||
protocol_path = run_dir / "protocol.md"
|
||||
stage_runtimes: dict[str, float | None] = {
|
||||
"validation": round(validation_runtime, 3),
|
||||
"setup": None,
|
||||
"whisper": None,
|
||||
"transcript_validation": None,
|
||||
"protocol": None,
|
||||
}
|
||||
metadata: dict[str, Any] = {
|
||||
"run_id": run_dir.name,
|
||||
"timestamp": timestamp,
|
||||
"input_audio": str(args.audio_file.resolve()),
|
||||
"transcript_output": str(transcript_path.resolve()),
|
||||
"protocol_output": str(protocol_path.resolve()),
|
||||
"whisper_model": str(args.whisper_model.resolve()),
|
||||
"model": args.model,
|
||||
"ollama_endpoint": args.ollama_endpoint,
|
||||
"status": "running",
|
||||
"stage_runtimes_seconds": stage_runtimes,
|
||||
"total_runtime_seconds": None,
|
||||
"failure": None,
|
||||
}
|
||||
current_stage = "setup"
|
||||
stage_started = time.perf_counter()
|
||||
|
||||
try:
|
||||
audio_dir = run_dir / "audio"
|
||||
transcript_dir = run_dir / "transcript"
|
||||
context_dir = run_dir / "context"
|
||||
protocol_dir = run_dir / "protocol"
|
||||
audio_dir.mkdir()
|
||||
transcript_dir.mkdir()
|
||||
context_dir.mkdir()
|
||||
protocol_dir.mkdir()
|
||||
write_json(
|
||||
audio_dir / "input_manifest.json",
|
||||
{
|
||||
"source_file": str(args.audio_file.resolve()),
|
||||
"filename": args.audio_file.name,
|
||||
"size_bytes": args.audio_file.stat().st_size,
|
||||
},
|
||||
)
|
||||
|
||||
preserved_context: Path | None = None
|
||||
if args.context is not None:
|
||||
preserved_context = context_dir / "meeting_context.yaml"
|
||||
shutil.copy2(args.context, preserved_context)
|
||||
stage_runtimes["setup"] = round(time.perf_counter() - stage_started, 3)
|
||||
|
||||
current_stage = "whisper"
|
||||
stage_started = time.perf_counter()
|
||||
transcription = transcribe_audio(
|
||||
args.audio_file,
|
||||
args.whisper_model,
|
||||
transcript_dir,
|
||||
args.language,
|
||||
executable=args.whisper_executable,
|
||||
threads=args.threads,
|
||||
)
|
||||
stage_runtimes["whisper"] = round(time.perf_counter() - stage_started, 3)
|
||||
|
||||
current_stage = "transcript_validation"
|
||||
stage_started = time.perf_counter()
|
||||
load_compact_transcript(transcription.transcript_json)
|
||||
stage_runtimes["transcript_validation"] = round(
|
||||
time.perf_counter() - stage_started, 3
|
||||
)
|
||||
|
||||
current_stage = "protocol"
|
||||
stage_started = time.perf_counter()
|
||||
result = generate_direct_protocol(
|
||||
transcription.transcript_json,
|
||||
preserved_context,
|
||||
model=args.model,
|
||||
endpoint=args.ollama_endpoint,
|
||||
)
|
||||
stage_runtimes["protocol"] = round(time.perf_counter() - stage_started, 3)
|
||||
protocol_path = persist_protocol(run_dir, result)
|
||||
metadata["status"] = "completed"
|
||||
except Exception as exc:
|
||||
if current_stage in stage_runtimes and stage_runtimes[current_stage] is None:
|
||||
stage_runtimes[current_stage] = round(time.perf_counter() - stage_started, 3)
|
||||
metadata["status"] = "failed"
|
||||
metadata["failure"] = {
|
||||
"stage": current_stage,
|
||||
"type": type(exc).__name__,
|
||||
"message": str(exc),
|
||||
}
|
||||
protocol_path = None
|
||||
print(f"Error: {type(exc).__name__}: {exc}", file=sys.stderr)
|
||||
finally:
|
||||
metadata["total_runtime_seconds"] = round(time.perf_counter() - overall_started, 3)
|
||||
write_json(run_dir / "run_metadata.json", metadata)
|
||||
|
||||
return (0 if metadata["status"] == "completed" else 2), run_dir, protocol_path
|
||||
|
||||
|
||||
def main(argv: list[str] | None = None) -> int:
|
||||
args = parse_args(argv)
|
||||
code, _run_dir, protocol_path = run(args)
|
||||
if protocol_path is not None:
|
||||
print(protocol_path)
|
||||
return code
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
@@ -0,0 +1,51 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Transcribe one audio file with whisper.cpp; do not generate a protocol."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
REPO_ROOT = Path(__file__).resolve().parents[1]
|
||||
if str(REPO_ROOT) not in sys.path:
|
||||
sys.path.insert(0, str(REPO_ROOT))
|
||||
|
||||
from src.meeting_lab.transcription.whisper import ( # noqa: E402
|
||||
TranscriptionError,
|
||||
transcribe_audio,
|
||||
)
|
||||
|
||||
|
||||
def parse_args(argv: list[str] | None = None) -> argparse.Namespace:
|
||||
parser = argparse.ArgumentParser(description="Create a compact Meeting Lab transcript with whisper.cpp.")
|
||||
parser.add_argument("audio_file", type=Path)
|
||||
parser.add_argument("--model", type=Path, required=True, help="Path to a whisper.cpp GGML model.")
|
||||
parser.add_argument("--output-dir", type=Path, required=True)
|
||||
parser.add_argument("--language", default="auto", help="Language code or 'auto' (default: auto).")
|
||||
parser.add_argument("--threads", default="auto", help="Thread count or 'auto' for physical CPU cores (default: auto).")
|
||||
parser.add_argument("--whisper-executable", default="whisper-cli", help="whisper.cpp CLI executable (default: whisper-cli).")
|
||||
return parser.parse_args(argv)
|
||||
|
||||
|
||||
def main(argv: list[str] | None = None) -> int:
|
||||
args = parse_args(argv)
|
||||
try:
|
||||
result = transcribe_audio(
|
||||
args.audio_file,
|
||||
args.model,
|
||||
args.output_dir,
|
||||
args.language,
|
||||
executable=args.whisper_executable,
|
||||
threads=args.threads,
|
||||
)
|
||||
except TranscriptionError as exc:
|
||||
print(f"Error: {exc}")
|
||||
return 1
|
||||
print(f"Transcript: {result.transcript_json}")
|
||||
print(f"Runtime: {result.runtime_seconds:.3f} seconds")
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
Reference in New Issue
Block a user