Add one-command Meeting Lab benchmark runner

Add a repository-native runner for reproducible Meeting Lab benchmark runs on machines without Codex.

The runner:

- validates normalized Whisper input and Meeting Context
- checks Ollama availability and the requested model
- rejects preloaded Ollama models by default for clean benchmarks
- supports an explicit --allow-loaded-models override
- uses the selected production configuration:
  - qwen3.5:9b
  - target_chars=4500
  - max_chars=5500
  - min_chars=2500
  - overlap_blocks=0
  - think=false
  - temperature=0
  - num_ctx=32768
- executes the complete current pipeline
- creates unique benchmark output directories
- preserves artifacts up to failure
- records runtime, environment and validation metadata
- writes working_protocol.md only when the renderer contract passes

Add focused mocked tests and Linux-first setup documentation for the AI-PC.
The runner does not include Whisper execution.
This commit is contained in:
2026-08-05 14:48:42 +02:00
parent 03bc6b1d90
commit e04e2533fc
5 changed files with 1467 additions and 0 deletions
+489
View File
@@ -0,0 +1,489 @@
import json
import unittest
import uuid
from datetime import datetime
from pathlib import Path
from unittest.mock import Mock, patch
from scripts import run_meeting
TEST_TMP = Path("tmp_run_meeting_tests")
VALID_CONTEXT = """schema_version: "1"
meeting:
meeting_id: "test-meeting"
title: "Test Meeting"
language: "de"
participants:
- participant_id: "martin"
display_name: "Martin"
aliases: []
attendance_status: "present"
mentioned_people: []
organization:
departments: []
known_entities: {}
"""
def write_whisper_json(path: Path) -> None:
path.write_text(
json.dumps(
{
"text": "Martin: Hallo.\nJames: Weiter.",
"segments": [
{"text": "Martin: Hallo."},
{"text": "James: Weiter."},
],
}
),
encoding="utf-8",
)
def write_context(path: Path, text: str = VALID_CONTEXT) -> None:
path.write_text(text, encoding="utf-8")
def response_json(payload):
response = Mock()
response.raise_for_status.return_value = None
response.json.return_value = payload
return response
def workspace_tempdir():
TEST_TMP.mkdir(exist_ok=True)
return ManualTempDirectory(TEST_TMP / f"run-meeting-{uuid.uuid4().hex}")
class ManualTempDirectory:
def __init__(self, path: Path) -> None:
self.path = path
def __enter__(self) -> str:
self.path.mkdir(parents=True)
return str(self.path)
def __exit__(self, exc_type, exc, tb) -> None:
return None
class RunMeetingTests(unittest.TestCase):
def test_cli_argument_parsing_defaults(self) -> None:
args = run_meeting.parse_args(
[
"--input",
"input.json",
"--context",
"context.yaml",
]
)
self.assertEqual(args.model, "qwen3.5:9b")
self.assertEqual(args.target_chars, 4500)
self.assertEqual(args.max_chars, 5500)
self.assertEqual(args.min_chars, 2500)
self.assertEqual(args.overlap_blocks, 0)
self.assertEqual(args.temperature, 0.0)
self.assertEqual(args.num_ctx, 32768)
self.assertFalse(args.allow_loaded_models)
def test_explicit_parameter_overrides(self) -> None:
args = run_meeting.parse_args(
[
"--input",
"input.json",
"--context",
"context.yaml",
"--model",
"other:model",
"--target-chars",
"3000",
"--max-chars",
"4000",
"--min-chars",
"1000",
"--overlap-blocks",
"1",
"--num-ctx",
"8192",
"--temperature",
"0.2",
]
)
self.assertEqual(args.model, "other:model")
self.assertEqual(args.target_chars, 3000)
self.assertEqual(args.max_chars, 4000)
self.assertEqual(args.min_chars, 1000)
self.assertEqual(args.overlap_blocks, 1)
self.assertEqual(args.num_ctx, 8192)
self.assertEqual(args.temperature, 0.2)
def test_unique_output_directory_creation(self) -> None:
with workspace_tempdir() as directory:
root = Path(directory)
fixed = datetime(2026, 8, 5, 12, 0, 0)
first = run_meeting.create_unique_output_dir(root, "progeo test", now=lambda: fixed)
second = run_meeting.create_unique_output_dir(root, "progeo test", now=lambda: fixed)
self.assertEqual(first.name, "progeo_test_20260805_120000")
self.assertEqual(second.name, "progeo_test_20260805_120000_01")
def test_invalid_input_path_fails_early(self) -> None:
with workspace_tempdir() as directory:
root = Path(directory)
context = root / "context.yaml"
write_context(context)
args = run_meeting.parse_args(
[
"--input",
str(root / "missing.json"),
"--context",
str(context),
"--output-root",
str(root / "benchmarks"),
]
)
code, output_dir, _report, _protocol = run_meeting.run_pipeline(args)
self.assertEqual(code, 2)
self.assertTrue((output_dir / "run_metadata.json").exists())
def test_invalid_meeting_context_fails(self) -> None:
with workspace_tempdir() as directory:
root = Path(directory)
input_path = root / "input.json"
context = root / "context.yaml"
write_whisper_json(input_path)
write_context(context, "schema_version: '['\n")
args = run_meeting.parse_args(
[
"--input",
str(input_path),
"--context",
str(context),
"--output-root",
str(root / "benchmarks"),
]
)
code, _output_dir, _report, _protocol = run_meeting.run_pipeline(args)
self.assertEqual(code, 2)
def test_unavailable_ollama_endpoint_fails(self) -> None:
with patch("scripts.run_meeting.requests.get", side_effect=run_meeting.requests.ConnectionError("down")):
with self.assertRaisesRegex(run_meeting.RunnerError, "not reachable"):
run_meeting.require_ollama("http://127.0.0.1:11434/api/generate", "qwen3.5:9b")
def test_missing_model_fails(self) -> None:
response = response_json({"models": [{"name": "other:model"}]})
with patch("scripts.run_meeting.requests.get", return_value=response):
with self.assertRaisesRegex(run_meeting.RunnerError, "not installed"):
run_meeting.require_ollama("http://127.0.0.1:11434/api/generate", "qwen3.5:9b")
def test_loaded_model_fails_preflight_by_default(self) -> None:
responses = [
response_json({"models": [{"name": "qwen3.5:9b"}]}),
response_json({"models": [{"name": "other:model"}]}),
]
with patch("scripts.run_meeting.requests.get", side_effect=responses):
with self.assertRaisesRegex(run_meeting.RunnerError, "loaded model"):
run_meeting.require_ollama("http://127.0.0.1:11434/api/generate", "qwen3.5:9b")
def test_loaded_model_override_permits_execution(self) -> None:
responses = [
response_json({"models": [{"name": "qwen3.5:9b"}]}),
response_json({"models": [{"name": "other:model"}]}),
]
with patch("scripts.run_meeting.requests.get", side_effect=responses):
data = run_meeting.require_ollama(
"http://127.0.0.1:11434/api/generate",
"qwen3.5:9b",
allow_loaded_models=True,
)
self.assertEqual(data["active_models"], ["other:model"])
self.assertTrue(data["allow_loaded_models"])
def test_loaded_model_preflight_failure_records_metadata(self) -> None:
with workspace_tempdir() as directory:
root = Path(directory)
input_path = root / "input.json"
context = root / "context.yaml"
write_whisper_json(input_path)
write_context(context)
args = run_meeting.parse_args(
[
"--input",
str(input_path),
"--context",
str(context),
"--output-root",
str(root / "benchmarks"),
]
)
error = run_meeting.RunnerError(
"Ollama has loaded model(s).",
details={
"preflight": {
"ollama": {
"active_models": ["other:model"],
"allow_loaded_models": False,
}
}
},
)
with patch("scripts.run_meeting.require_ollama", side_effect=error):
code, output_dir, report_path, _protocol = run_meeting.run_pipeline(args)
metadata = json.loads((output_dir / "run_metadata.json").read_text(encoding="utf-8"))
report = report_path.read_text(encoding="utf-8")
self.assertEqual(code, 2)
self.assertEqual(metadata["preflight"]["ollama"]["active_models"], ["other:model"])
self.assertFalse(metadata["preflight"]["ollama"]["allow_loaded_models"])
self.assertIn("Detected loaded models: `['other:model']`", report)
self.assertIn("Loaded-model override: `False`", report)
def test_stage_failure_preserves_artifacts(self) -> None:
with workspace_tempdir() as directory:
root = Path(directory)
input_path = root / "input.json"
context = root / "context.yaml"
write_whisper_json(input_path)
write_context(context)
args = run_meeting.parse_args(
[
"--input",
str(input_path),
"--context",
str(context),
"--output-root",
str(root / "benchmarks"),
]
)
def fail_extraction(**kwargs):
marker = kwargs["output_dir"] / "partial.raw.txt"
marker.parent.mkdir(parents=True, exist_ok=True)
marker.write_text("partial", encoding="utf-8")
raise ValueError("model failed")
with patch("scripts.run_meeting.require_ollama", return_value={"active_models": []}), patch(
"scripts.run_meeting.run_extraction_stage",
side_effect=fail_extraction,
):
code, output_dir, report_path, _protocol = run_meeting.run_pipeline(args)
self.assertEqual(code, 2)
self.assertTrue((output_dir / "extractions" / "partial.raw.txt").exists())
self.assertTrue(report_path.exists())
self.assertIn("model failed", (output_dir / "run_metadata.json").read_text(encoding="utf-8"))
def test_benchmark_report_generation(self) -> None:
with workspace_tempdir() as directory:
path = Path(directory) / "report.md"
run_meeting.write_benchmark_report(
path,
{
"status": "completed",
"input": "input.json",
"context": "context.yaml",
"model": "qwen3.5:9b",
"output_dir": "out",
"configuration": {
"target_chars": 4500,
"max_chars": 5500,
"min_chars": 2500,
"overlap_blocks": 0,
"temperature": 0.0,
"num_ctx": 32768,
},
"runtime_seconds_by_stage": {"chunking": {"runtime_seconds": 1.0, "status": "passed"}},
"preflight": {
"ollama": {"active_models": ["other:model"], "allow_loaded_models": True}
},
"total_runtime_seconds": 1.0,
"average_extraction_seconds_per_chunk": None,
"chunk_distribution": {},
"extraction_counts": {},
"canonicalizer_stats": {},
"semantic_consolidator": {
"validator_before": {"valid": True, "violations": []},
"validator_after": {"valid": True, "violations": []},
"repair_count": 0,
},
"renderer": {"valid": False, "violations": [{"type": "x"}]},
},
)
text = path.read_text(encoding="utf-8")
self.assertIn("# Meeting Lab Benchmark Report", text)
self.assertIn("Renderer contract valid: False", text)
self.assertIn("Detected loaded models: `['other:model']`", text)
self.assertIn("Loaded-model override: `True`", text)
def test_metadata_records_loaded_model_condition_and_override(self) -> None:
with workspace_tempdir() as directory:
root = Path(directory)
input_path = root / "input.json"
context = root / "context.yaml"
write_whisper_json(input_path)
write_context(context)
args = run_meeting.parse_args(
[
"--input",
str(input_path),
"--context",
str(context),
"--output-root",
str(root / "benchmarks"),
"--allow-loaded-models",
]
)
def fake_extraction(**kwargs):
output = kwargs["output_dir"] / "chunk_01_extraction.json"
output.parent.mkdir(parents=True, exist_ok=True)
output.write_text(
json.dumps(
{
"facts": ["Martin | Test fact | clear | evidence"],
"decisions": [],
"todos": [],
"questions": [],
"positions": [],
"technical": [],
"context": {"source_file": str(context)},
}
),
encoding="utf-8",
)
return [output]
def fake_semantic(**_kwargs):
return {
"validator_before": {"valid": True, "violations": []},
"validator_after": {"valid": True, "violations": []},
"repair_count": 0,
"response_metadata": {},
}
def fake_renderer(**kwargs):
output_dir = kwargs["output_dir"]
output_dir.mkdir(parents=True, exist_ok=True)
(output_dir / "validation_report.json").write_text(
json.dumps({"valid": False, "violations": []}),
encoding="utf-8",
)
return {"valid": False, "runtime_seconds": 0.1, "output_path": None}
with patch(
"scripts.run_meeting.require_ollama",
return_value={"active_models": ["other:model"], "allow_loaded_models": True},
), patch(
"scripts.run_meeting.run_extraction_stage",
side_effect=fake_extraction,
), patch(
"scripts.run_meeting.run_semantic_consolidator_stage",
side_effect=fake_semantic,
), patch(
"scripts.run_meeting.render_working_protocol",
side_effect=fake_renderer,
):
code, output_dir, report_path, _protocol_path = run_meeting.run_pipeline(args)
metadata = json.loads((output_dir / "run_metadata.json").read_text(encoding="utf-8"))
report = report_path.read_text(encoding="utf-8")
self.assertEqual(code, 0)
self.assertTrue(metadata["configuration"]["allow_loaded_models"])
self.assertEqual(metadata["preflight"]["ollama"]["active_models"], ["other:model"])
self.assertTrue(metadata["preflight"]["ollama"]["allow_loaded_models"])
self.assertIn("Detected loaded models: `['other:model']`", report)
self.assertIn("Loaded-model override: `True`", report)
def test_successful_run_uses_mocked_llm_stages(self) -> None:
with workspace_tempdir() as directory:
root = Path(directory)
input_path = root / "input.json"
context = root / "context.yaml"
write_whisper_json(input_path)
write_context(context)
args = run_meeting.parse_args(
[
"--input",
str(input_path),
"--context",
str(context),
"--output-root",
str(root / "benchmarks"),
]
)
def fake_extraction(**kwargs):
output = kwargs["output_dir"] / "chunk_01_extraction.json"
output.parent.mkdir(parents=True, exist_ok=True)
output.write_text(
json.dumps(
{
"facts": ["Martin | Test fact | clear | evidence"],
"decisions": [],
"todos": [],
"questions": [],
"positions": [],
"technical": [],
"context": {"source_file": str(context)},
}
),
encoding="utf-8",
)
output.with_suffix(".raw.txt").write_text("{}", encoding="utf-8")
return [output]
def fake_semantic(**_kwargs):
out = root / "placeholder"
return {
"validator_before": {"valid": True, "violations": []},
"validator_after": {"valid": True, "violations": []},
"repair_count": 0,
"response_metadata": {},
"output": str(out),
}
def fake_renderer(**kwargs):
output_dir = kwargs["output_dir"]
output_dir.mkdir(parents=True, exist_ok=True)
(output_dir / "validation_report.json").write_text(
json.dumps({"valid": False, "violations": [{"type": "missing_required_heading"}]}),
encoding="utf-8",
)
return {"valid": False, "runtime_seconds": 0.1, "output_path": None}
with patch("scripts.run_meeting.require_ollama", return_value={"active_models": []}), patch(
"scripts.run_meeting.run_extraction_stage",
side_effect=fake_extraction,
), patch(
"scripts.run_meeting.run_semantic_consolidator_stage",
side_effect=fake_semantic,
), patch(
"scripts.run_meeting.render_working_protocol",
side_effect=fake_renderer,
):
code, output_dir, report_path, protocol_path = run_meeting.run_pipeline(args)
self.assertEqual(code, 0)
self.assertTrue((output_dir / "canonicalizer" / "canonicalized_extractions.json").exists())
self.assertTrue(report_path.exists())
self.assertIsNone(protocol_path)
if __name__ == "__main__":
unittest.main()