Add a repository-native runner for reproducible Meeting Lab benchmark runs on machines without Codex. The runner: - validates normalized Whisper input and Meeting Context - checks Ollama availability and the requested model - rejects preloaded Ollama models by default for clean benchmarks - supports an explicit --allow-loaded-models override - uses the selected production configuration: - qwen3.5:9b - target_chars=4500 - max_chars=5500 - min_chars=2500 - overlap_blocks=0 - think=false - temperature=0 - num_ctx=32768 - executes the complete current pipeline - creates unique benchmark output directories - preserves artifacts up to failure - records runtime, environment and validation metadata - writes working_protocol.md only when the renderer contract passes Add focused mocked tests and Linux-first setup documentation for the AI-PC. The runner does not include Whisper execution.
490 lines
18 KiB
Python
490 lines
18 KiB
Python
import json
|
|
import unittest
|
|
import uuid
|
|
from datetime import datetime
|
|
from pathlib import Path
|
|
from unittest.mock import Mock, patch
|
|
|
|
from scripts import run_meeting
|
|
|
|
|
|
TEST_TMP = Path("tmp_run_meeting_tests")
|
|
|
|
|
|
VALID_CONTEXT = """schema_version: "1"
|
|
|
|
meeting:
|
|
meeting_id: "test-meeting"
|
|
title: "Test Meeting"
|
|
language: "de"
|
|
|
|
participants:
|
|
- participant_id: "martin"
|
|
display_name: "Martin"
|
|
aliases: []
|
|
attendance_status: "present"
|
|
|
|
mentioned_people: []
|
|
organization:
|
|
departments: []
|
|
known_entities: {}
|
|
"""
|
|
|
|
|
|
def write_whisper_json(path: Path) -> None:
|
|
path.write_text(
|
|
json.dumps(
|
|
{
|
|
"text": "Martin: Hallo.\nJames: Weiter.",
|
|
"segments": [
|
|
{"text": "Martin: Hallo."},
|
|
{"text": "James: Weiter."},
|
|
],
|
|
}
|
|
),
|
|
encoding="utf-8",
|
|
)
|
|
|
|
|
|
def write_context(path: Path, text: str = VALID_CONTEXT) -> None:
|
|
path.write_text(text, encoding="utf-8")
|
|
|
|
|
|
def response_json(payload):
|
|
response = Mock()
|
|
response.raise_for_status.return_value = None
|
|
response.json.return_value = payload
|
|
return response
|
|
|
|
|
|
def workspace_tempdir():
|
|
TEST_TMP.mkdir(exist_ok=True)
|
|
return ManualTempDirectory(TEST_TMP / f"run-meeting-{uuid.uuid4().hex}")
|
|
|
|
|
|
class ManualTempDirectory:
|
|
def __init__(self, path: Path) -> None:
|
|
self.path = path
|
|
|
|
def __enter__(self) -> str:
|
|
self.path.mkdir(parents=True)
|
|
return str(self.path)
|
|
|
|
def __exit__(self, exc_type, exc, tb) -> None:
|
|
return None
|
|
|
|
|
|
class RunMeetingTests(unittest.TestCase):
|
|
def test_cli_argument_parsing_defaults(self) -> None:
|
|
args = run_meeting.parse_args(
|
|
[
|
|
"--input",
|
|
"input.json",
|
|
"--context",
|
|
"context.yaml",
|
|
]
|
|
)
|
|
|
|
self.assertEqual(args.model, "qwen3.5:9b")
|
|
self.assertEqual(args.target_chars, 4500)
|
|
self.assertEqual(args.max_chars, 5500)
|
|
self.assertEqual(args.min_chars, 2500)
|
|
self.assertEqual(args.overlap_blocks, 0)
|
|
self.assertEqual(args.temperature, 0.0)
|
|
self.assertEqual(args.num_ctx, 32768)
|
|
self.assertFalse(args.allow_loaded_models)
|
|
|
|
def test_explicit_parameter_overrides(self) -> None:
|
|
args = run_meeting.parse_args(
|
|
[
|
|
"--input",
|
|
"input.json",
|
|
"--context",
|
|
"context.yaml",
|
|
"--model",
|
|
"other:model",
|
|
"--target-chars",
|
|
"3000",
|
|
"--max-chars",
|
|
"4000",
|
|
"--min-chars",
|
|
"1000",
|
|
"--overlap-blocks",
|
|
"1",
|
|
"--num-ctx",
|
|
"8192",
|
|
"--temperature",
|
|
"0.2",
|
|
]
|
|
)
|
|
|
|
self.assertEqual(args.model, "other:model")
|
|
self.assertEqual(args.target_chars, 3000)
|
|
self.assertEqual(args.max_chars, 4000)
|
|
self.assertEqual(args.min_chars, 1000)
|
|
self.assertEqual(args.overlap_blocks, 1)
|
|
self.assertEqual(args.num_ctx, 8192)
|
|
self.assertEqual(args.temperature, 0.2)
|
|
|
|
def test_unique_output_directory_creation(self) -> None:
|
|
with workspace_tempdir() as directory:
|
|
root = Path(directory)
|
|
fixed = datetime(2026, 8, 5, 12, 0, 0)
|
|
first = run_meeting.create_unique_output_dir(root, "progeo test", now=lambda: fixed)
|
|
second = run_meeting.create_unique_output_dir(root, "progeo test", now=lambda: fixed)
|
|
|
|
self.assertEqual(first.name, "progeo_test_20260805_120000")
|
|
self.assertEqual(second.name, "progeo_test_20260805_120000_01")
|
|
|
|
def test_invalid_input_path_fails_early(self) -> None:
|
|
with workspace_tempdir() as directory:
|
|
root = Path(directory)
|
|
context = root / "context.yaml"
|
|
write_context(context)
|
|
args = run_meeting.parse_args(
|
|
[
|
|
"--input",
|
|
str(root / "missing.json"),
|
|
"--context",
|
|
str(context),
|
|
"--output-root",
|
|
str(root / "benchmarks"),
|
|
]
|
|
)
|
|
code, output_dir, _report, _protocol = run_meeting.run_pipeline(args)
|
|
|
|
self.assertEqual(code, 2)
|
|
self.assertTrue((output_dir / "run_metadata.json").exists())
|
|
|
|
def test_invalid_meeting_context_fails(self) -> None:
|
|
with workspace_tempdir() as directory:
|
|
root = Path(directory)
|
|
input_path = root / "input.json"
|
|
context = root / "context.yaml"
|
|
write_whisper_json(input_path)
|
|
write_context(context, "schema_version: '['\n")
|
|
args = run_meeting.parse_args(
|
|
[
|
|
"--input",
|
|
str(input_path),
|
|
"--context",
|
|
str(context),
|
|
"--output-root",
|
|
str(root / "benchmarks"),
|
|
]
|
|
)
|
|
code, _output_dir, _report, _protocol = run_meeting.run_pipeline(args)
|
|
|
|
self.assertEqual(code, 2)
|
|
|
|
def test_unavailable_ollama_endpoint_fails(self) -> None:
|
|
with patch("scripts.run_meeting.requests.get", side_effect=run_meeting.requests.ConnectionError("down")):
|
|
with self.assertRaisesRegex(run_meeting.RunnerError, "not reachable"):
|
|
run_meeting.require_ollama("http://127.0.0.1:11434/api/generate", "qwen3.5:9b")
|
|
|
|
def test_missing_model_fails(self) -> None:
|
|
response = response_json({"models": [{"name": "other:model"}]})
|
|
with patch("scripts.run_meeting.requests.get", return_value=response):
|
|
with self.assertRaisesRegex(run_meeting.RunnerError, "not installed"):
|
|
run_meeting.require_ollama("http://127.0.0.1:11434/api/generate", "qwen3.5:9b")
|
|
|
|
def test_loaded_model_fails_preflight_by_default(self) -> None:
|
|
responses = [
|
|
response_json({"models": [{"name": "qwen3.5:9b"}]}),
|
|
response_json({"models": [{"name": "other:model"}]}),
|
|
]
|
|
with patch("scripts.run_meeting.requests.get", side_effect=responses):
|
|
with self.assertRaisesRegex(run_meeting.RunnerError, "loaded model"):
|
|
run_meeting.require_ollama("http://127.0.0.1:11434/api/generate", "qwen3.5:9b")
|
|
|
|
def test_loaded_model_override_permits_execution(self) -> None:
|
|
responses = [
|
|
response_json({"models": [{"name": "qwen3.5:9b"}]}),
|
|
response_json({"models": [{"name": "other:model"}]}),
|
|
]
|
|
with patch("scripts.run_meeting.requests.get", side_effect=responses):
|
|
data = run_meeting.require_ollama(
|
|
"http://127.0.0.1:11434/api/generate",
|
|
"qwen3.5:9b",
|
|
allow_loaded_models=True,
|
|
)
|
|
|
|
self.assertEqual(data["active_models"], ["other:model"])
|
|
self.assertTrue(data["allow_loaded_models"])
|
|
|
|
def test_loaded_model_preflight_failure_records_metadata(self) -> None:
|
|
with workspace_tempdir() as directory:
|
|
root = Path(directory)
|
|
input_path = root / "input.json"
|
|
context = root / "context.yaml"
|
|
write_whisper_json(input_path)
|
|
write_context(context)
|
|
args = run_meeting.parse_args(
|
|
[
|
|
"--input",
|
|
str(input_path),
|
|
"--context",
|
|
str(context),
|
|
"--output-root",
|
|
str(root / "benchmarks"),
|
|
]
|
|
)
|
|
error = run_meeting.RunnerError(
|
|
"Ollama has loaded model(s).",
|
|
details={
|
|
"preflight": {
|
|
"ollama": {
|
|
"active_models": ["other:model"],
|
|
"allow_loaded_models": False,
|
|
}
|
|
}
|
|
},
|
|
)
|
|
|
|
with patch("scripts.run_meeting.require_ollama", side_effect=error):
|
|
code, output_dir, report_path, _protocol = run_meeting.run_pipeline(args)
|
|
|
|
metadata = json.loads((output_dir / "run_metadata.json").read_text(encoding="utf-8"))
|
|
report = report_path.read_text(encoding="utf-8")
|
|
self.assertEqual(code, 2)
|
|
self.assertEqual(metadata["preflight"]["ollama"]["active_models"], ["other:model"])
|
|
self.assertFalse(metadata["preflight"]["ollama"]["allow_loaded_models"])
|
|
self.assertIn("Detected loaded models: `['other:model']`", report)
|
|
self.assertIn("Loaded-model override: `False`", report)
|
|
|
|
def test_stage_failure_preserves_artifacts(self) -> None:
|
|
with workspace_tempdir() as directory:
|
|
root = Path(directory)
|
|
input_path = root / "input.json"
|
|
context = root / "context.yaml"
|
|
write_whisper_json(input_path)
|
|
write_context(context)
|
|
args = run_meeting.parse_args(
|
|
[
|
|
"--input",
|
|
str(input_path),
|
|
"--context",
|
|
str(context),
|
|
"--output-root",
|
|
str(root / "benchmarks"),
|
|
]
|
|
)
|
|
|
|
def fail_extraction(**kwargs):
|
|
marker = kwargs["output_dir"] / "partial.raw.txt"
|
|
marker.parent.mkdir(parents=True, exist_ok=True)
|
|
marker.write_text("partial", encoding="utf-8")
|
|
raise ValueError("model failed")
|
|
|
|
with patch("scripts.run_meeting.require_ollama", return_value={"active_models": []}), patch(
|
|
"scripts.run_meeting.run_extraction_stage",
|
|
side_effect=fail_extraction,
|
|
):
|
|
code, output_dir, report_path, _protocol = run_meeting.run_pipeline(args)
|
|
|
|
self.assertEqual(code, 2)
|
|
self.assertTrue((output_dir / "extractions" / "partial.raw.txt").exists())
|
|
self.assertTrue(report_path.exists())
|
|
self.assertIn("model failed", (output_dir / "run_metadata.json").read_text(encoding="utf-8"))
|
|
|
|
def test_benchmark_report_generation(self) -> None:
|
|
with workspace_tempdir() as directory:
|
|
path = Path(directory) / "report.md"
|
|
run_meeting.write_benchmark_report(
|
|
path,
|
|
{
|
|
"status": "completed",
|
|
"input": "input.json",
|
|
"context": "context.yaml",
|
|
"model": "qwen3.5:9b",
|
|
"output_dir": "out",
|
|
"configuration": {
|
|
"target_chars": 4500,
|
|
"max_chars": 5500,
|
|
"min_chars": 2500,
|
|
"overlap_blocks": 0,
|
|
"temperature": 0.0,
|
|
"num_ctx": 32768,
|
|
},
|
|
"runtime_seconds_by_stage": {"chunking": {"runtime_seconds": 1.0, "status": "passed"}},
|
|
"preflight": {
|
|
"ollama": {"active_models": ["other:model"], "allow_loaded_models": True}
|
|
},
|
|
"total_runtime_seconds": 1.0,
|
|
"average_extraction_seconds_per_chunk": None,
|
|
"chunk_distribution": {},
|
|
"extraction_counts": {},
|
|
"canonicalizer_stats": {},
|
|
"semantic_consolidator": {
|
|
"validator_before": {"valid": True, "violations": []},
|
|
"validator_after": {"valid": True, "violations": []},
|
|
"repair_count": 0,
|
|
},
|
|
"renderer": {"valid": False, "violations": [{"type": "x"}]},
|
|
},
|
|
)
|
|
|
|
text = path.read_text(encoding="utf-8")
|
|
|
|
self.assertIn("# Meeting Lab Benchmark Report", text)
|
|
self.assertIn("Renderer contract valid: False", text)
|
|
self.assertIn("Detected loaded models: `['other:model']`", text)
|
|
self.assertIn("Loaded-model override: `True`", text)
|
|
|
|
def test_metadata_records_loaded_model_condition_and_override(self) -> None:
|
|
with workspace_tempdir() as directory:
|
|
root = Path(directory)
|
|
input_path = root / "input.json"
|
|
context = root / "context.yaml"
|
|
write_whisper_json(input_path)
|
|
write_context(context)
|
|
args = run_meeting.parse_args(
|
|
[
|
|
"--input",
|
|
str(input_path),
|
|
"--context",
|
|
str(context),
|
|
"--output-root",
|
|
str(root / "benchmarks"),
|
|
"--allow-loaded-models",
|
|
]
|
|
)
|
|
|
|
def fake_extraction(**kwargs):
|
|
output = kwargs["output_dir"] / "chunk_01_extraction.json"
|
|
output.parent.mkdir(parents=True, exist_ok=True)
|
|
output.write_text(
|
|
json.dumps(
|
|
{
|
|
"facts": ["Martin | Test fact | clear | evidence"],
|
|
"decisions": [],
|
|
"todos": [],
|
|
"questions": [],
|
|
"positions": [],
|
|
"technical": [],
|
|
"context": {"source_file": str(context)},
|
|
}
|
|
),
|
|
encoding="utf-8",
|
|
)
|
|
return [output]
|
|
|
|
def fake_semantic(**_kwargs):
|
|
return {
|
|
"validator_before": {"valid": True, "violations": []},
|
|
"validator_after": {"valid": True, "violations": []},
|
|
"repair_count": 0,
|
|
"response_metadata": {},
|
|
}
|
|
|
|
def fake_renderer(**kwargs):
|
|
output_dir = kwargs["output_dir"]
|
|
output_dir.mkdir(parents=True, exist_ok=True)
|
|
(output_dir / "validation_report.json").write_text(
|
|
json.dumps({"valid": False, "violations": []}),
|
|
encoding="utf-8",
|
|
)
|
|
return {"valid": False, "runtime_seconds": 0.1, "output_path": None}
|
|
|
|
with patch(
|
|
"scripts.run_meeting.require_ollama",
|
|
return_value={"active_models": ["other:model"], "allow_loaded_models": True},
|
|
), patch(
|
|
"scripts.run_meeting.run_extraction_stage",
|
|
side_effect=fake_extraction,
|
|
), patch(
|
|
"scripts.run_meeting.run_semantic_consolidator_stage",
|
|
side_effect=fake_semantic,
|
|
), patch(
|
|
"scripts.run_meeting.render_working_protocol",
|
|
side_effect=fake_renderer,
|
|
):
|
|
code, output_dir, report_path, _protocol_path = run_meeting.run_pipeline(args)
|
|
|
|
metadata = json.loads((output_dir / "run_metadata.json").read_text(encoding="utf-8"))
|
|
report = report_path.read_text(encoding="utf-8")
|
|
self.assertEqual(code, 0)
|
|
self.assertTrue(metadata["configuration"]["allow_loaded_models"])
|
|
self.assertEqual(metadata["preflight"]["ollama"]["active_models"], ["other:model"])
|
|
self.assertTrue(metadata["preflight"]["ollama"]["allow_loaded_models"])
|
|
self.assertIn("Detected loaded models: `['other:model']`", report)
|
|
self.assertIn("Loaded-model override: `True`", report)
|
|
|
|
def test_successful_run_uses_mocked_llm_stages(self) -> None:
|
|
with workspace_tempdir() as directory:
|
|
root = Path(directory)
|
|
input_path = root / "input.json"
|
|
context = root / "context.yaml"
|
|
write_whisper_json(input_path)
|
|
write_context(context)
|
|
args = run_meeting.parse_args(
|
|
[
|
|
"--input",
|
|
str(input_path),
|
|
"--context",
|
|
str(context),
|
|
"--output-root",
|
|
str(root / "benchmarks"),
|
|
]
|
|
)
|
|
|
|
def fake_extraction(**kwargs):
|
|
output = kwargs["output_dir"] / "chunk_01_extraction.json"
|
|
output.parent.mkdir(parents=True, exist_ok=True)
|
|
output.write_text(
|
|
json.dumps(
|
|
{
|
|
"facts": ["Martin | Test fact | clear | evidence"],
|
|
"decisions": [],
|
|
"todos": [],
|
|
"questions": [],
|
|
"positions": [],
|
|
"technical": [],
|
|
"context": {"source_file": str(context)},
|
|
}
|
|
),
|
|
encoding="utf-8",
|
|
)
|
|
output.with_suffix(".raw.txt").write_text("{}", encoding="utf-8")
|
|
return [output]
|
|
|
|
def fake_semantic(**_kwargs):
|
|
out = root / "placeholder"
|
|
return {
|
|
"validator_before": {"valid": True, "violations": []},
|
|
"validator_after": {"valid": True, "violations": []},
|
|
"repair_count": 0,
|
|
"response_metadata": {},
|
|
"output": str(out),
|
|
}
|
|
|
|
def fake_renderer(**kwargs):
|
|
output_dir = kwargs["output_dir"]
|
|
output_dir.mkdir(parents=True, exist_ok=True)
|
|
(output_dir / "validation_report.json").write_text(
|
|
json.dumps({"valid": False, "violations": [{"type": "missing_required_heading"}]}),
|
|
encoding="utf-8",
|
|
)
|
|
return {"valid": False, "runtime_seconds": 0.1, "output_path": None}
|
|
|
|
with patch("scripts.run_meeting.require_ollama", return_value={"active_models": []}), patch(
|
|
"scripts.run_meeting.run_extraction_stage",
|
|
side_effect=fake_extraction,
|
|
), patch(
|
|
"scripts.run_meeting.run_semantic_consolidator_stage",
|
|
side_effect=fake_semantic,
|
|
), patch(
|
|
"scripts.run_meeting.render_working_protocol",
|
|
side_effect=fake_renderer,
|
|
):
|
|
code, output_dir, report_path, protocol_path = run_meeting.run_pipeline(args)
|
|
|
|
self.assertEqual(code, 0)
|
|
self.assertTrue((output_dir / "canonicalizer" / "canonicalized_extractions.json").exists())
|
|
self.assertTrue(report_path.exists())
|
|
self.assertIsNone(protocol_path)
|
|
|
|
|
|
if __name__ == "__main__":
|
|
unittest.main()
|