- introduce Gold Standard evaluation corpus - document decision taxonomy - define prompt-engineering methodology - add regression workflow - establish Prompt Version 2 baseline - validate decision_simple, decision_deferred and decision_none
52 lines
1.8 KiB
Python
52 lines
1.8 KiB
Python
import json
|
|
import tempfile
|
|
import unittest
|
|
from pathlib import Path
|
|
|
|
from scripts.run_gold_test import read_json_object, scenario_paths, validate_required_keys
|
|
|
|
|
|
class GoldRunnerTests(unittest.TestCase):
|
|
def test_missing_transcript_is_reported(self) -> None:
|
|
with tempfile.TemporaryDirectory() as directory:
|
|
scenario = Path(directory)
|
|
(scenario / "expected.json").write_text("{}", encoding="utf-8")
|
|
|
|
with self.assertRaisesRegex(FileNotFoundError, "Missing transcript.txt"):
|
|
scenario_paths(scenario)
|
|
|
|
def test_missing_expected_is_reported(self) -> None:
|
|
with tempfile.TemporaryDirectory() as directory:
|
|
scenario = Path(directory)
|
|
(scenario / "transcript.txt").write_text("A: Hello.", encoding="utf-8")
|
|
|
|
with self.assertRaisesRegex(FileNotFoundError, "Missing expected.json"):
|
|
scenario_paths(scenario)
|
|
|
|
def test_invalid_expected_json_is_reported(self) -> None:
|
|
with tempfile.TemporaryDirectory() as directory:
|
|
path = Path(directory) / "expected.json"
|
|
path.write_text("{invalid", encoding="utf-8")
|
|
|
|
with self.assertRaisesRegex(ValueError, "Invalid JSON"):
|
|
read_json_object(path)
|
|
|
|
def test_missing_required_schema_keys_are_reported(self) -> None:
|
|
with tempfile.TemporaryDirectory() as directory:
|
|
path = Path(directory) / "expected.json"
|
|
data = {
|
|
"facts": [],
|
|
"decisions": [],
|
|
"todos": [],
|
|
"questions": [],
|
|
"positions": [],
|
|
}
|
|
path.write_text(json.dumps(data), encoding="utf-8")
|
|
|
|
with self.assertRaisesRegex(ValueError, "technical"):
|
|
validate_required_keys(read_json_object(path), path)
|
|
|
|
|
|
if __name__ == "__main__":
|
|
unittest.main()
|