Establish prompt engineering baseline with Gold Standard tests
- introduce Gold Standard evaluation corpus - document decision taxonomy - define prompt-engineering methodology - add regression workflow - establish Prompt Version 2 baseline - validate decision_simple, decision_deferred and decision_none
This commit is contained in:
@@ -0,0 +1,51 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from pathlib import Path
|
||||
from typing import Any, Iterable
|
||||
|
||||
|
||||
PROJECT_ROOT = Path(__file__).resolve().parents[3]
|
||||
PROMPTS_DIR = PROJECT_ROOT / "prompts"
|
||||
|
||||
|
||||
def load_prompt(name: str, prompts_dir: Path = PROMPTS_DIR) -> str:
|
||||
path = prompts_dir / name
|
||||
return path.read_text(encoding="utf-8").strip()
|
||||
|
||||
|
||||
def load_existing_prompts(
|
||||
names: Iterable[str],
|
||||
prompts_dir: Path = PROMPTS_DIR,
|
||||
) -> list[str]:
|
||||
prompts: list[str] = []
|
||||
for name in names:
|
||||
text = load_prompt(name, prompts_dir)
|
||||
if text:
|
||||
prompts.append(text)
|
||||
return prompts
|
||||
|
||||
|
||||
def build_extraction_prompt(
|
||||
source_name: str,
|
||||
transcript: str,
|
||||
output_schema: dict[str, Any],
|
||||
task_prompt_names: Iterable[str] = ("decisions.md",),
|
||||
prompts_dir: Path = PROMPTS_DIR,
|
||||
) -> str:
|
||||
schema_text = json.dumps(output_schema, ensure_ascii=False, indent=2)
|
||||
prompt_parts = [
|
||||
load_prompt("common.md", prompts_dir),
|
||||
*load_existing_prompts(task_prompt_names, prompts_dir),
|
||||
f"""Quelldatei:
|
||||
{source_name}
|
||||
|
||||
Erwartete JSON-Struktur:
|
||||
{schema_text}
|
||||
|
||||
TRANSKRIPT:
|
||||
--- BEGINN TRANSKRIPT ---
|
||||
{transcript}
|
||||
--- ENDE TRANSKRIPT ---""",
|
||||
]
|
||||
return "\n\n".join(part for part in prompt_parts if part).strip() + "\n"
|
||||
|
||||
Reference in New Issue
Block a user