Implement Meeting Context V1 and extraction improvements

Introduce Meeting Context V1 with YAML schema, validation and template.
Support optional --meeting-context during chunk extraction.
Inject authoritative Meeting Context into extraction prompts.
Record Meeting Context provenance in extraction output.
Activate todos.md in shared prompt assembly.
Strengthen responsibility attribution and decision/todo boundaries.
Add focused Gold scenarios and validation tests.
Update architecture and pipeline documentation.
This commit is contained in:
2026-08-01 16:49:48 +02:00
parent 63075eaca9
commit 06f0e7e651
25 changed files with 1453 additions and 22 deletions
+45 -4
View File
@@ -20,6 +20,11 @@ from typing import Any
import requests
from src.meeting_lab.llm.prompts import build_extraction_prompt
from src.meeting_lab.models.meeting_context import (
MeetingContext,
load_meeting_context,
render_meeting_context_for_prompt,
)
DEFAULT_MODEL = "qwen3:8b"
@@ -36,6 +41,8 @@ EXTRACTION_CATEGORIES = (
NORMALIZED_CHUNK_RE = re.compile(r"^(chunk_\d+)_normalized\.txt$")
EXTRACTION_TASK_PROMPT_NAMES = ("decisions.md", "todos.md")
OUTPUT_SCHEMA = {
"chunk": {
@@ -138,11 +145,31 @@ def parse_args() -> argparse.Namespace:
default=32768,
help="Context window tokens per chunk (default: 32768)",
)
parser.add_argument(
"--meeting-context",
type=Path,
help="Optional Meeting Context V1 YAML file to inject into extraction prompts.",
)
return parser.parse_args()
def build_prompt(source_name: str, transcript: str) -> str:
return build_extraction_prompt(source_name, transcript, OUTPUT_SCHEMA)
def build_prompt(
source_name: str,
transcript: str,
meeting_context: MeetingContext | None = None,
) -> str:
context_text = (
render_meeting_context_for_prompt(meeting_context)
if meeting_context is not None
else None
)
return build_extraction_prompt(
source_name,
transcript,
OUTPUT_SCHEMA,
meeting_context=context_text,
task_prompt_names=EXTRACTION_TASK_PROMPT_NAMES,
)
def call_ollama(
@@ -376,12 +403,13 @@ def extract_chunk(
temperature: float,
num_predict: int | None,
num_ctx: int | None,
) -> dict[str, list[str]]:
meeting_context: MeetingContext | None = None,
) -> dict[str, Any]:
transcript = chunk_path.read_text(encoding="utf-8-sig").strip()
if not transcript:
raise ValueError(f"The input file is empty: {chunk_path}")
prompt = build_prompt(chunk_path.name, transcript)
prompt = build_prompt(chunk_path.name, transcript, meeting_context=meeting_context)
raw_text, _metadata = call_ollama(
endpoint=endpoint,
model=model,
@@ -402,6 +430,8 @@ def extract_chunk(
) from exc
extraction = normalize_current_schema(parsed)
if meeting_context is not None:
extraction["context"] = meeting_context.provenance()
output_path.parent.mkdir(parents=True, exist_ok=True)
output_path.write_text(
json.dumps(extraction, ensure_ascii=False, indent=2) + "\n",
@@ -419,6 +449,7 @@ def extract_input(
temperature: float,
num_predict: int | None,
num_ctx: int | None,
meeting_context: MeetingContext | None = None,
) -> list[Path]:
if input_path.is_file():
output_path = output or extraction_path_for_chunk(input_path)
@@ -431,6 +462,7 @@ def extract_input(
temperature,
num_predict,
num_ctx,
meeting_context=meeting_context,
)
return [output_path]
@@ -459,6 +491,7 @@ def extract_input(
temperature,
num_predict,
num_ctx,
meeting_context=meeting_context,
)
output_paths.append(output_path)
@@ -469,6 +502,11 @@ def main() -> int:
args = parse_args()
try:
meeting_context = (
load_meeting_context(args.meeting_context)
if args.meeting_context is not None
else None
)
output_paths = extract_input(
input_path=args.input,
output=args.output,
@@ -478,6 +516,7 @@ def main() -> int:
temperature=args.temperature,
num_predict=args.num_predict,
num_ctx=args.num_ctx,
meeting_context=meeting_context,
)
except requests.ConnectionError:
print(
@@ -496,6 +535,8 @@ def main() -> int:
return 1
print(f"Input: {args.input}")
if args.meeting_context is not None:
print(f"Meeting context: {args.meeting_context}")
print(f"Processed chunks: {len(output_paths)}")
print(f"Extraction JSON files: {len(output_paths)}")
for output_path in output_paths:
+2
View File
@@ -30,12 +30,14 @@ def build_extraction_prompt(
source_name: str,
transcript: str,
output_schema: dict[str, Any],
meeting_context: str | None = None,
task_prompt_names: Iterable[str] = ("decisions.md",),
prompts_dir: Path = PROMPTS_DIR,
) -> str:
schema_text = json.dumps(output_schema, ensure_ascii=False, indent=2)
prompt_parts = [
load_prompt("common.md", prompts_dir),
meeting_context,
*load_existing_prompts(task_prompt_names, prompts_dir),
f"""Quelldatei:
{source_name}
+396
View File
@@ -0,0 +1,396 @@
"""Meeting Context V1 loading, validation and prompt rendering."""
from __future__ import annotations
import ast
from dataclasses import dataclass
from pathlib import Path
from typing import Any
SUPPORTED_SCHEMA_VERSIONS = {"1"}
VALID_ATTENDANCE_STATUSES = {"present", "not_present", "absent"}
class MeetingContextValidationError(ValueError):
"""Raised when a Meeting Context file is structurally invalid."""
@dataclass(frozen=True)
class MeetingContext:
data: dict[str, Any]
source_file: Path
@property
def schema_version(self) -> str:
return str(self.data["schema_version"])
@property
def meeting_id(self) -> str:
return str(self.data["meeting"]["meeting_id"])
def provenance(self) -> dict[str, str]:
return {
"meeting_id": self.meeting_id,
"source_file": str(self.source_file),
"schema_version": self.schema_version,
}
def load_meeting_context(path: Path) -> MeetingContext:
loaded = _load_yaml(path)
if not isinstance(loaded, dict):
raise MeetingContextValidationError("Meeting Context must be a YAML object.")
validate_meeting_context(loaded)
return MeetingContext(data=loaded, source_file=path)
def validate_meeting_context(data: dict[str, Any]) -> None:
schema_version = str(data.get("schema_version", "")).strip()
if schema_version not in SUPPORTED_SCHEMA_VERSIONS:
raise MeetingContextValidationError(
f"Unsupported meeting context schema_version: {schema_version!r}."
)
meeting = _require_mapping(data, "meeting")
_require_non_empty_string(meeting, "meeting.meeting_id")
_require_non_empty_string(meeting, "meeting.title")
_require_non_empty_string(meeting, "meeting.language")
organization = _optional_mapping(data.get("organization"), "organization")
departments = _optional_list(organization.get("departments"), "organization.departments")
department_ids = _collect_department_ids(departments)
participants = _optional_list(data.get("participants"), "participants")
mentioned_people = _optional_list(data.get("mentioned_people"), "mentioned_people")
participant_ids = _collect_unique_ids(participants, "participant_id", "participants")
person_ids = _collect_unique_ids(mentioned_people, "person_id", "mentioned_people")
collisions = sorted(participant_ids & person_ids)
if collisions:
raise MeetingContextValidationError(
"Participant IDs and mentioned-person IDs must not collide: "
+ ", ".join(collisions)
)
for index, participant in enumerate(participants):
item_path = f"participants[{index}]"
_validate_attendance(participant, item_path)
if participant.get("attendance_status") != "present":
raise MeetingContextValidationError(
f"{item_path}.attendance_status must be 'present'."
)
_validate_department_reference(participant, item_path, department_ids)
for index, person in enumerate(mentioned_people):
item_path = f"mentioned_people[{index}]"
_validate_attendance(person, item_path)
if person.get("attendance_status") == "present":
raise MeetingContextValidationError(
f"{item_path}.attendance_status must not be 'present'."
)
_validate_department_reference(person, item_path, department_ids)
def render_meeting_context_for_prompt(context: MeetingContext) -> str:
data = context.data
meeting = data["meeting"]
organization = _optional_mapping(data.get("organization"), "organization")
departments_by_id = {
str(department.get("id")): str(department.get("name"))
for department in _optional_list(
organization.get("departments"), "organization.departments"
)
if isinstance(department, dict) and department.get("id") and department.get("name")
}
lines = [
"MEETING CONTEXT V1 (AUTHORITATIVE METADATA)",
"",
"Rules:",
"- The participant list is authoritative.",
"- Mentioned people did not attend this meeting.",
"- Roles and departments must not be inferred or changed.",
"- Discussion of a department does not establish responsibility.",
"- An action item may name a responsible person only when assignment or acceptance is explicit in the transcript.",
"- Objections, suggestions and expertise do not establish ownership.",
"",
"Meeting:",
f"- Title: {_text(meeting.get('title'))}",
f"- Language: {_text(meeting.get('language'))}",
]
objective = _text(meeting.get("objective"))
if objective:
lines.append(f"- Objective: {objective}")
participants = _optional_list(data.get("participants"), "participants")
if participants:
lines.extend(["", "Actual participants:"])
for participant in participants:
lines.append(_render_person_line(participant, "participant_id", departments_by_id))
mentioned_people = _optional_list(data.get("mentioned_people"), "mentioned_people")
if mentioned_people:
lines.extend(["", "Mentioned but absent people:"])
for person in mentioned_people:
lines.append(_render_person_line(person, "person_id", departments_by_id))
abbreviations = _optional_mapping(organization.get("abbreviations"), "organization.abbreviations")
abbreviation_lines = [
f"- {key}: {_text(value)}"
for key, value in sorted(abbreviations.items())
if _text(value)
]
if abbreviation_lines:
lines.extend(["", "Abbreviations:", *abbreviation_lines])
known_entities = _optional_mapping(data.get("known_entities"), "known_entities")
entity_lines = []
for key in sorted(known_entities):
values = [_text(value) for value in _optional_list(known_entities.get(key), key)]
values = [value for value in values if value]
if values:
entity_lines.append(f"- {key}: {', '.join(values)}")
if entity_lines:
lines.extend(["", "Relevant known entities:", *entity_lines])
context_rules = _optional_mapping(data.get("context_rules"), "context_rules")
rule_lines = [
f"- {key}: {str(value).lower() if isinstance(value, bool) else _text(value)}"
for key, value in sorted(context_rules.items())
]
if rule_lines:
lines.extend(["", "Context rules:", *rule_lines])
return "\n".join(lines).strip() + "\n"
def _render_person_line(
person: dict[str, Any],
id_key: str,
departments_by_id: dict[str, str],
) -> str:
parts = [_text(person.get("display_name"))]
aliases = [_text(alias) for alias in _optional_list(person.get("aliases"), "aliases")]
aliases = [alias for alias in aliases if alias]
if aliases:
parts.append(f"aliases: {', '.join(aliases)}")
roles = _roles(person)
if roles:
parts.append(f"roles: {', '.join(roles)}")
department_id = _text(person.get("department_id") or person.get("department"))
if department_id:
department_name = departments_by_id.get(department_id, department_id)
parts.append(f"department: {department_name}")
identifier = _text(person.get(id_key))
if identifier:
parts.append(f"id: {identifier}")
return "- " + "; ".join(part for part in parts if part)
def _roles(person: dict[str, Any]) -> list[str]:
roles = [_text(role) for role in _optional_list(person.get("roles"), "roles")]
role = _text(person.get("role"))
if role:
roles.insert(0, role)
return [role for role in roles if role]
def _load_yaml(path: Path) -> Any:
text = path.read_text(encoding="utf-8-sig")
try:
import yaml # type: ignore[import-not-found]
except ModuleNotFoundError:
return _parse_simple_yaml(text)
return yaml.safe_load(text)
def _collect_department_ids(departments: list[Any]) -> set[str]:
ids: set[str] = set()
for index, department in enumerate(departments):
if not isinstance(department, dict):
raise MeetingContextValidationError(
f"organization.departments[{index}] must be an object."
)
department_id = _text(department.get("id"))
if not department_id:
raise MeetingContextValidationError(
f"organization.departments[{index}].id must be non-empty."
)
if department_id in ids:
raise MeetingContextValidationError(
f"Duplicate department id: {department_id!r}."
)
ids.add(department_id)
return ids
def _collect_unique_ids(items: list[Any], key: str, path: str) -> set[str]:
ids: set[str] = set()
for index, item in enumerate(items):
if not isinstance(item, dict):
raise MeetingContextValidationError(f"{path}[{index}] must be an object.")
identifier = _text(item.get(key))
if not identifier:
raise MeetingContextValidationError(f"{path}[{index}].{key} must be non-empty.")
if identifier in ids:
raise MeetingContextValidationError(f"Duplicate {key}: {identifier!r}.")
ids.add(identifier)
return ids
def _validate_attendance(item: dict[str, Any], path: str) -> None:
status = item.get("attendance_status")
if status not in VALID_ATTENDANCE_STATUSES:
raise MeetingContextValidationError(
f"{path}.attendance_status has invalid value: {status!r}."
)
def _validate_department_reference(
item: dict[str, Any],
path: str,
department_ids: set[str],
) -> None:
department_id = _text(item.get("department_id") or item.get("department"))
if department_id and department_id not in department_ids:
raise MeetingContextValidationError(
f"{path}.department_id references unknown department: {department_id!r}."
)
def _require_mapping(data: dict[str, Any], key: str) -> dict[str, Any]:
value = data.get(key)
if not isinstance(value, dict):
raise MeetingContextValidationError(f"{key} must be an object.")
return value
def _optional_mapping(value: Any, path: str) -> dict[str, Any]:
if value is None:
return {}
if not isinstance(value, dict):
raise MeetingContextValidationError(f"{path} must be an object.")
return value
def _optional_list(value: Any, path: str) -> list[Any]:
if value is None:
return []
if not isinstance(value, list):
raise MeetingContextValidationError(f"{path} must be a list.")
return value
def _require_non_empty_string(data: dict[str, Any], key_path: str) -> None:
key = key_path.split(".")[-1]
if not _text(data.get(key)):
raise MeetingContextValidationError(f"{key_path} must be present and non-empty.")
def _text(value: Any) -> str:
if value is None:
return ""
return str(value).strip()
def _parse_simple_yaml(text: str) -> Any:
lines = []
for raw_line in text.splitlines():
stripped = _strip_yaml_comment(raw_line.rstrip())
if stripped.strip():
lines.append((len(stripped) - len(stripped.lstrip(" ")), stripped.lstrip(" ")))
if not lines:
return None
parsed, index = _parse_yaml_block(lines, 0, lines[0][0])
if index != len(lines):
raise MeetingContextValidationError("Could not parse Meeting Context YAML.")
return parsed
def _parse_yaml_block(
lines: list[tuple[int, str]],
index: int,
indent: int,
) -> tuple[Any, int]:
if lines[index][1].startswith("- "):
result = []
while index < len(lines) and lines[index][0] == indent and lines[index][1].startswith("- "):
content = lines[index][1][2:].strip()
index += 1
if not content:
value, index = _parse_yaml_block(lines, index, lines[index][0])
result.append(value)
continue
if ":" in content:
key, raw_value = content.split(":", 1)
item = {key.strip(): _parse_scalar(raw_value.strip()) if raw_value.strip() else {}}
while index < len(lines) and lines[index][0] > indent:
child_indent, child_content = lines[index]
if child_content.startswith("- "):
break
child_key, child_raw_value = child_content.split(":", 1)
index += 1
child_raw_value = child_raw_value.strip()
if child_raw_value:
item[child_key.strip()] = _parse_scalar(child_raw_value)
elif index < len(lines) and lines[index][0] > child_indent:
item[child_key.strip()], index = _parse_yaml_block(lines, index, lines[index][0])
else:
item[child_key.strip()] = None
result.append(item)
else:
result.append(_parse_scalar(content))
return result, index
result = {}
while index < len(lines) and lines[index][0] == indent and not lines[index][1].startswith("- "):
key, raw_value = lines[index][1].split(":", 1)
index += 1
raw_value = raw_value.strip()
if raw_value:
result[key.strip()] = _parse_scalar(raw_value)
elif index < len(lines) and lines[index][0] > indent:
result[key.strip()], index = _parse_yaml_block(lines, index, lines[index][0])
else:
result[key.strip()] = None
return result, index
def _parse_scalar(value: str) -> Any:
if value in {"null", "Null", "NULL", "~"}:
return None
if value in {"true", "True", "TRUE"}:
return True
if value in {"false", "False", "FALSE"}:
return False
if value in {"[]", "{}"} or (
value.startswith("[") and value.endswith("]")
):
return ast.literal_eval(value)
if (
(value.startswith('"') and value.endswith('"'))
or (value.startswith("'") and value.endswith("'"))
):
return ast.literal_eval(value)
return value
def _strip_yaml_comment(line: str) -> str:
in_single = False
in_double = False
for index, char in enumerate(line):
if char == "'" and not in_double:
in_single = not in_single
elif char == '"' and not in_single:
in_double = not in_double
elif char == "#" and not in_single and not in_double:
return line[:index].rstrip()
return line