Introduce Meeting Context V1 with YAML schema, validation and template. Support optional --meeting-context during chunk extraction. Inject authoritative Meeting Context into extraction prompts. Record Meeting Context provenance in extraction output. Activate todos.md in shared prompt assembly. Strengthen responsibility attribution and decision/todo boundaries. Add focused Gold scenarios and validation tests. Update architecture and pipeline documentation.
397 lines
14 KiB
Python
397 lines
14 KiB
Python
"""Meeting Context V1 loading, validation and prompt rendering."""
|
|
|
|
from __future__ import annotations
|
|
|
|
import ast
|
|
from dataclasses import dataclass
|
|
from pathlib import Path
|
|
from typing import Any
|
|
|
|
|
|
SUPPORTED_SCHEMA_VERSIONS = {"1"}
|
|
VALID_ATTENDANCE_STATUSES = {"present", "not_present", "absent"}
|
|
|
|
|
|
class MeetingContextValidationError(ValueError):
|
|
"""Raised when a Meeting Context file is structurally invalid."""
|
|
|
|
|
|
@dataclass(frozen=True)
|
|
class MeetingContext:
|
|
data: dict[str, Any]
|
|
source_file: Path
|
|
|
|
@property
|
|
def schema_version(self) -> str:
|
|
return str(self.data["schema_version"])
|
|
|
|
@property
|
|
def meeting_id(self) -> str:
|
|
return str(self.data["meeting"]["meeting_id"])
|
|
|
|
def provenance(self) -> dict[str, str]:
|
|
return {
|
|
"meeting_id": self.meeting_id,
|
|
"source_file": str(self.source_file),
|
|
"schema_version": self.schema_version,
|
|
}
|
|
|
|
|
|
def load_meeting_context(path: Path) -> MeetingContext:
|
|
loaded = _load_yaml(path)
|
|
if not isinstance(loaded, dict):
|
|
raise MeetingContextValidationError("Meeting Context must be a YAML object.")
|
|
|
|
validate_meeting_context(loaded)
|
|
return MeetingContext(data=loaded, source_file=path)
|
|
|
|
|
|
def validate_meeting_context(data: dict[str, Any]) -> None:
|
|
schema_version = str(data.get("schema_version", "")).strip()
|
|
if schema_version not in SUPPORTED_SCHEMA_VERSIONS:
|
|
raise MeetingContextValidationError(
|
|
f"Unsupported meeting context schema_version: {schema_version!r}."
|
|
)
|
|
|
|
meeting = _require_mapping(data, "meeting")
|
|
_require_non_empty_string(meeting, "meeting.meeting_id")
|
|
_require_non_empty_string(meeting, "meeting.title")
|
|
_require_non_empty_string(meeting, "meeting.language")
|
|
|
|
organization = _optional_mapping(data.get("organization"), "organization")
|
|
departments = _optional_list(organization.get("departments"), "organization.departments")
|
|
department_ids = _collect_department_ids(departments)
|
|
|
|
participants = _optional_list(data.get("participants"), "participants")
|
|
mentioned_people = _optional_list(data.get("mentioned_people"), "mentioned_people")
|
|
|
|
participant_ids = _collect_unique_ids(participants, "participant_id", "participants")
|
|
person_ids = _collect_unique_ids(mentioned_people, "person_id", "mentioned_people")
|
|
collisions = sorted(participant_ids & person_ids)
|
|
if collisions:
|
|
raise MeetingContextValidationError(
|
|
"Participant IDs and mentioned-person IDs must not collide: "
|
|
+ ", ".join(collisions)
|
|
)
|
|
|
|
for index, participant in enumerate(participants):
|
|
item_path = f"participants[{index}]"
|
|
_validate_attendance(participant, item_path)
|
|
if participant.get("attendance_status") != "present":
|
|
raise MeetingContextValidationError(
|
|
f"{item_path}.attendance_status must be 'present'."
|
|
)
|
|
_validate_department_reference(participant, item_path, department_ids)
|
|
|
|
for index, person in enumerate(mentioned_people):
|
|
item_path = f"mentioned_people[{index}]"
|
|
_validate_attendance(person, item_path)
|
|
if person.get("attendance_status") == "present":
|
|
raise MeetingContextValidationError(
|
|
f"{item_path}.attendance_status must not be 'present'."
|
|
)
|
|
_validate_department_reference(person, item_path, department_ids)
|
|
|
|
|
|
def render_meeting_context_for_prompt(context: MeetingContext) -> str:
|
|
data = context.data
|
|
meeting = data["meeting"]
|
|
organization = _optional_mapping(data.get("organization"), "organization")
|
|
departments_by_id = {
|
|
str(department.get("id")): str(department.get("name"))
|
|
for department in _optional_list(
|
|
organization.get("departments"), "organization.departments"
|
|
)
|
|
if isinstance(department, dict) and department.get("id") and department.get("name")
|
|
}
|
|
|
|
lines = [
|
|
"MEETING CONTEXT V1 (AUTHORITATIVE METADATA)",
|
|
"",
|
|
"Rules:",
|
|
"- The participant list is authoritative.",
|
|
"- Mentioned people did not attend this meeting.",
|
|
"- Roles and departments must not be inferred or changed.",
|
|
"- Discussion of a department does not establish responsibility.",
|
|
"- An action item may name a responsible person only when assignment or acceptance is explicit in the transcript.",
|
|
"- Objections, suggestions and expertise do not establish ownership.",
|
|
"",
|
|
"Meeting:",
|
|
f"- Title: {_text(meeting.get('title'))}",
|
|
f"- Language: {_text(meeting.get('language'))}",
|
|
]
|
|
|
|
objective = _text(meeting.get("objective"))
|
|
if objective:
|
|
lines.append(f"- Objective: {objective}")
|
|
|
|
participants = _optional_list(data.get("participants"), "participants")
|
|
if participants:
|
|
lines.extend(["", "Actual participants:"])
|
|
for participant in participants:
|
|
lines.append(_render_person_line(participant, "participant_id", departments_by_id))
|
|
|
|
mentioned_people = _optional_list(data.get("mentioned_people"), "mentioned_people")
|
|
if mentioned_people:
|
|
lines.extend(["", "Mentioned but absent people:"])
|
|
for person in mentioned_people:
|
|
lines.append(_render_person_line(person, "person_id", departments_by_id))
|
|
|
|
abbreviations = _optional_mapping(organization.get("abbreviations"), "organization.abbreviations")
|
|
abbreviation_lines = [
|
|
f"- {key}: {_text(value)}"
|
|
for key, value in sorted(abbreviations.items())
|
|
if _text(value)
|
|
]
|
|
if abbreviation_lines:
|
|
lines.extend(["", "Abbreviations:", *abbreviation_lines])
|
|
|
|
known_entities = _optional_mapping(data.get("known_entities"), "known_entities")
|
|
entity_lines = []
|
|
for key in sorted(known_entities):
|
|
values = [_text(value) for value in _optional_list(known_entities.get(key), key)]
|
|
values = [value for value in values if value]
|
|
if values:
|
|
entity_lines.append(f"- {key}: {', '.join(values)}")
|
|
if entity_lines:
|
|
lines.extend(["", "Relevant known entities:", *entity_lines])
|
|
|
|
context_rules = _optional_mapping(data.get("context_rules"), "context_rules")
|
|
rule_lines = [
|
|
f"- {key}: {str(value).lower() if isinstance(value, bool) else _text(value)}"
|
|
for key, value in sorted(context_rules.items())
|
|
]
|
|
if rule_lines:
|
|
lines.extend(["", "Context rules:", *rule_lines])
|
|
|
|
return "\n".join(lines).strip() + "\n"
|
|
|
|
|
|
def _render_person_line(
|
|
person: dict[str, Any],
|
|
id_key: str,
|
|
departments_by_id: dict[str, str],
|
|
) -> str:
|
|
parts = [_text(person.get("display_name"))]
|
|
aliases = [_text(alias) for alias in _optional_list(person.get("aliases"), "aliases")]
|
|
aliases = [alias for alias in aliases if alias]
|
|
if aliases:
|
|
parts.append(f"aliases: {', '.join(aliases)}")
|
|
|
|
roles = _roles(person)
|
|
if roles:
|
|
parts.append(f"roles: {', '.join(roles)}")
|
|
|
|
department_id = _text(person.get("department_id") or person.get("department"))
|
|
if department_id:
|
|
department_name = departments_by_id.get(department_id, department_id)
|
|
parts.append(f"department: {department_name}")
|
|
|
|
identifier = _text(person.get(id_key))
|
|
if identifier:
|
|
parts.append(f"id: {identifier}")
|
|
|
|
return "- " + "; ".join(part for part in parts if part)
|
|
|
|
|
|
def _roles(person: dict[str, Any]) -> list[str]:
|
|
roles = [_text(role) for role in _optional_list(person.get("roles"), "roles")]
|
|
role = _text(person.get("role"))
|
|
if role:
|
|
roles.insert(0, role)
|
|
return [role for role in roles if role]
|
|
|
|
|
|
def _load_yaml(path: Path) -> Any:
|
|
text = path.read_text(encoding="utf-8-sig")
|
|
try:
|
|
import yaml # type: ignore[import-not-found]
|
|
except ModuleNotFoundError:
|
|
return _parse_simple_yaml(text)
|
|
return yaml.safe_load(text)
|
|
|
|
|
|
def _collect_department_ids(departments: list[Any]) -> set[str]:
|
|
ids: set[str] = set()
|
|
for index, department in enumerate(departments):
|
|
if not isinstance(department, dict):
|
|
raise MeetingContextValidationError(
|
|
f"organization.departments[{index}] must be an object."
|
|
)
|
|
department_id = _text(department.get("id"))
|
|
if not department_id:
|
|
raise MeetingContextValidationError(
|
|
f"organization.departments[{index}].id must be non-empty."
|
|
)
|
|
if department_id in ids:
|
|
raise MeetingContextValidationError(
|
|
f"Duplicate department id: {department_id!r}."
|
|
)
|
|
ids.add(department_id)
|
|
return ids
|
|
|
|
|
|
def _collect_unique_ids(items: list[Any], key: str, path: str) -> set[str]:
|
|
ids: set[str] = set()
|
|
for index, item in enumerate(items):
|
|
if not isinstance(item, dict):
|
|
raise MeetingContextValidationError(f"{path}[{index}] must be an object.")
|
|
identifier = _text(item.get(key))
|
|
if not identifier:
|
|
raise MeetingContextValidationError(f"{path}[{index}].{key} must be non-empty.")
|
|
if identifier in ids:
|
|
raise MeetingContextValidationError(f"Duplicate {key}: {identifier!r}.")
|
|
ids.add(identifier)
|
|
return ids
|
|
|
|
|
|
def _validate_attendance(item: dict[str, Any], path: str) -> None:
|
|
status = item.get("attendance_status")
|
|
if status not in VALID_ATTENDANCE_STATUSES:
|
|
raise MeetingContextValidationError(
|
|
f"{path}.attendance_status has invalid value: {status!r}."
|
|
)
|
|
|
|
|
|
def _validate_department_reference(
|
|
item: dict[str, Any],
|
|
path: str,
|
|
department_ids: set[str],
|
|
) -> None:
|
|
department_id = _text(item.get("department_id") or item.get("department"))
|
|
if department_id and department_id not in department_ids:
|
|
raise MeetingContextValidationError(
|
|
f"{path}.department_id references unknown department: {department_id!r}."
|
|
)
|
|
|
|
|
|
def _require_mapping(data: dict[str, Any], key: str) -> dict[str, Any]:
|
|
value = data.get(key)
|
|
if not isinstance(value, dict):
|
|
raise MeetingContextValidationError(f"{key} must be an object.")
|
|
return value
|
|
|
|
|
|
def _optional_mapping(value: Any, path: str) -> dict[str, Any]:
|
|
if value is None:
|
|
return {}
|
|
if not isinstance(value, dict):
|
|
raise MeetingContextValidationError(f"{path} must be an object.")
|
|
return value
|
|
|
|
|
|
def _optional_list(value: Any, path: str) -> list[Any]:
|
|
if value is None:
|
|
return []
|
|
if not isinstance(value, list):
|
|
raise MeetingContextValidationError(f"{path} must be a list.")
|
|
return value
|
|
|
|
|
|
def _require_non_empty_string(data: dict[str, Any], key_path: str) -> None:
|
|
key = key_path.split(".")[-1]
|
|
if not _text(data.get(key)):
|
|
raise MeetingContextValidationError(f"{key_path} must be present and non-empty.")
|
|
|
|
|
|
def _text(value: Any) -> str:
|
|
if value is None:
|
|
return ""
|
|
return str(value).strip()
|
|
|
|
|
|
def _parse_simple_yaml(text: str) -> Any:
|
|
lines = []
|
|
for raw_line in text.splitlines():
|
|
stripped = _strip_yaml_comment(raw_line.rstrip())
|
|
if stripped.strip():
|
|
lines.append((len(stripped) - len(stripped.lstrip(" ")), stripped.lstrip(" ")))
|
|
if not lines:
|
|
return None
|
|
parsed, index = _parse_yaml_block(lines, 0, lines[0][0])
|
|
if index != len(lines):
|
|
raise MeetingContextValidationError("Could not parse Meeting Context YAML.")
|
|
return parsed
|
|
|
|
|
|
def _parse_yaml_block(
|
|
lines: list[tuple[int, str]],
|
|
index: int,
|
|
indent: int,
|
|
) -> tuple[Any, int]:
|
|
if lines[index][1].startswith("- "):
|
|
result = []
|
|
while index < len(lines) and lines[index][0] == indent and lines[index][1].startswith("- "):
|
|
content = lines[index][1][2:].strip()
|
|
index += 1
|
|
if not content:
|
|
value, index = _parse_yaml_block(lines, index, lines[index][0])
|
|
result.append(value)
|
|
continue
|
|
|
|
if ":" in content:
|
|
key, raw_value = content.split(":", 1)
|
|
item = {key.strip(): _parse_scalar(raw_value.strip()) if raw_value.strip() else {}}
|
|
while index < len(lines) and lines[index][0] > indent:
|
|
child_indent, child_content = lines[index]
|
|
if child_content.startswith("- "):
|
|
break
|
|
child_key, child_raw_value = child_content.split(":", 1)
|
|
index += 1
|
|
child_raw_value = child_raw_value.strip()
|
|
if child_raw_value:
|
|
item[child_key.strip()] = _parse_scalar(child_raw_value)
|
|
elif index < len(lines) and lines[index][0] > child_indent:
|
|
item[child_key.strip()], index = _parse_yaml_block(lines, index, lines[index][0])
|
|
else:
|
|
item[child_key.strip()] = None
|
|
result.append(item)
|
|
else:
|
|
result.append(_parse_scalar(content))
|
|
return result, index
|
|
|
|
result = {}
|
|
while index < len(lines) and lines[index][0] == indent and not lines[index][1].startswith("- "):
|
|
key, raw_value = lines[index][1].split(":", 1)
|
|
index += 1
|
|
raw_value = raw_value.strip()
|
|
if raw_value:
|
|
result[key.strip()] = _parse_scalar(raw_value)
|
|
elif index < len(lines) and lines[index][0] > indent:
|
|
result[key.strip()], index = _parse_yaml_block(lines, index, lines[index][0])
|
|
else:
|
|
result[key.strip()] = None
|
|
return result, index
|
|
|
|
|
|
def _parse_scalar(value: str) -> Any:
|
|
if value in {"null", "Null", "NULL", "~"}:
|
|
return None
|
|
if value in {"true", "True", "TRUE"}:
|
|
return True
|
|
if value in {"false", "False", "FALSE"}:
|
|
return False
|
|
if value in {"[]", "{}"} or (
|
|
value.startswith("[") and value.endswith("]")
|
|
):
|
|
return ast.literal_eval(value)
|
|
if (
|
|
(value.startswith('"') and value.endswith('"'))
|
|
or (value.startswith("'") and value.endswith("'"))
|
|
):
|
|
return ast.literal_eval(value)
|
|
return value
|
|
|
|
|
|
def _strip_yaml_comment(line: str) -> str:
|
|
in_single = False
|
|
in_double = False
|
|
for index, char in enumerate(line):
|
|
if char == "'" and not in_double:
|
|
in_single = not in_single
|
|
elif char == '"' and not in_single:
|
|
in_double = not in_double
|
|
elif char == "#" and not in_single and not in_double:
|
|
return line[:index].rstrip()
|
|
return line
|