Implement Meeting Context V1 and extraction improvements
Introduce Meeting Context V1 with YAML schema, validation and template. Support optional --meeting-context during chunk extraction. Inject authoritative Meeting Context into extraction prompts. Record Meeting Context provenance in extraction output. Activate todos.md in shared prompt assembly. Strengthen responsibility attribution and decision/todo boundaries. Add focused Gold scenarios and validation tests. Update architecture and pipeline documentation.
This commit is contained in:
@@ -0,0 +1,396 @@
|
||||
"""Meeting Context V1 loading, validation and prompt rendering."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import ast
|
||||
from dataclasses import dataclass
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
|
||||
SUPPORTED_SCHEMA_VERSIONS = {"1"}
|
||||
VALID_ATTENDANCE_STATUSES = {"present", "not_present", "absent"}
|
||||
|
||||
|
||||
class MeetingContextValidationError(ValueError):
|
||||
"""Raised when a Meeting Context file is structurally invalid."""
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class MeetingContext:
|
||||
data: dict[str, Any]
|
||||
source_file: Path
|
||||
|
||||
@property
|
||||
def schema_version(self) -> str:
|
||||
return str(self.data["schema_version"])
|
||||
|
||||
@property
|
||||
def meeting_id(self) -> str:
|
||||
return str(self.data["meeting"]["meeting_id"])
|
||||
|
||||
def provenance(self) -> dict[str, str]:
|
||||
return {
|
||||
"meeting_id": self.meeting_id,
|
||||
"source_file": str(self.source_file),
|
||||
"schema_version": self.schema_version,
|
||||
}
|
||||
|
||||
|
||||
def load_meeting_context(path: Path) -> MeetingContext:
|
||||
loaded = _load_yaml(path)
|
||||
if not isinstance(loaded, dict):
|
||||
raise MeetingContextValidationError("Meeting Context must be a YAML object.")
|
||||
|
||||
validate_meeting_context(loaded)
|
||||
return MeetingContext(data=loaded, source_file=path)
|
||||
|
||||
|
||||
def validate_meeting_context(data: dict[str, Any]) -> None:
|
||||
schema_version = str(data.get("schema_version", "")).strip()
|
||||
if schema_version not in SUPPORTED_SCHEMA_VERSIONS:
|
||||
raise MeetingContextValidationError(
|
||||
f"Unsupported meeting context schema_version: {schema_version!r}."
|
||||
)
|
||||
|
||||
meeting = _require_mapping(data, "meeting")
|
||||
_require_non_empty_string(meeting, "meeting.meeting_id")
|
||||
_require_non_empty_string(meeting, "meeting.title")
|
||||
_require_non_empty_string(meeting, "meeting.language")
|
||||
|
||||
organization = _optional_mapping(data.get("organization"), "organization")
|
||||
departments = _optional_list(organization.get("departments"), "organization.departments")
|
||||
department_ids = _collect_department_ids(departments)
|
||||
|
||||
participants = _optional_list(data.get("participants"), "participants")
|
||||
mentioned_people = _optional_list(data.get("mentioned_people"), "mentioned_people")
|
||||
|
||||
participant_ids = _collect_unique_ids(participants, "participant_id", "participants")
|
||||
person_ids = _collect_unique_ids(mentioned_people, "person_id", "mentioned_people")
|
||||
collisions = sorted(participant_ids & person_ids)
|
||||
if collisions:
|
||||
raise MeetingContextValidationError(
|
||||
"Participant IDs and mentioned-person IDs must not collide: "
|
||||
+ ", ".join(collisions)
|
||||
)
|
||||
|
||||
for index, participant in enumerate(participants):
|
||||
item_path = f"participants[{index}]"
|
||||
_validate_attendance(participant, item_path)
|
||||
if participant.get("attendance_status") != "present":
|
||||
raise MeetingContextValidationError(
|
||||
f"{item_path}.attendance_status must be 'present'."
|
||||
)
|
||||
_validate_department_reference(participant, item_path, department_ids)
|
||||
|
||||
for index, person in enumerate(mentioned_people):
|
||||
item_path = f"mentioned_people[{index}]"
|
||||
_validate_attendance(person, item_path)
|
||||
if person.get("attendance_status") == "present":
|
||||
raise MeetingContextValidationError(
|
||||
f"{item_path}.attendance_status must not be 'present'."
|
||||
)
|
||||
_validate_department_reference(person, item_path, department_ids)
|
||||
|
||||
|
||||
def render_meeting_context_for_prompt(context: MeetingContext) -> str:
|
||||
data = context.data
|
||||
meeting = data["meeting"]
|
||||
organization = _optional_mapping(data.get("organization"), "organization")
|
||||
departments_by_id = {
|
||||
str(department.get("id")): str(department.get("name"))
|
||||
for department in _optional_list(
|
||||
organization.get("departments"), "organization.departments"
|
||||
)
|
||||
if isinstance(department, dict) and department.get("id") and department.get("name")
|
||||
}
|
||||
|
||||
lines = [
|
||||
"MEETING CONTEXT V1 (AUTHORITATIVE METADATA)",
|
||||
"",
|
||||
"Rules:",
|
||||
"- The participant list is authoritative.",
|
||||
"- Mentioned people did not attend this meeting.",
|
||||
"- Roles and departments must not be inferred or changed.",
|
||||
"- Discussion of a department does not establish responsibility.",
|
||||
"- An action item may name a responsible person only when assignment or acceptance is explicit in the transcript.",
|
||||
"- Objections, suggestions and expertise do not establish ownership.",
|
||||
"",
|
||||
"Meeting:",
|
||||
f"- Title: {_text(meeting.get('title'))}",
|
||||
f"- Language: {_text(meeting.get('language'))}",
|
||||
]
|
||||
|
||||
objective = _text(meeting.get("objective"))
|
||||
if objective:
|
||||
lines.append(f"- Objective: {objective}")
|
||||
|
||||
participants = _optional_list(data.get("participants"), "participants")
|
||||
if participants:
|
||||
lines.extend(["", "Actual participants:"])
|
||||
for participant in participants:
|
||||
lines.append(_render_person_line(participant, "participant_id", departments_by_id))
|
||||
|
||||
mentioned_people = _optional_list(data.get("mentioned_people"), "mentioned_people")
|
||||
if mentioned_people:
|
||||
lines.extend(["", "Mentioned but absent people:"])
|
||||
for person in mentioned_people:
|
||||
lines.append(_render_person_line(person, "person_id", departments_by_id))
|
||||
|
||||
abbreviations = _optional_mapping(organization.get("abbreviations"), "organization.abbreviations")
|
||||
abbreviation_lines = [
|
||||
f"- {key}: {_text(value)}"
|
||||
for key, value in sorted(abbreviations.items())
|
||||
if _text(value)
|
||||
]
|
||||
if abbreviation_lines:
|
||||
lines.extend(["", "Abbreviations:", *abbreviation_lines])
|
||||
|
||||
known_entities = _optional_mapping(data.get("known_entities"), "known_entities")
|
||||
entity_lines = []
|
||||
for key in sorted(known_entities):
|
||||
values = [_text(value) for value in _optional_list(known_entities.get(key), key)]
|
||||
values = [value for value in values if value]
|
||||
if values:
|
||||
entity_lines.append(f"- {key}: {', '.join(values)}")
|
||||
if entity_lines:
|
||||
lines.extend(["", "Relevant known entities:", *entity_lines])
|
||||
|
||||
context_rules = _optional_mapping(data.get("context_rules"), "context_rules")
|
||||
rule_lines = [
|
||||
f"- {key}: {str(value).lower() if isinstance(value, bool) else _text(value)}"
|
||||
for key, value in sorted(context_rules.items())
|
||||
]
|
||||
if rule_lines:
|
||||
lines.extend(["", "Context rules:", *rule_lines])
|
||||
|
||||
return "\n".join(lines).strip() + "\n"
|
||||
|
||||
|
||||
def _render_person_line(
|
||||
person: dict[str, Any],
|
||||
id_key: str,
|
||||
departments_by_id: dict[str, str],
|
||||
) -> str:
|
||||
parts = [_text(person.get("display_name"))]
|
||||
aliases = [_text(alias) for alias in _optional_list(person.get("aliases"), "aliases")]
|
||||
aliases = [alias for alias in aliases if alias]
|
||||
if aliases:
|
||||
parts.append(f"aliases: {', '.join(aliases)}")
|
||||
|
||||
roles = _roles(person)
|
||||
if roles:
|
||||
parts.append(f"roles: {', '.join(roles)}")
|
||||
|
||||
department_id = _text(person.get("department_id") or person.get("department"))
|
||||
if department_id:
|
||||
department_name = departments_by_id.get(department_id, department_id)
|
||||
parts.append(f"department: {department_name}")
|
||||
|
||||
identifier = _text(person.get(id_key))
|
||||
if identifier:
|
||||
parts.append(f"id: {identifier}")
|
||||
|
||||
return "- " + "; ".join(part for part in parts if part)
|
||||
|
||||
|
||||
def _roles(person: dict[str, Any]) -> list[str]:
|
||||
roles = [_text(role) for role in _optional_list(person.get("roles"), "roles")]
|
||||
role = _text(person.get("role"))
|
||||
if role:
|
||||
roles.insert(0, role)
|
||||
return [role for role in roles if role]
|
||||
|
||||
|
||||
def _load_yaml(path: Path) -> Any:
|
||||
text = path.read_text(encoding="utf-8-sig")
|
||||
try:
|
||||
import yaml # type: ignore[import-not-found]
|
||||
except ModuleNotFoundError:
|
||||
return _parse_simple_yaml(text)
|
||||
return yaml.safe_load(text)
|
||||
|
||||
|
||||
def _collect_department_ids(departments: list[Any]) -> set[str]:
|
||||
ids: set[str] = set()
|
||||
for index, department in enumerate(departments):
|
||||
if not isinstance(department, dict):
|
||||
raise MeetingContextValidationError(
|
||||
f"organization.departments[{index}] must be an object."
|
||||
)
|
||||
department_id = _text(department.get("id"))
|
||||
if not department_id:
|
||||
raise MeetingContextValidationError(
|
||||
f"organization.departments[{index}].id must be non-empty."
|
||||
)
|
||||
if department_id in ids:
|
||||
raise MeetingContextValidationError(
|
||||
f"Duplicate department id: {department_id!r}."
|
||||
)
|
||||
ids.add(department_id)
|
||||
return ids
|
||||
|
||||
|
||||
def _collect_unique_ids(items: list[Any], key: str, path: str) -> set[str]:
|
||||
ids: set[str] = set()
|
||||
for index, item in enumerate(items):
|
||||
if not isinstance(item, dict):
|
||||
raise MeetingContextValidationError(f"{path}[{index}] must be an object.")
|
||||
identifier = _text(item.get(key))
|
||||
if not identifier:
|
||||
raise MeetingContextValidationError(f"{path}[{index}].{key} must be non-empty.")
|
||||
if identifier in ids:
|
||||
raise MeetingContextValidationError(f"Duplicate {key}: {identifier!r}.")
|
||||
ids.add(identifier)
|
||||
return ids
|
||||
|
||||
|
||||
def _validate_attendance(item: dict[str, Any], path: str) -> None:
|
||||
status = item.get("attendance_status")
|
||||
if status not in VALID_ATTENDANCE_STATUSES:
|
||||
raise MeetingContextValidationError(
|
||||
f"{path}.attendance_status has invalid value: {status!r}."
|
||||
)
|
||||
|
||||
|
||||
def _validate_department_reference(
|
||||
item: dict[str, Any],
|
||||
path: str,
|
||||
department_ids: set[str],
|
||||
) -> None:
|
||||
department_id = _text(item.get("department_id") or item.get("department"))
|
||||
if department_id and department_id not in department_ids:
|
||||
raise MeetingContextValidationError(
|
||||
f"{path}.department_id references unknown department: {department_id!r}."
|
||||
)
|
||||
|
||||
|
||||
def _require_mapping(data: dict[str, Any], key: str) -> dict[str, Any]:
|
||||
value = data.get(key)
|
||||
if not isinstance(value, dict):
|
||||
raise MeetingContextValidationError(f"{key} must be an object.")
|
||||
return value
|
||||
|
||||
|
||||
def _optional_mapping(value: Any, path: str) -> dict[str, Any]:
|
||||
if value is None:
|
||||
return {}
|
||||
if not isinstance(value, dict):
|
||||
raise MeetingContextValidationError(f"{path} must be an object.")
|
||||
return value
|
||||
|
||||
|
||||
def _optional_list(value: Any, path: str) -> list[Any]:
|
||||
if value is None:
|
||||
return []
|
||||
if not isinstance(value, list):
|
||||
raise MeetingContextValidationError(f"{path} must be a list.")
|
||||
return value
|
||||
|
||||
|
||||
def _require_non_empty_string(data: dict[str, Any], key_path: str) -> None:
|
||||
key = key_path.split(".")[-1]
|
||||
if not _text(data.get(key)):
|
||||
raise MeetingContextValidationError(f"{key_path} must be present and non-empty.")
|
||||
|
||||
|
||||
def _text(value: Any) -> str:
|
||||
if value is None:
|
||||
return ""
|
||||
return str(value).strip()
|
||||
|
||||
|
||||
def _parse_simple_yaml(text: str) -> Any:
|
||||
lines = []
|
||||
for raw_line in text.splitlines():
|
||||
stripped = _strip_yaml_comment(raw_line.rstrip())
|
||||
if stripped.strip():
|
||||
lines.append((len(stripped) - len(stripped.lstrip(" ")), stripped.lstrip(" ")))
|
||||
if not lines:
|
||||
return None
|
||||
parsed, index = _parse_yaml_block(lines, 0, lines[0][0])
|
||||
if index != len(lines):
|
||||
raise MeetingContextValidationError("Could not parse Meeting Context YAML.")
|
||||
return parsed
|
||||
|
||||
|
||||
def _parse_yaml_block(
|
||||
lines: list[tuple[int, str]],
|
||||
index: int,
|
||||
indent: int,
|
||||
) -> tuple[Any, int]:
|
||||
if lines[index][1].startswith("- "):
|
||||
result = []
|
||||
while index < len(lines) and lines[index][0] == indent and lines[index][1].startswith("- "):
|
||||
content = lines[index][1][2:].strip()
|
||||
index += 1
|
||||
if not content:
|
||||
value, index = _parse_yaml_block(lines, index, lines[index][0])
|
||||
result.append(value)
|
||||
continue
|
||||
|
||||
if ":" in content:
|
||||
key, raw_value = content.split(":", 1)
|
||||
item = {key.strip(): _parse_scalar(raw_value.strip()) if raw_value.strip() else {}}
|
||||
while index < len(lines) and lines[index][0] > indent:
|
||||
child_indent, child_content = lines[index]
|
||||
if child_content.startswith("- "):
|
||||
break
|
||||
child_key, child_raw_value = child_content.split(":", 1)
|
||||
index += 1
|
||||
child_raw_value = child_raw_value.strip()
|
||||
if child_raw_value:
|
||||
item[child_key.strip()] = _parse_scalar(child_raw_value)
|
||||
elif index < len(lines) and lines[index][0] > child_indent:
|
||||
item[child_key.strip()], index = _parse_yaml_block(lines, index, lines[index][0])
|
||||
else:
|
||||
item[child_key.strip()] = None
|
||||
result.append(item)
|
||||
else:
|
||||
result.append(_parse_scalar(content))
|
||||
return result, index
|
||||
|
||||
result = {}
|
||||
while index < len(lines) and lines[index][0] == indent and not lines[index][1].startswith("- "):
|
||||
key, raw_value = lines[index][1].split(":", 1)
|
||||
index += 1
|
||||
raw_value = raw_value.strip()
|
||||
if raw_value:
|
||||
result[key.strip()] = _parse_scalar(raw_value)
|
||||
elif index < len(lines) and lines[index][0] > indent:
|
||||
result[key.strip()], index = _parse_yaml_block(lines, index, lines[index][0])
|
||||
else:
|
||||
result[key.strip()] = None
|
||||
return result, index
|
||||
|
||||
|
||||
def _parse_scalar(value: str) -> Any:
|
||||
if value in {"null", "Null", "NULL", "~"}:
|
||||
return None
|
||||
if value in {"true", "True", "TRUE"}:
|
||||
return True
|
||||
if value in {"false", "False", "FALSE"}:
|
||||
return False
|
||||
if value in {"[]", "{}"} or (
|
||||
value.startswith("[") and value.endswith("]")
|
||||
):
|
||||
return ast.literal_eval(value)
|
||||
if (
|
||||
(value.startswith('"') and value.endswith('"'))
|
||||
or (value.startswith("'") and value.endswith("'"))
|
||||
):
|
||||
return ast.literal_eval(value)
|
||||
return value
|
||||
|
||||
|
||||
def _strip_yaml_comment(line: str) -> str:
|
||||
in_single = False
|
||||
in_double = False
|
||||
for index, char in enumerate(line):
|
||||
if char == "'" and not in_double:
|
||||
in_single = not in_single
|
||||
elif char == '"' and not in_single:
|
||||
in_double = not in_double
|
||||
elif char == "#" and not in_single and not in_double:
|
||||
return line[:index].rstrip()
|
||||
return line
|
||||
Reference in New Issue
Block a user