Implement Meeting Context V1 and extraction improvements
Introduce Meeting Context V1 with YAML schema, validation and template. Support optional --meeting-context during chunk extraction. Inject authoritative Meeting Context into extraction prompts. Record Meeting Context provenance in extraction output. Activate todos.md in shared prompt assembly. Strengthen responsibility attribution and decision/todo boundaries. Add focused Gold scenarios and validation tests. Update architecture and pipeline documentation.
This commit is contained in:
@@ -20,6 +20,11 @@ from typing import Any
|
||||
import requests
|
||||
|
||||
from src.meeting_lab.llm.prompts import build_extraction_prompt
|
||||
from src.meeting_lab.models.meeting_context import (
|
||||
MeetingContext,
|
||||
load_meeting_context,
|
||||
render_meeting_context_for_prompt,
|
||||
)
|
||||
|
||||
|
||||
DEFAULT_MODEL = "qwen3:8b"
|
||||
@@ -36,6 +41,8 @@ EXTRACTION_CATEGORIES = (
|
||||
|
||||
NORMALIZED_CHUNK_RE = re.compile(r"^(chunk_\d+)_normalized\.txt$")
|
||||
|
||||
EXTRACTION_TASK_PROMPT_NAMES = ("decisions.md", "todos.md")
|
||||
|
||||
|
||||
OUTPUT_SCHEMA = {
|
||||
"chunk": {
|
||||
@@ -138,11 +145,31 @@ def parse_args() -> argparse.Namespace:
|
||||
default=32768,
|
||||
help="Context window tokens per chunk (default: 32768)",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--meeting-context",
|
||||
type=Path,
|
||||
help="Optional Meeting Context V1 YAML file to inject into extraction prompts.",
|
||||
)
|
||||
return parser.parse_args()
|
||||
|
||||
|
||||
def build_prompt(source_name: str, transcript: str) -> str:
|
||||
return build_extraction_prompt(source_name, transcript, OUTPUT_SCHEMA)
|
||||
def build_prompt(
|
||||
source_name: str,
|
||||
transcript: str,
|
||||
meeting_context: MeetingContext | None = None,
|
||||
) -> str:
|
||||
context_text = (
|
||||
render_meeting_context_for_prompt(meeting_context)
|
||||
if meeting_context is not None
|
||||
else None
|
||||
)
|
||||
return build_extraction_prompt(
|
||||
source_name,
|
||||
transcript,
|
||||
OUTPUT_SCHEMA,
|
||||
meeting_context=context_text,
|
||||
task_prompt_names=EXTRACTION_TASK_PROMPT_NAMES,
|
||||
)
|
||||
|
||||
|
||||
def call_ollama(
|
||||
@@ -376,12 +403,13 @@ def extract_chunk(
|
||||
temperature: float,
|
||||
num_predict: int | None,
|
||||
num_ctx: int | None,
|
||||
) -> dict[str, list[str]]:
|
||||
meeting_context: MeetingContext | None = None,
|
||||
) -> dict[str, Any]:
|
||||
transcript = chunk_path.read_text(encoding="utf-8-sig").strip()
|
||||
if not transcript:
|
||||
raise ValueError(f"The input file is empty: {chunk_path}")
|
||||
|
||||
prompt = build_prompt(chunk_path.name, transcript)
|
||||
prompt = build_prompt(chunk_path.name, transcript, meeting_context=meeting_context)
|
||||
raw_text, _metadata = call_ollama(
|
||||
endpoint=endpoint,
|
||||
model=model,
|
||||
@@ -402,6 +430,8 @@ def extract_chunk(
|
||||
) from exc
|
||||
|
||||
extraction = normalize_current_schema(parsed)
|
||||
if meeting_context is not None:
|
||||
extraction["context"] = meeting_context.provenance()
|
||||
output_path.parent.mkdir(parents=True, exist_ok=True)
|
||||
output_path.write_text(
|
||||
json.dumps(extraction, ensure_ascii=False, indent=2) + "\n",
|
||||
@@ -419,6 +449,7 @@ def extract_input(
|
||||
temperature: float,
|
||||
num_predict: int | None,
|
||||
num_ctx: int | None,
|
||||
meeting_context: MeetingContext | None = None,
|
||||
) -> list[Path]:
|
||||
if input_path.is_file():
|
||||
output_path = output or extraction_path_for_chunk(input_path)
|
||||
@@ -431,6 +462,7 @@ def extract_input(
|
||||
temperature,
|
||||
num_predict,
|
||||
num_ctx,
|
||||
meeting_context=meeting_context,
|
||||
)
|
||||
return [output_path]
|
||||
|
||||
@@ -459,6 +491,7 @@ def extract_input(
|
||||
temperature,
|
||||
num_predict,
|
||||
num_ctx,
|
||||
meeting_context=meeting_context,
|
||||
)
|
||||
output_paths.append(output_path)
|
||||
|
||||
@@ -469,6 +502,11 @@ def main() -> int:
|
||||
args = parse_args()
|
||||
|
||||
try:
|
||||
meeting_context = (
|
||||
load_meeting_context(args.meeting_context)
|
||||
if args.meeting_context is not None
|
||||
else None
|
||||
)
|
||||
output_paths = extract_input(
|
||||
input_path=args.input,
|
||||
output=args.output,
|
||||
@@ -478,6 +516,7 @@ def main() -> int:
|
||||
temperature=args.temperature,
|
||||
num_predict=args.num_predict,
|
||||
num_ctx=args.num_ctx,
|
||||
meeting_context=meeting_context,
|
||||
)
|
||||
except requests.ConnectionError:
|
||||
print(
|
||||
@@ -496,6 +535,8 @@ def main() -> int:
|
||||
return 1
|
||||
|
||||
print(f"Input: {args.input}")
|
||||
if args.meeting_context is not None:
|
||||
print(f"Meeting context: {args.meeting_context}")
|
||||
print(f"Processed chunks: {len(output_paths)}")
|
||||
print(f"Extraction JSON files: {len(output_paths)}")
|
||||
for output_path in output_paths:
|
||||
|
||||
@@ -30,12 +30,14 @@ def build_extraction_prompt(
|
||||
source_name: str,
|
||||
transcript: str,
|
||||
output_schema: dict[str, Any],
|
||||
meeting_context: str | None = None,
|
||||
task_prompt_names: Iterable[str] = ("decisions.md",),
|
||||
prompts_dir: Path = PROMPTS_DIR,
|
||||
) -> str:
|
||||
schema_text = json.dumps(output_schema, ensure_ascii=False, indent=2)
|
||||
prompt_parts = [
|
||||
load_prompt("common.md", prompts_dir),
|
||||
meeting_context,
|
||||
*load_existing_prompts(task_prompt_names, prompts_dir),
|
||||
f"""Quelldatei:
|
||||
{source_name}
|
||||
|
||||
@@ -0,0 +1,396 @@
|
||||
"""Meeting Context V1 loading, validation and prompt rendering."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import ast
|
||||
from dataclasses import dataclass
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
|
||||
SUPPORTED_SCHEMA_VERSIONS = {"1"}
|
||||
VALID_ATTENDANCE_STATUSES = {"present", "not_present", "absent"}
|
||||
|
||||
|
||||
class MeetingContextValidationError(ValueError):
|
||||
"""Raised when a Meeting Context file is structurally invalid."""
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class MeetingContext:
|
||||
data: dict[str, Any]
|
||||
source_file: Path
|
||||
|
||||
@property
|
||||
def schema_version(self) -> str:
|
||||
return str(self.data["schema_version"])
|
||||
|
||||
@property
|
||||
def meeting_id(self) -> str:
|
||||
return str(self.data["meeting"]["meeting_id"])
|
||||
|
||||
def provenance(self) -> dict[str, str]:
|
||||
return {
|
||||
"meeting_id": self.meeting_id,
|
||||
"source_file": str(self.source_file),
|
||||
"schema_version": self.schema_version,
|
||||
}
|
||||
|
||||
|
||||
def load_meeting_context(path: Path) -> MeetingContext:
|
||||
loaded = _load_yaml(path)
|
||||
if not isinstance(loaded, dict):
|
||||
raise MeetingContextValidationError("Meeting Context must be a YAML object.")
|
||||
|
||||
validate_meeting_context(loaded)
|
||||
return MeetingContext(data=loaded, source_file=path)
|
||||
|
||||
|
||||
def validate_meeting_context(data: dict[str, Any]) -> None:
|
||||
schema_version = str(data.get("schema_version", "")).strip()
|
||||
if schema_version not in SUPPORTED_SCHEMA_VERSIONS:
|
||||
raise MeetingContextValidationError(
|
||||
f"Unsupported meeting context schema_version: {schema_version!r}."
|
||||
)
|
||||
|
||||
meeting = _require_mapping(data, "meeting")
|
||||
_require_non_empty_string(meeting, "meeting.meeting_id")
|
||||
_require_non_empty_string(meeting, "meeting.title")
|
||||
_require_non_empty_string(meeting, "meeting.language")
|
||||
|
||||
organization = _optional_mapping(data.get("organization"), "organization")
|
||||
departments = _optional_list(organization.get("departments"), "organization.departments")
|
||||
department_ids = _collect_department_ids(departments)
|
||||
|
||||
participants = _optional_list(data.get("participants"), "participants")
|
||||
mentioned_people = _optional_list(data.get("mentioned_people"), "mentioned_people")
|
||||
|
||||
participant_ids = _collect_unique_ids(participants, "participant_id", "participants")
|
||||
person_ids = _collect_unique_ids(mentioned_people, "person_id", "mentioned_people")
|
||||
collisions = sorted(participant_ids & person_ids)
|
||||
if collisions:
|
||||
raise MeetingContextValidationError(
|
||||
"Participant IDs and mentioned-person IDs must not collide: "
|
||||
+ ", ".join(collisions)
|
||||
)
|
||||
|
||||
for index, participant in enumerate(participants):
|
||||
item_path = f"participants[{index}]"
|
||||
_validate_attendance(participant, item_path)
|
||||
if participant.get("attendance_status") != "present":
|
||||
raise MeetingContextValidationError(
|
||||
f"{item_path}.attendance_status must be 'present'."
|
||||
)
|
||||
_validate_department_reference(participant, item_path, department_ids)
|
||||
|
||||
for index, person in enumerate(mentioned_people):
|
||||
item_path = f"mentioned_people[{index}]"
|
||||
_validate_attendance(person, item_path)
|
||||
if person.get("attendance_status") == "present":
|
||||
raise MeetingContextValidationError(
|
||||
f"{item_path}.attendance_status must not be 'present'."
|
||||
)
|
||||
_validate_department_reference(person, item_path, department_ids)
|
||||
|
||||
|
||||
def render_meeting_context_for_prompt(context: MeetingContext) -> str:
|
||||
data = context.data
|
||||
meeting = data["meeting"]
|
||||
organization = _optional_mapping(data.get("organization"), "organization")
|
||||
departments_by_id = {
|
||||
str(department.get("id")): str(department.get("name"))
|
||||
for department in _optional_list(
|
||||
organization.get("departments"), "organization.departments"
|
||||
)
|
||||
if isinstance(department, dict) and department.get("id") and department.get("name")
|
||||
}
|
||||
|
||||
lines = [
|
||||
"MEETING CONTEXT V1 (AUTHORITATIVE METADATA)",
|
||||
"",
|
||||
"Rules:",
|
||||
"- The participant list is authoritative.",
|
||||
"- Mentioned people did not attend this meeting.",
|
||||
"- Roles and departments must not be inferred or changed.",
|
||||
"- Discussion of a department does not establish responsibility.",
|
||||
"- An action item may name a responsible person only when assignment or acceptance is explicit in the transcript.",
|
||||
"- Objections, suggestions and expertise do not establish ownership.",
|
||||
"",
|
||||
"Meeting:",
|
||||
f"- Title: {_text(meeting.get('title'))}",
|
||||
f"- Language: {_text(meeting.get('language'))}",
|
||||
]
|
||||
|
||||
objective = _text(meeting.get("objective"))
|
||||
if objective:
|
||||
lines.append(f"- Objective: {objective}")
|
||||
|
||||
participants = _optional_list(data.get("participants"), "participants")
|
||||
if participants:
|
||||
lines.extend(["", "Actual participants:"])
|
||||
for participant in participants:
|
||||
lines.append(_render_person_line(participant, "participant_id", departments_by_id))
|
||||
|
||||
mentioned_people = _optional_list(data.get("mentioned_people"), "mentioned_people")
|
||||
if mentioned_people:
|
||||
lines.extend(["", "Mentioned but absent people:"])
|
||||
for person in mentioned_people:
|
||||
lines.append(_render_person_line(person, "person_id", departments_by_id))
|
||||
|
||||
abbreviations = _optional_mapping(organization.get("abbreviations"), "organization.abbreviations")
|
||||
abbreviation_lines = [
|
||||
f"- {key}: {_text(value)}"
|
||||
for key, value in sorted(abbreviations.items())
|
||||
if _text(value)
|
||||
]
|
||||
if abbreviation_lines:
|
||||
lines.extend(["", "Abbreviations:", *abbreviation_lines])
|
||||
|
||||
known_entities = _optional_mapping(data.get("known_entities"), "known_entities")
|
||||
entity_lines = []
|
||||
for key in sorted(known_entities):
|
||||
values = [_text(value) for value in _optional_list(known_entities.get(key), key)]
|
||||
values = [value for value in values if value]
|
||||
if values:
|
||||
entity_lines.append(f"- {key}: {', '.join(values)}")
|
||||
if entity_lines:
|
||||
lines.extend(["", "Relevant known entities:", *entity_lines])
|
||||
|
||||
context_rules = _optional_mapping(data.get("context_rules"), "context_rules")
|
||||
rule_lines = [
|
||||
f"- {key}: {str(value).lower() if isinstance(value, bool) else _text(value)}"
|
||||
for key, value in sorted(context_rules.items())
|
||||
]
|
||||
if rule_lines:
|
||||
lines.extend(["", "Context rules:", *rule_lines])
|
||||
|
||||
return "\n".join(lines).strip() + "\n"
|
||||
|
||||
|
||||
def _render_person_line(
|
||||
person: dict[str, Any],
|
||||
id_key: str,
|
||||
departments_by_id: dict[str, str],
|
||||
) -> str:
|
||||
parts = [_text(person.get("display_name"))]
|
||||
aliases = [_text(alias) for alias in _optional_list(person.get("aliases"), "aliases")]
|
||||
aliases = [alias for alias in aliases if alias]
|
||||
if aliases:
|
||||
parts.append(f"aliases: {', '.join(aliases)}")
|
||||
|
||||
roles = _roles(person)
|
||||
if roles:
|
||||
parts.append(f"roles: {', '.join(roles)}")
|
||||
|
||||
department_id = _text(person.get("department_id") or person.get("department"))
|
||||
if department_id:
|
||||
department_name = departments_by_id.get(department_id, department_id)
|
||||
parts.append(f"department: {department_name}")
|
||||
|
||||
identifier = _text(person.get(id_key))
|
||||
if identifier:
|
||||
parts.append(f"id: {identifier}")
|
||||
|
||||
return "- " + "; ".join(part for part in parts if part)
|
||||
|
||||
|
||||
def _roles(person: dict[str, Any]) -> list[str]:
|
||||
roles = [_text(role) for role in _optional_list(person.get("roles"), "roles")]
|
||||
role = _text(person.get("role"))
|
||||
if role:
|
||||
roles.insert(0, role)
|
||||
return [role for role in roles if role]
|
||||
|
||||
|
||||
def _load_yaml(path: Path) -> Any:
|
||||
text = path.read_text(encoding="utf-8-sig")
|
||||
try:
|
||||
import yaml # type: ignore[import-not-found]
|
||||
except ModuleNotFoundError:
|
||||
return _parse_simple_yaml(text)
|
||||
return yaml.safe_load(text)
|
||||
|
||||
|
||||
def _collect_department_ids(departments: list[Any]) -> set[str]:
|
||||
ids: set[str] = set()
|
||||
for index, department in enumerate(departments):
|
||||
if not isinstance(department, dict):
|
||||
raise MeetingContextValidationError(
|
||||
f"organization.departments[{index}] must be an object."
|
||||
)
|
||||
department_id = _text(department.get("id"))
|
||||
if not department_id:
|
||||
raise MeetingContextValidationError(
|
||||
f"organization.departments[{index}].id must be non-empty."
|
||||
)
|
||||
if department_id in ids:
|
||||
raise MeetingContextValidationError(
|
||||
f"Duplicate department id: {department_id!r}."
|
||||
)
|
||||
ids.add(department_id)
|
||||
return ids
|
||||
|
||||
|
||||
def _collect_unique_ids(items: list[Any], key: str, path: str) -> set[str]:
|
||||
ids: set[str] = set()
|
||||
for index, item in enumerate(items):
|
||||
if not isinstance(item, dict):
|
||||
raise MeetingContextValidationError(f"{path}[{index}] must be an object.")
|
||||
identifier = _text(item.get(key))
|
||||
if not identifier:
|
||||
raise MeetingContextValidationError(f"{path}[{index}].{key} must be non-empty.")
|
||||
if identifier in ids:
|
||||
raise MeetingContextValidationError(f"Duplicate {key}: {identifier!r}.")
|
||||
ids.add(identifier)
|
||||
return ids
|
||||
|
||||
|
||||
def _validate_attendance(item: dict[str, Any], path: str) -> None:
|
||||
status = item.get("attendance_status")
|
||||
if status not in VALID_ATTENDANCE_STATUSES:
|
||||
raise MeetingContextValidationError(
|
||||
f"{path}.attendance_status has invalid value: {status!r}."
|
||||
)
|
||||
|
||||
|
||||
def _validate_department_reference(
|
||||
item: dict[str, Any],
|
||||
path: str,
|
||||
department_ids: set[str],
|
||||
) -> None:
|
||||
department_id = _text(item.get("department_id") or item.get("department"))
|
||||
if department_id and department_id not in department_ids:
|
||||
raise MeetingContextValidationError(
|
||||
f"{path}.department_id references unknown department: {department_id!r}."
|
||||
)
|
||||
|
||||
|
||||
def _require_mapping(data: dict[str, Any], key: str) -> dict[str, Any]:
|
||||
value = data.get(key)
|
||||
if not isinstance(value, dict):
|
||||
raise MeetingContextValidationError(f"{key} must be an object.")
|
||||
return value
|
||||
|
||||
|
||||
def _optional_mapping(value: Any, path: str) -> dict[str, Any]:
|
||||
if value is None:
|
||||
return {}
|
||||
if not isinstance(value, dict):
|
||||
raise MeetingContextValidationError(f"{path} must be an object.")
|
||||
return value
|
||||
|
||||
|
||||
def _optional_list(value: Any, path: str) -> list[Any]:
|
||||
if value is None:
|
||||
return []
|
||||
if not isinstance(value, list):
|
||||
raise MeetingContextValidationError(f"{path} must be a list.")
|
||||
return value
|
||||
|
||||
|
||||
def _require_non_empty_string(data: dict[str, Any], key_path: str) -> None:
|
||||
key = key_path.split(".")[-1]
|
||||
if not _text(data.get(key)):
|
||||
raise MeetingContextValidationError(f"{key_path} must be present and non-empty.")
|
||||
|
||||
|
||||
def _text(value: Any) -> str:
|
||||
if value is None:
|
||||
return ""
|
||||
return str(value).strip()
|
||||
|
||||
|
||||
def _parse_simple_yaml(text: str) -> Any:
|
||||
lines = []
|
||||
for raw_line in text.splitlines():
|
||||
stripped = _strip_yaml_comment(raw_line.rstrip())
|
||||
if stripped.strip():
|
||||
lines.append((len(stripped) - len(stripped.lstrip(" ")), stripped.lstrip(" ")))
|
||||
if not lines:
|
||||
return None
|
||||
parsed, index = _parse_yaml_block(lines, 0, lines[0][0])
|
||||
if index != len(lines):
|
||||
raise MeetingContextValidationError("Could not parse Meeting Context YAML.")
|
||||
return parsed
|
||||
|
||||
|
||||
def _parse_yaml_block(
|
||||
lines: list[tuple[int, str]],
|
||||
index: int,
|
||||
indent: int,
|
||||
) -> tuple[Any, int]:
|
||||
if lines[index][1].startswith("- "):
|
||||
result = []
|
||||
while index < len(lines) and lines[index][0] == indent and lines[index][1].startswith("- "):
|
||||
content = lines[index][1][2:].strip()
|
||||
index += 1
|
||||
if not content:
|
||||
value, index = _parse_yaml_block(lines, index, lines[index][0])
|
||||
result.append(value)
|
||||
continue
|
||||
|
||||
if ":" in content:
|
||||
key, raw_value = content.split(":", 1)
|
||||
item = {key.strip(): _parse_scalar(raw_value.strip()) if raw_value.strip() else {}}
|
||||
while index < len(lines) and lines[index][0] > indent:
|
||||
child_indent, child_content = lines[index]
|
||||
if child_content.startswith("- "):
|
||||
break
|
||||
child_key, child_raw_value = child_content.split(":", 1)
|
||||
index += 1
|
||||
child_raw_value = child_raw_value.strip()
|
||||
if child_raw_value:
|
||||
item[child_key.strip()] = _parse_scalar(child_raw_value)
|
||||
elif index < len(lines) and lines[index][0] > child_indent:
|
||||
item[child_key.strip()], index = _parse_yaml_block(lines, index, lines[index][0])
|
||||
else:
|
||||
item[child_key.strip()] = None
|
||||
result.append(item)
|
||||
else:
|
||||
result.append(_parse_scalar(content))
|
||||
return result, index
|
||||
|
||||
result = {}
|
||||
while index < len(lines) and lines[index][0] == indent and not lines[index][1].startswith("- "):
|
||||
key, raw_value = lines[index][1].split(":", 1)
|
||||
index += 1
|
||||
raw_value = raw_value.strip()
|
||||
if raw_value:
|
||||
result[key.strip()] = _parse_scalar(raw_value)
|
||||
elif index < len(lines) and lines[index][0] > indent:
|
||||
result[key.strip()], index = _parse_yaml_block(lines, index, lines[index][0])
|
||||
else:
|
||||
result[key.strip()] = None
|
||||
return result, index
|
||||
|
||||
|
||||
def _parse_scalar(value: str) -> Any:
|
||||
if value in {"null", "Null", "NULL", "~"}:
|
||||
return None
|
||||
if value in {"true", "True", "TRUE"}:
|
||||
return True
|
||||
if value in {"false", "False", "FALSE"}:
|
||||
return False
|
||||
if value in {"[]", "{}"} or (
|
||||
value.startswith("[") and value.endswith("]")
|
||||
):
|
||||
return ast.literal_eval(value)
|
||||
if (
|
||||
(value.startswith('"') and value.endswith('"'))
|
||||
or (value.startswith("'") and value.endswith("'"))
|
||||
):
|
||||
return ast.literal_eval(value)
|
||||
return value
|
||||
|
||||
|
||||
def _strip_yaml_comment(line: str) -> str:
|
||||
in_single = False
|
||||
in_double = False
|
||||
for index, char in enumerate(line):
|
||||
if char == "'" and not in_double:
|
||||
in_single = not in_single
|
||||
elif char == '"' and not in_single:
|
||||
in_double = not in_double
|
||||
elif char == "#" and not in_single and not in_double:
|
||||
return line[:index].rstrip()
|
||||
return line
|
||||
Reference in New Issue
Block a user