Files
meeting-lab/src/meeting_lab/protocol/render_working_protocol.py
T
admin 950284e236 Stabilize Meeting Lab pipeline for RC1 evaluation
This commit significantly improves the robustness and determinism of the Meeting Lab processing pipeline and establishes the first Release Candidate baseline for end-to-end evaluation.

Highlights

- BUG-009
  - Implement deterministic responsible-party validation
  - Normalize participant aliases using Meeting Context
  - Reject invalid responsible values (dates, locations, technical terms, projects, products, unknown entities)
  - Record structured responsibility validation metadata
  - Add focused regression tests

- BUG-010
  - Implement adaptive num_predict estimation for Semantic Consolidator
  - Eliminate JSON truncation caused by fixed output limits
  - Add deterministic source coverage repair
  - Preserve strict post-repair validation
  - Add regression tests

- BUG-011
  - Implement Working Protocol V2 renderer contract enforcement
  - Preserve raw renderer responses
  - Reject invalid protocol output instead of accepting malformed documents
  - Add deterministic cleanup for harmless formatting deviations
  - Add focused renderer regression tests

- Meeting Context
  - Validate Meeting Context V1
  - Integrate authoritative participant alias normalization

- Documentation
  - Update architecture documentation
  - Update output documentation
  - Update regression bug tracker

The pipeline now fails safely instead of silently accepting invalid intermediate or final artifacts.

Remaining work focuses primarily on extraction quality and semantic classification (decisions, action items, protocol faithfulness), rather than pipeline robustness.
2026-08-04 13:11:54 +02:00

520 lines
17 KiB
Python

#!/usr/bin/env python3
"""LLM-backed Working Protocol V2 renderer with deterministic contract checks."""
from __future__ import annotations
import argparse
import json
import re
import sys
import time
from dataclasses import dataclass
from datetime import datetime
from pathlib import Path
from typing import Any
import requests
try:
from meeting_lab.llm.prompts import load_prompt
except ModuleNotFoundError: # pragma: no cover - used by repository-root tests.
from src.meeting_lab.llm.prompts import load_prompt
DEFAULT_MODEL = "qwen3.5:9b"
DEFAULT_ENDPOINT = "http://127.0.0.1:11434/api/generate"
DEFAULT_NUM_CTX = 32768
DEFAULT_NUM_PREDICT = 4096
DEFAULT_TIMEOUT = 1800
PROMPT_NAME = "working_protocol.md"
REQUIRED_TITLE = "# Working Protocol"
ALLOWED_TOPIC_SECTIONS = {
"Background",
"Decisions",
"Action Items",
"Open Questions",
}
GENERIC_SUMMARY_HEADINGS = {
"Entscheidungen",
"Decisions",
"Handlungsaufträge",
"Handlungsauftraege",
"Action Items",
"Offene Fragen",
"Open Questions",
"Technische Details",
"Technical Details",
"Fakten",
"Facts",
"Zusammenfassung",
"Summary",
"Konsolidierter Projektstatusbericht",
}
HEADING_RE = re.compile(r"^(#{1,6})\s+(.+?)\s*$")
class WorkingProtocolValidationError(ValueError):
"""Raised when rendered Markdown violates the Working Protocol V2 contract."""
@dataclass(frozen=True)
class WorkingProtocolValidationReport:
valid: bool
violations: list[dict[str, Any]]
def to_dict(self) -> dict[str, Any]:
return {"valid": self.valid, "violations": self.violations}
def parse_args() -> argparse.Namespace:
parser = argparse.ArgumentParser(
description="Render a consolidated meeting representation as Working Protocol V2."
)
parser.add_argument("input", type=Path, help="Consolidated meeting JSON input.")
parser.add_argument(
"-o",
"--output-dir",
type=Path,
required=True,
help="Directory for working_protocol.md, raw response, validation and metadata.",
)
parser.add_argument(
"--model",
default=DEFAULT_MODEL,
help=f"Ollama model name (default: {DEFAULT_MODEL}).",
)
parser.add_argument(
"--endpoint",
default=DEFAULT_ENDPOINT,
help=f"Ollama generate endpoint (default: {DEFAULT_ENDPOINT}).",
)
parser.add_argument(
"--timeout",
type=int,
default=DEFAULT_TIMEOUT,
help=f"HTTP timeout in seconds (default: {DEFAULT_TIMEOUT}).",
)
parser.add_argument(
"--num-ctx",
type=int,
default=DEFAULT_NUM_CTX,
help=f"Context window tokens (default: {DEFAULT_NUM_CTX}).",
)
parser.add_argument(
"--num-predict",
type=int,
default=DEFAULT_NUM_PREDICT,
help=f"Maximum generated tokens (default: {DEFAULT_NUM_PREDICT}).",
)
thinking = parser.add_mutually_exclusive_group()
thinking.add_argument("--think", dest="think", action="store_true")
thinking.add_argument("--no-think", dest="think", action="store_false")
parser.set_defaults(think=False)
return parser.parse_args()
def build_renderer_prompt(input_text: str) -> str:
return f"{load_prompt(PROMPT_NAME)}\n\nINPUT JSON:\n{input_text}\n"
def response_text_from_ollama_data(data: dict[str, Any]) -> str:
text = data.get("response")
if isinstance(text, str):
return text
message = data.get("message")
if isinstance(message, dict) and isinstance(message.get("content"), str):
return message["content"]
return ""
def build_ollama_payload(
model: str,
prompt: str,
num_ctx: int,
num_predict: int,
think: bool,
) -> dict[str, Any]:
return {
"model": model,
"prompt": prompt,
"think": think,
"stream": False,
"options": {
"temperature": 0.0,
"num_ctx": num_ctx,
"num_predict": num_predict,
},
}
def call_ollama(
endpoint: str,
payload: dict[str, Any],
timeout: int,
) -> tuple[dict[str, Any], float]:
start = time.perf_counter()
response = requests.post(endpoint, json=payload, timeout=timeout)
runtime = time.perf_counter() - start
response.raise_for_status()
data = response.json()
if not isinstance(data, dict):
raise ValueError("Ollama response must be a JSON object.")
return data, runtime
def normalize_heading_whitespace(line: str) -> str:
match = HEADING_RE.match(line.strip())
if not match:
return line.rstrip()
return f"{match.group(1)} {match.group(2).strip()}"
def clean_working_protocol_markdown(text: str) -> str:
"""
Remove only non-semantic wrapper text before an existing protocol heading.
The cleanup does not create headings or transform a categorized summary into
a Working Protocol. It only makes an already present contract heading the
first byte of the candidate document.
"""
text = text.replace("\r\n", "\n").replace("\r", "\n")
lines = text.split("\n")
start_index: int | None = None
for index, line in enumerate(lines):
if normalize_heading_whitespace(line) == REQUIRED_TITLE:
start_index = index
break
if start_index is None:
return text.lstrip()
cleaned_lines = lines[start_index:]
if cleaned_lines:
cleaned_lines[0] = REQUIRED_TITLE
return "\n".join(normalize_heading_whitespace(line) for line in cleaned_lines).strip() + "\n"
def validate_markdown_shape(markdown: str) -> list[dict[str, Any]]:
violations: list[dict[str, Any]] = []
if markdown.count("```") % 2 != 0:
violations.append(
{
"type": "malformed_markdown",
"reason": "unclosed_fenced_code_block",
}
)
seen_title = False
seen_topic = False
current_topic_has_section = False
topic_count = 0
section_count = 0
for line_number, line in enumerate(markdown.splitlines(), start=1):
match = HEADING_RE.match(line)
if not match:
continue
level = len(match.group(1))
title = match.group(2).strip().strip("*")
if level == 1:
if line_number != 1 or title != "Working Protocol":
violations.append(
{
"type": "invalid_heading",
"line": line_number,
"heading": line,
"reason": "only the first line may be '# Working Protocol'",
}
)
seen_title = True
continue
if not seen_title:
violations.append(
{
"type": "invalid_heading_order",
"line": line_number,
"heading": line,
"reason": "heading appears before required title",
}
)
continue
if level == 2:
topic_count += 1
seen_topic = True
current_topic_has_section = False
if title in GENERIC_SUMMARY_HEADINGS:
violations.append(
{
"type": "generic_summary_framing",
"line": line_number,
"heading": line,
"reason": "top-level category heading is not a topic",
}
)
continue
if level == 3:
if not seen_topic:
violations.append(
{
"type": "missing_topic",
"line": line_number,
"heading": line,
"reason": "section appears before any topic heading",
}
)
if title not in ALLOWED_TOPIC_SECTIONS:
violations.append(
{
"type": "unknown_topic_section",
"line": line_number,
"heading": line,
"allowed": sorted(ALLOWED_TOPIC_SECTIONS),
}
)
else:
section_count += 1
current_topic_has_section = True
continue
violations.append(
{
"type": "unsupported_heading_level",
"line": line_number,
"heading": line,
"reason": "Working Protocol V2 uses h1 title, h2 topics and h3 sections only",
}
)
if seen_topic and not current_topic_has_section:
# This catches the last topic; earlier empty topics are caught below by
# counting consecutive h2 headings.
pass
h2_without_section = _topic_headings_without_sections(markdown)
violations.extend(h2_without_section)
if topic_count == 0:
violations.append(
{
"type": "missing_topic",
"reason": "document must contain at least one '## Topic title' section",
}
)
if section_count == 0:
violations.append(
{
"type": "missing_topic_sections",
"reason": "document must contain at least one allowed h3 topic section",
}
)
return violations
def _topic_headings_without_sections(markdown: str) -> list[dict[str, Any]]:
violations: list[dict[str, Any]] = []
current_topic: tuple[int, str] | None = None
current_has_section = False
for line_number, line in enumerate(markdown.splitlines(), start=1):
match = HEADING_RE.match(line)
if not match:
continue
level = len(match.group(1))
if level == 2:
if current_topic is not None and not current_has_section:
violations.append(
{
"type": "empty_topic",
"line": current_topic[0],
"heading": current_topic[1],
"reason": "topic has no allowed h3 section",
}
)
current_topic = (line_number, line)
current_has_section = False
elif level == 3 and current_topic is not None:
title = match.group(2).strip().strip("*")
if title in ALLOWED_TOPIC_SECTIONS:
current_has_section = True
if current_topic is not None and not current_has_section:
violations.append(
{
"type": "empty_topic",
"line": current_topic[0],
"heading": current_topic[1],
"reason": "topic has no allowed h3 section",
}
)
return violations
def validate_working_protocol_markdown(markdown: str) -> WorkingProtocolValidationReport:
violations: list[dict[str, Any]] = []
if not markdown.startswith(REQUIRED_TITLE):
violations.append(
{
"type": "missing_required_heading",
"expected": REQUIRED_TITLE,
"reason": "document must begin exactly with '# Working Protocol'",
}
)
elif not markdown.startswith(REQUIRED_TITLE + "\n"):
violations.append(
{
"type": "invalid_required_heading",
"expected": REQUIRED_TITLE,
"reason": "required heading must occupy the complete first line",
}
)
if markdown.startswith(REQUIRED_TITLE):
violations.extend(validate_markdown_shape(markdown))
return WorkingProtocolValidationReport(
valid=len(violations) == 0,
violations=violations,
)
def write_json(path: Path, data: Any) -> None:
path.parent.mkdir(parents=True, exist_ok=True)
path.write_text(json.dumps(data, ensure_ascii=False, indent=2) + "\n", encoding="utf-8")
def render_working_protocol(
input_path: Path,
output_dir: Path,
model: str = DEFAULT_MODEL,
endpoint: str = DEFAULT_ENDPOINT,
timeout: int = DEFAULT_TIMEOUT,
num_ctx: int = DEFAULT_NUM_CTX,
num_predict: int = DEFAULT_NUM_PREDICT,
think: bool = False,
) -> dict[str, Any]:
output_dir.mkdir(parents=True, exist_ok=True)
prompt_path = Path("prompts") / PROMPT_NAME
raw_response_path = output_dir / "raw_model_response.json"
raw_text_path = output_dir / "raw_model_response.txt"
candidate_path = output_dir / "cleaned_candidate.md"
validation_path = output_dir / "validation_report.json"
protocol_path = output_dir / "working_protocol.md"
metadata_path = output_dir / "metadata.json"
report_path = output_dir / "report.md"
input_text = input_path.read_text(encoding="utf-8-sig")
prompt = build_renderer_prompt(input_text)
payload = build_ollama_payload(model, prompt, num_ctx, num_predict, think)
data, runtime = call_ollama(endpoint, payload, timeout)
response_text = response_text_from_ollama_data(data)
raw_response_path.write_text(
json.dumps(data, ensure_ascii=False, indent=2) + "\n",
encoding="utf-8",
)
raw_text_path.write_text(response_text, encoding="utf-8")
cleaned = clean_working_protocol_markdown(response_text)
candidate_path.write_text(cleaned, encoding="utf-8")
validation = validate_working_protocol_markdown(cleaned)
write_json(validation_path, validation.to_dict())
if validation.valid:
protocol_path.write_text(cleaned, encoding="utf-8")
elif protocol_path.exists():
protocol_path.unlink()
metadata = {
"model": model,
"endpoint": endpoint,
"think": think,
"stream": False,
"temperature": 0.0,
"num_ctx": num_ctx,
"num_predict": num_predict,
"timeout_seconds": timeout,
"request_count": 1,
"runtime_seconds": runtime,
"prompt_path": str(prompt_path.resolve()),
"input_path": str(input_path.resolve()),
"output_path": str(protocol_path.resolve()) if validation.valid else None,
"raw_response_path": str(raw_response_path.resolve()),
"raw_text_path": str(raw_text_path.resolve()),
"cleaned_candidate_path": str(candidate_path.resolve()),
"validation_report_path": str(validation_path.resolve()),
"http_status_code": 200,
"response_text_length": len(response_text),
"candidate_text_length": len(cleaned),
"valid": validation.valid,
"readable_markdown": validation.valid,
"top_level_json_keys": sorted(data.keys()),
"done": data.get("done"),
"done_reason": data.get("done_reason"),
"total_duration": data.get("total_duration"),
"load_duration": data.get("load_duration"),
"prompt_eval_count": data.get("prompt_eval_count"),
"prompt_eval_duration": data.get("prompt_eval_duration"),
"eval_count": data.get("eval_count"),
"eval_duration": data.get("eval_duration"),
"created_at": datetime.now().isoformat(timespec="seconds"),
}
write_json(metadata_path, metadata)
lines = [
"# Working Protocol Renderer V2 Report",
"",
f"- Result: {'valid renderer run' if validation.valid else 'invalid renderer run'}",
f"- Model: `{model}`",
f"- Runtime: {runtime:.3f} seconds",
f"- Request count: 1",
f"- Valid: {validation.valid}",
f"- Violations: {len(validation.violations)}",
f"- Raw response path: `{raw_response_path}`",
f"- Cleaned candidate path: `{candidate_path}`",
f"- Validation report path: `{validation_path}`",
f"- Output path: `{protocol_path if validation.valid else 'not written'}`",
]
report_path.write_text("\n".join(lines) + "\n", encoding="utf-8")
return metadata
def main() -> int:
args = parse_args()
try:
metadata = render_working_protocol(
input_path=args.input,
output_dir=args.output_dir,
model=args.model,
endpoint=args.endpoint,
timeout=args.timeout,
num_ctx=args.num_ctx,
num_predict=args.num_predict,
think=args.think,
)
except requests.ConnectionError as exc:
print(f"Error: Ollama is not reachable at {args.endpoint}: {exc}", file=sys.stderr)
return 1
except requests.Timeout as exc:
print(f"Error: Ollama request timed out after {args.timeout} seconds: {exc}", file=sys.stderr)
return 1
except requests.HTTPError as exc:
print(f"Error: Ollama returned an HTTP error: {exc}", file=sys.stderr)
return 1
except (OSError, UnicodeError, ValueError, json.JSONDecodeError) as exc:
print(f"Error: {exc}", file=sys.stderr)
return 1
print(f"Runtime seconds: {metadata['runtime_seconds']:.3f}")
print(f"Validation result: {'passed' if metadata['valid'] else 'failed'}")
print(f"Output: {metadata['output_path'] or 'not written'}")
print(f"Raw model response: {metadata['raw_response_path']}")
print(f"Validation report: {metadata['validation_report_path']}")
return 0 if metadata["valid"] else 1
if __name__ == "__main__":
raise SystemExit(main())