This commit significantly improves the robustness and determinism of the Meeting Lab processing pipeline and establishes the first Release Candidate baseline for end-to-end evaluation. Highlights - BUG-009 - Implement deterministic responsible-party validation - Normalize participant aliases using Meeting Context - Reject invalid responsible values (dates, locations, technical terms, projects, products, unknown entities) - Record structured responsibility validation metadata - Add focused regression tests - BUG-010 - Implement adaptive num_predict estimation for Semantic Consolidator - Eliminate JSON truncation caused by fixed output limits - Add deterministic source coverage repair - Preserve strict post-repair validation - Add regression tests - BUG-011 - Implement Working Protocol V2 renderer contract enforcement - Preserve raw renderer responses - Reject invalid protocol output instead of accepting malformed documents - Add deterministic cleanup for harmless formatting deviations - Add focused renderer regression tests - Meeting Context - Validate Meeting Context V1 - Integrate authoritative participant alias normalization - Documentation - Update architecture documentation - Update output documentation - Update regression bug tracker The pipeline now fails safely instead of silently accepting invalid intermediate or final artifacts. Remaining work focuses primarily on extraction quality and semantic classification (decisions, action items, protocol faithfulness), rather than pipeline robustness.
520 lines
17 KiB
Python
520 lines
17 KiB
Python
#!/usr/bin/env python3
|
|
"""LLM-backed Working Protocol V2 renderer with deterministic contract checks."""
|
|
|
|
from __future__ import annotations
|
|
|
|
import argparse
|
|
import json
|
|
import re
|
|
import sys
|
|
import time
|
|
from dataclasses import dataclass
|
|
from datetime import datetime
|
|
from pathlib import Path
|
|
from typing import Any
|
|
|
|
import requests
|
|
|
|
try:
|
|
from meeting_lab.llm.prompts import load_prompt
|
|
except ModuleNotFoundError: # pragma: no cover - used by repository-root tests.
|
|
from src.meeting_lab.llm.prompts import load_prompt
|
|
|
|
|
|
DEFAULT_MODEL = "qwen3.5:9b"
|
|
DEFAULT_ENDPOINT = "http://127.0.0.1:11434/api/generate"
|
|
DEFAULT_NUM_CTX = 32768
|
|
DEFAULT_NUM_PREDICT = 4096
|
|
DEFAULT_TIMEOUT = 1800
|
|
PROMPT_NAME = "working_protocol.md"
|
|
REQUIRED_TITLE = "# Working Protocol"
|
|
ALLOWED_TOPIC_SECTIONS = {
|
|
"Background",
|
|
"Decisions",
|
|
"Action Items",
|
|
"Open Questions",
|
|
}
|
|
GENERIC_SUMMARY_HEADINGS = {
|
|
"Entscheidungen",
|
|
"Decisions",
|
|
"Handlungsaufträge",
|
|
"Handlungsauftraege",
|
|
"Action Items",
|
|
"Offene Fragen",
|
|
"Open Questions",
|
|
"Technische Details",
|
|
"Technical Details",
|
|
"Fakten",
|
|
"Facts",
|
|
"Zusammenfassung",
|
|
"Summary",
|
|
"Konsolidierter Projektstatusbericht",
|
|
}
|
|
HEADING_RE = re.compile(r"^(#{1,6})\s+(.+?)\s*$")
|
|
|
|
|
|
class WorkingProtocolValidationError(ValueError):
|
|
"""Raised when rendered Markdown violates the Working Protocol V2 contract."""
|
|
|
|
|
|
@dataclass(frozen=True)
|
|
class WorkingProtocolValidationReport:
|
|
valid: bool
|
|
violations: list[dict[str, Any]]
|
|
|
|
def to_dict(self) -> dict[str, Any]:
|
|
return {"valid": self.valid, "violations": self.violations}
|
|
|
|
|
|
def parse_args() -> argparse.Namespace:
|
|
parser = argparse.ArgumentParser(
|
|
description="Render a consolidated meeting representation as Working Protocol V2."
|
|
)
|
|
parser.add_argument("input", type=Path, help="Consolidated meeting JSON input.")
|
|
parser.add_argument(
|
|
"-o",
|
|
"--output-dir",
|
|
type=Path,
|
|
required=True,
|
|
help="Directory for working_protocol.md, raw response, validation and metadata.",
|
|
)
|
|
parser.add_argument(
|
|
"--model",
|
|
default=DEFAULT_MODEL,
|
|
help=f"Ollama model name (default: {DEFAULT_MODEL}).",
|
|
)
|
|
parser.add_argument(
|
|
"--endpoint",
|
|
default=DEFAULT_ENDPOINT,
|
|
help=f"Ollama generate endpoint (default: {DEFAULT_ENDPOINT}).",
|
|
)
|
|
parser.add_argument(
|
|
"--timeout",
|
|
type=int,
|
|
default=DEFAULT_TIMEOUT,
|
|
help=f"HTTP timeout in seconds (default: {DEFAULT_TIMEOUT}).",
|
|
)
|
|
parser.add_argument(
|
|
"--num-ctx",
|
|
type=int,
|
|
default=DEFAULT_NUM_CTX,
|
|
help=f"Context window tokens (default: {DEFAULT_NUM_CTX}).",
|
|
)
|
|
parser.add_argument(
|
|
"--num-predict",
|
|
type=int,
|
|
default=DEFAULT_NUM_PREDICT,
|
|
help=f"Maximum generated tokens (default: {DEFAULT_NUM_PREDICT}).",
|
|
)
|
|
thinking = parser.add_mutually_exclusive_group()
|
|
thinking.add_argument("--think", dest="think", action="store_true")
|
|
thinking.add_argument("--no-think", dest="think", action="store_false")
|
|
parser.set_defaults(think=False)
|
|
return parser.parse_args()
|
|
|
|
|
|
def build_renderer_prompt(input_text: str) -> str:
|
|
return f"{load_prompt(PROMPT_NAME)}\n\nINPUT JSON:\n{input_text}\n"
|
|
|
|
|
|
def response_text_from_ollama_data(data: dict[str, Any]) -> str:
|
|
text = data.get("response")
|
|
if isinstance(text, str):
|
|
return text
|
|
message = data.get("message")
|
|
if isinstance(message, dict) and isinstance(message.get("content"), str):
|
|
return message["content"]
|
|
return ""
|
|
|
|
|
|
def build_ollama_payload(
|
|
model: str,
|
|
prompt: str,
|
|
num_ctx: int,
|
|
num_predict: int,
|
|
think: bool,
|
|
) -> dict[str, Any]:
|
|
return {
|
|
"model": model,
|
|
"prompt": prompt,
|
|
"think": think,
|
|
"stream": False,
|
|
"options": {
|
|
"temperature": 0.0,
|
|
"num_ctx": num_ctx,
|
|
"num_predict": num_predict,
|
|
},
|
|
}
|
|
|
|
|
|
def call_ollama(
|
|
endpoint: str,
|
|
payload: dict[str, Any],
|
|
timeout: int,
|
|
) -> tuple[dict[str, Any], float]:
|
|
start = time.perf_counter()
|
|
response = requests.post(endpoint, json=payload, timeout=timeout)
|
|
runtime = time.perf_counter() - start
|
|
response.raise_for_status()
|
|
data = response.json()
|
|
if not isinstance(data, dict):
|
|
raise ValueError("Ollama response must be a JSON object.")
|
|
return data, runtime
|
|
|
|
|
|
def normalize_heading_whitespace(line: str) -> str:
|
|
match = HEADING_RE.match(line.strip())
|
|
if not match:
|
|
return line.rstrip()
|
|
return f"{match.group(1)} {match.group(2).strip()}"
|
|
|
|
|
|
def clean_working_protocol_markdown(text: str) -> str:
|
|
"""
|
|
Remove only non-semantic wrapper text before an existing protocol heading.
|
|
|
|
The cleanup does not create headings or transform a categorized summary into
|
|
a Working Protocol. It only makes an already present contract heading the
|
|
first byte of the candidate document.
|
|
"""
|
|
text = text.replace("\r\n", "\n").replace("\r", "\n")
|
|
lines = text.split("\n")
|
|
start_index: int | None = None
|
|
for index, line in enumerate(lines):
|
|
if normalize_heading_whitespace(line) == REQUIRED_TITLE:
|
|
start_index = index
|
|
break
|
|
if start_index is None:
|
|
return text.lstrip()
|
|
|
|
cleaned_lines = lines[start_index:]
|
|
if cleaned_lines:
|
|
cleaned_lines[0] = REQUIRED_TITLE
|
|
return "\n".join(normalize_heading_whitespace(line) for line in cleaned_lines).strip() + "\n"
|
|
|
|
|
|
def validate_markdown_shape(markdown: str) -> list[dict[str, Any]]:
|
|
violations: list[dict[str, Any]] = []
|
|
if markdown.count("```") % 2 != 0:
|
|
violations.append(
|
|
{
|
|
"type": "malformed_markdown",
|
|
"reason": "unclosed_fenced_code_block",
|
|
}
|
|
)
|
|
|
|
seen_title = False
|
|
seen_topic = False
|
|
current_topic_has_section = False
|
|
topic_count = 0
|
|
section_count = 0
|
|
|
|
for line_number, line in enumerate(markdown.splitlines(), start=1):
|
|
match = HEADING_RE.match(line)
|
|
if not match:
|
|
continue
|
|
level = len(match.group(1))
|
|
title = match.group(2).strip().strip("*")
|
|
|
|
if level == 1:
|
|
if line_number != 1 or title != "Working Protocol":
|
|
violations.append(
|
|
{
|
|
"type": "invalid_heading",
|
|
"line": line_number,
|
|
"heading": line,
|
|
"reason": "only the first line may be '# Working Protocol'",
|
|
}
|
|
)
|
|
seen_title = True
|
|
continue
|
|
|
|
if not seen_title:
|
|
violations.append(
|
|
{
|
|
"type": "invalid_heading_order",
|
|
"line": line_number,
|
|
"heading": line,
|
|
"reason": "heading appears before required title",
|
|
}
|
|
)
|
|
continue
|
|
|
|
if level == 2:
|
|
topic_count += 1
|
|
seen_topic = True
|
|
current_topic_has_section = False
|
|
if title in GENERIC_SUMMARY_HEADINGS:
|
|
violations.append(
|
|
{
|
|
"type": "generic_summary_framing",
|
|
"line": line_number,
|
|
"heading": line,
|
|
"reason": "top-level category heading is not a topic",
|
|
}
|
|
)
|
|
continue
|
|
|
|
if level == 3:
|
|
if not seen_topic:
|
|
violations.append(
|
|
{
|
|
"type": "missing_topic",
|
|
"line": line_number,
|
|
"heading": line,
|
|
"reason": "section appears before any topic heading",
|
|
}
|
|
)
|
|
if title not in ALLOWED_TOPIC_SECTIONS:
|
|
violations.append(
|
|
{
|
|
"type": "unknown_topic_section",
|
|
"line": line_number,
|
|
"heading": line,
|
|
"allowed": sorted(ALLOWED_TOPIC_SECTIONS),
|
|
}
|
|
)
|
|
else:
|
|
section_count += 1
|
|
current_topic_has_section = True
|
|
continue
|
|
|
|
violations.append(
|
|
{
|
|
"type": "unsupported_heading_level",
|
|
"line": line_number,
|
|
"heading": line,
|
|
"reason": "Working Protocol V2 uses h1 title, h2 topics and h3 sections only",
|
|
}
|
|
)
|
|
|
|
if seen_topic and not current_topic_has_section:
|
|
# This catches the last topic; earlier empty topics are caught below by
|
|
# counting consecutive h2 headings.
|
|
pass
|
|
|
|
h2_without_section = _topic_headings_without_sections(markdown)
|
|
violations.extend(h2_without_section)
|
|
|
|
if topic_count == 0:
|
|
violations.append(
|
|
{
|
|
"type": "missing_topic",
|
|
"reason": "document must contain at least one '## Topic title' section",
|
|
}
|
|
)
|
|
if section_count == 0:
|
|
violations.append(
|
|
{
|
|
"type": "missing_topic_sections",
|
|
"reason": "document must contain at least one allowed h3 topic section",
|
|
}
|
|
)
|
|
return violations
|
|
|
|
|
|
def _topic_headings_without_sections(markdown: str) -> list[dict[str, Any]]:
|
|
violations: list[dict[str, Any]] = []
|
|
current_topic: tuple[int, str] | None = None
|
|
current_has_section = False
|
|
|
|
for line_number, line in enumerate(markdown.splitlines(), start=1):
|
|
match = HEADING_RE.match(line)
|
|
if not match:
|
|
continue
|
|
level = len(match.group(1))
|
|
if level == 2:
|
|
if current_topic is not None and not current_has_section:
|
|
violations.append(
|
|
{
|
|
"type": "empty_topic",
|
|
"line": current_topic[0],
|
|
"heading": current_topic[1],
|
|
"reason": "topic has no allowed h3 section",
|
|
}
|
|
)
|
|
current_topic = (line_number, line)
|
|
current_has_section = False
|
|
elif level == 3 and current_topic is not None:
|
|
title = match.group(2).strip().strip("*")
|
|
if title in ALLOWED_TOPIC_SECTIONS:
|
|
current_has_section = True
|
|
|
|
if current_topic is not None and not current_has_section:
|
|
violations.append(
|
|
{
|
|
"type": "empty_topic",
|
|
"line": current_topic[0],
|
|
"heading": current_topic[1],
|
|
"reason": "topic has no allowed h3 section",
|
|
}
|
|
)
|
|
return violations
|
|
|
|
|
|
def validate_working_protocol_markdown(markdown: str) -> WorkingProtocolValidationReport:
|
|
violations: list[dict[str, Any]] = []
|
|
if not markdown.startswith(REQUIRED_TITLE):
|
|
violations.append(
|
|
{
|
|
"type": "missing_required_heading",
|
|
"expected": REQUIRED_TITLE,
|
|
"reason": "document must begin exactly with '# Working Protocol'",
|
|
}
|
|
)
|
|
elif not markdown.startswith(REQUIRED_TITLE + "\n"):
|
|
violations.append(
|
|
{
|
|
"type": "invalid_required_heading",
|
|
"expected": REQUIRED_TITLE,
|
|
"reason": "required heading must occupy the complete first line",
|
|
}
|
|
)
|
|
|
|
if markdown.startswith(REQUIRED_TITLE):
|
|
violations.extend(validate_markdown_shape(markdown))
|
|
|
|
return WorkingProtocolValidationReport(
|
|
valid=len(violations) == 0,
|
|
violations=violations,
|
|
)
|
|
|
|
|
|
def write_json(path: Path, data: Any) -> None:
|
|
path.parent.mkdir(parents=True, exist_ok=True)
|
|
path.write_text(json.dumps(data, ensure_ascii=False, indent=2) + "\n", encoding="utf-8")
|
|
|
|
|
|
def render_working_protocol(
|
|
input_path: Path,
|
|
output_dir: Path,
|
|
model: str = DEFAULT_MODEL,
|
|
endpoint: str = DEFAULT_ENDPOINT,
|
|
timeout: int = DEFAULT_TIMEOUT,
|
|
num_ctx: int = DEFAULT_NUM_CTX,
|
|
num_predict: int = DEFAULT_NUM_PREDICT,
|
|
think: bool = False,
|
|
) -> dict[str, Any]:
|
|
output_dir.mkdir(parents=True, exist_ok=True)
|
|
prompt_path = Path("prompts") / PROMPT_NAME
|
|
raw_response_path = output_dir / "raw_model_response.json"
|
|
raw_text_path = output_dir / "raw_model_response.txt"
|
|
candidate_path = output_dir / "cleaned_candidate.md"
|
|
validation_path = output_dir / "validation_report.json"
|
|
protocol_path = output_dir / "working_protocol.md"
|
|
metadata_path = output_dir / "metadata.json"
|
|
report_path = output_dir / "report.md"
|
|
|
|
input_text = input_path.read_text(encoding="utf-8-sig")
|
|
prompt = build_renderer_prompt(input_text)
|
|
payload = build_ollama_payload(model, prompt, num_ctx, num_predict, think)
|
|
|
|
data, runtime = call_ollama(endpoint, payload, timeout)
|
|
response_text = response_text_from_ollama_data(data)
|
|
raw_response_path.write_text(
|
|
json.dumps(data, ensure_ascii=False, indent=2) + "\n",
|
|
encoding="utf-8",
|
|
)
|
|
raw_text_path.write_text(response_text, encoding="utf-8")
|
|
|
|
cleaned = clean_working_protocol_markdown(response_text)
|
|
candidate_path.write_text(cleaned, encoding="utf-8")
|
|
validation = validate_working_protocol_markdown(cleaned)
|
|
write_json(validation_path, validation.to_dict())
|
|
|
|
if validation.valid:
|
|
protocol_path.write_text(cleaned, encoding="utf-8")
|
|
elif protocol_path.exists():
|
|
protocol_path.unlink()
|
|
|
|
metadata = {
|
|
"model": model,
|
|
"endpoint": endpoint,
|
|
"think": think,
|
|
"stream": False,
|
|
"temperature": 0.0,
|
|
"num_ctx": num_ctx,
|
|
"num_predict": num_predict,
|
|
"timeout_seconds": timeout,
|
|
"request_count": 1,
|
|
"runtime_seconds": runtime,
|
|
"prompt_path": str(prompt_path.resolve()),
|
|
"input_path": str(input_path.resolve()),
|
|
"output_path": str(protocol_path.resolve()) if validation.valid else None,
|
|
"raw_response_path": str(raw_response_path.resolve()),
|
|
"raw_text_path": str(raw_text_path.resolve()),
|
|
"cleaned_candidate_path": str(candidate_path.resolve()),
|
|
"validation_report_path": str(validation_path.resolve()),
|
|
"http_status_code": 200,
|
|
"response_text_length": len(response_text),
|
|
"candidate_text_length": len(cleaned),
|
|
"valid": validation.valid,
|
|
"readable_markdown": validation.valid,
|
|
"top_level_json_keys": sorted(data.keys()),
|
|
"done": data.get("done"),
|
|
"done_reason": data.get("done_reason"),
|
|
"total_duration": data.get("total_duration"),
|
|
"load_duration": data.get("load_duration"),
|
|
"prompt_eval_count": data.get("prompt_eval_count"),
|
|
"prompt_eval_duration": data.get("prompt_eval_duration"),
|
|
"eval_count": data.get("eval_count"),
|
|
"eval_duration": data.get("eval_duration"),
|
|
"created_at": datetime.now().isoformat(timespec="seconds"),
|
|
}
|
|
write_json(metadata_path, metadata)
|
|
|
|
lines = [
|
|
"# Working Protocol Renderer V2 Report",
|
|
"",
|
|
f"- Result: {'valid renderer run' if validation.valid else 'invalid renderer run'}",
|
|
f"- Model: `{model}`",
|
|
f"- Runtime: {runtime:.3f} seconds",
|
|
f"- Request count: 1",
|
|
f"- Valid: {validation.valid}",
|
|
f"- Violations: {len(validation.violations)}",
|
|
f"- Raw response path: `{raw_response_path}`",
|
|
f"- Cleaned candidate path: `{candidate_path}`",
|
|
f"- Validation report path: `{validation_path}`",
|
|
f"- Output path: `{protocol_path if validation.valid else 'not written'}`",
|
|
]
|
|
report_path.write_text("\n".join(lines) + "\n", encoding="utf-8")
|
|
return metadata
|
|
|
|
|
|
def main() -> int:
|
|
args = parse_args()
|
|
try:
|
|
metadata = render_working_protocol(
|
|
input_path=args.input,
|
|
output_dir=args.output_dir,
|
|
model=args.model,
|
|
endpoint=args.endpoint,
|
|
timeout=args.timeout,
|
|
num_ctx=args.num_ctx,
|
|
num_predict=args.num_predict,
|
|
think=args.think,
|
|
)
|
|
except requests.ConnectionError as exc:
|
|
print(f"Error: Ollama is not reachable at {args.endpoint}: {exc}", file=sys.stderr)
|
|
return 1
|
|
except requests.Timeout as exc:
|
|
print(f"Error: Ollama request timed out after {args.timeout} seconds: {exc}", file=sys.stderr)
|
|
return 1
|
|
except requests.HTTPError as exc:
|
|
print(f"Error: Ollama returned an HTTP error: {exc}", file=sys.stderr)
|
|
return 1
|
|
except (OSError, UnicodeError, ValueError, json.JSONDecodeError) as exc:
|
|
print(f"Error: {exc}", file=sys.stderr)
|
|
return 1
|
|
|
|
print(f"Runtime seconds: {metadata['runtime_seconds']:.3f}")
|
|
print(f"Validation result: {'passed' if metadata['valid'] else 'failed'}")
|
|
print(f"Output: {metadata['output_path'] or 'not written'}")
|
|
print(f"Raw model response: {metadata['raw_response_path']}")
|
|
print(f"Validation report: {metadata['validation_report_path']}")
|
|
return 0 if metadata["valid"] else 1
|
|
|
|
|
|
if __name__ == "__main__":
|
|
raise SystemExit(main())
|