Stabilize Meeting Lab pipeline for RC1 evaluation

This commit significantly improves the robustness and determinism of the Meeting Lab processing pipeline and establishes the first Release Candidate baseline for end-to-end evaluation.

Highlights

- BUG-009
  - Implement deterministic responsible-party validation
  - Normalize participant aliases using Meeting Context
  - Reject invalid responsible values (dates, locations, technical terms, projects, products, unknown entities)
  - Record structured responsibility validation metadata
  - Add focused regression tests

- BUG-010
  - Implement adaptive num_predict estimation for Semantic Consolidator
  - Eliminate JSON truncation caused by fixed output limits
  - Add deterministic source coverage repair
  - Preserve strict post-repair validation
  - Add regression tests

- BUG-011
  - Implement Working Protocol V2 renderer contract enforcement
  - Preserve raw renderer responses
  - Reject invalid protocol output instead of accepting malformed documents
  - Add deterministic cleanup for harmless formatting deviations
  - Add focused renderer regression tests

- Meeting Context
  - Validate Meeting Context V1
  - Integrate authoritative participant alias normalization

- Documentation
  - Update architecture documentation
  - Update output documentation
  - Update regression bug tracker

The pipeline now fails safely instead of silently accepting invalid intermediate or final artifacts.

Remaining work focuses primarily on extraction quality and semantic classification (decisions, action items, protocol faithfulness), rather than pipeline robustness.
This commit is contained in:
2026-08-04 13:11:54 +02:00
parent 60a8acae91
commit 950284e236
10 changed files with 1816 additions and 18 deletions
@@ -4,6 +4,7 @@
from __future__ import annotations
import argparse
import copy
import json
import sys
import time
@@ -25,8 +26,13 @@ except ModuleNotFoundError: # pragma: no cover - used by repository-root tests.
DEFAULT_MODEL = "qwen3.5:9b"
DEFAULT_ENDPOINT = "http://127.0.0.1:11434/api/generate"
DEFAULT_NUM_CTX = 32768
DEFAULT_NUM_PREDICT = 4096
DEFAULT_MIN_NUM_PREDICT = 4096
DEFAULT_NUM_PREDICT = DEFAULT_MIN_NUM_PREDICT
DEFAULT_PROGRESS_INTERVAL = 30
OUTPUT_CONTEXT_RESERVE_TOKENS = 1024
OUTPUT_TOKEN_ESTIMATE_CHARS = 4
OUTPUT_GROUP_OVERHEAD_CHARS = 320
OUTPUT_SAFETY_MARGIN = 1.35
PROMPT_NAME = "consolidate_facts.md"
@@ -75,10 +81,11 @@ def parse_args() -> argparse.Namespace:
parser.add_argument(
"--num-predict",
type=int,
default=DEFAULT_NUM_PREDICT,
default=None,
help=(
"Maximum generated tokens. The default is bounded for the expected "
f"fact-group JSON while leaving truncation headroom (default: {DEFAULT_NUM_PREDICT})."
"Maximum generated tokens. By default this is estimated from the "
"fact payload size and bounded by the context window. Explicit "
"values preserve the previous fixed-budget behavior."
),
)
thinking = parser.add_mutually_exclusive_group()
@@ -158,6 +165,49 @@ def build_consolidation_prompt(facts: list[dict[str, Any]]) -> str:
return f"{task_prompt}\n\nFACT ITEMS:\n{payload}\n"
def estimate_response_tokens(facts: list[dict[str, Any]]) -> int:
"""
Estimate the token budget needed for the model's grouping JSON.
Semantic Consolidator V0 asks the model to return one group per source fact
unless it finds a conservative duplicate. The response therefore scales with
the number and text size of fact items. The estimate intentionally includes
per-group JSON overhead and a safety margin; strict validation still decides
whether the actual response is usable.
"""
text_chars = 0
for item in facts:
text_chars += len(str(item.get("text", "")))
text_chars += len(str(item.get("evidence", "")))
estimated_chars = int(
(text_chars + len(facts) * OUTPUT_GROUP_OVERHEAD_CHARS)
* OUTPUT_SAFETY_MARGIN
)
return max(
DEFAULT_MIN_NUM_PREDICT,
(estimated_chars + OUTPUT_TOKEN_ESTIMATE_CHARS - 1)
// OUTPUT_TOKEN_ESTIMATE_CHARS,
)
def resolve_num_predict(
requested_num_predict: int | None,
facts: list[dict[str, Any]],
prompt_token_estimate: int,
num_ctx: int,
) -> int:
if requested_num_predict is not None:
return requested_num_predict
estimated = estimate_response_tokens(facts)
max_available = max(
DEFAULT_MIN_NUM_PREDICT,
num_ctx - prompt_token_estimate - OUTPUT_CONTEXT_RESERVE_TOKENS,
)
return min(estimated, max_available)
def response_text_from_ollama_data(data: dict[str, Any]) -> str | None:
text = data.get("response")
if isinstance(text, str) and text.strip():
@@ -370,6 +420,96 @@ def validate_group_shapes(groups: list[dict[str, Any]]) -> None:
raise ConsolidationValidationError("Merged groups need at least two IDs.")
def repair_model_group_coverage(
model_output: dict[str, Any],
facts: list[dict[str, Any]],
) -> tuple[dict[str, Any], list[dict[str, Any]]]:
"""
Apply deterministic source-coverage repairs to model grouping JSON.
The repair is intentionally conservative. Repeated source IDs are removed
after their first occurrence, empty groups created by that removal are
dropped, and missing facts are restored as singleton groups using the
original canonicalized fact text. No existing group text, merge reason or
semantic merge is rewritten.
"""
groups = model_output.get("groups")
if not isinstance(groups, list):
return model_output, []
repaired = copy.deepcopy(model_output)
repaired_groups = repaired["groups"]
facts_by_id = {str(item.get("item_id")): item for item in facts}
expected_ids = set(facts_by_id)
seen: set[str] = set()
changes: list[dict[str, Any]] = []
for group_index, group in enumerate(repaired_groups):
if not isinstance(group, dict):
continue
source_ids = group.get("source_item_ids")
if not isinstance(source_ids, list):
continue
kept_ids: list[str] = []
for id_index, item_id in enumerate(source_ids):
if not isinstance(item_id, str) or item_id not in expected_ids:
kept_ids.append(item_id)
continue
if item_id in seen:
changes.append(
{
"operation": "remove_duplicate_source_id",
"id": item_id,
"group_index": group_index,
"id_index": id_index,
}
)
continue
seen.add(item_id)
kept_ids.append(item_id)
group["source_item_ids"] = kept_ids
non_empty_groups: list[dict[str, Any]] = []
for group_index, group in enumerate(repaired_groups):
if (
isinstance(group, dict)
and isinstance(group.get("source_item_ids"), list)
and len(group["source_item_ids"]) == 0
):
changes.append(
{
"operation": "remove_empty_group",
"group_index": group_index,
"canonical_text": group.get("canonical_text"),
}
)
continue
non_empty_groups.append(group)
repaired["groups"] = non_empty_groups
missing_ids = sorted(expected_ids - seen)
for item_id in missing_ids:
fact = facts_by_id[item_id]
repaired["groups"].append(
{
"canonical_text": str(fact.get("text", "")).strip(),
"source_item_ids": [item_id],
"merge_reason": (
"Deterministic coverage repair: source fact was missing "
"from the model grouping and is preserved as a singleton."
),
}
)
changes.append(
{
"operation": "restore_missing_source_id_as_singleton",
"id": item_id,
}
)
return repaired, changes
def build_consolidated_fact_item(
group: dict[str, Any],
fact_by_id: dict[str, dict[str, Any]],
@@ -481,9 +621,11 @@ def write_report(
runtime: float,
prompt_chars: int,
prompt_token_estimate: int,
num_predict: int,
fact_count: int,
groups: list[dict[str, Any]],
output_path: Path,
repair_changes: list[dict[str, Any]] | None = None,
) -> None:
merged = [group for group in groups if len(group["source_item_ids"]) > 1]
singletons = [group for group in groups if len(group["source_item_ids"]) == 1]
@@ -497,9 +639,11 @@ def write_report(
f"- Fact item count: {fact_count}",
f"- Prompt characters: {prompt_chars}",
f"- Estimated prompt tokens: {prompt_token_estimate}",
f"- num_predict: {num_predict}",
f"- Merged fact groups: {len(merged)}",
f"- Source facts involved in merges: {sum(len(group['source_item_ids']) for group in merged)}",
f"- Singleton fact groups: {len(singletons)}",
f"- Deterministic repair changes: {len(repair_changes or [])}",
f"- Output path: `{output_path}`",
"",
"## Actual Merges",
@@ -518,6 +662,10 @@ def write_report(
"",
]
)
if repair_changes:
lines.extend(["", "## Deterministic Coverage Repairs", ""])
for change in repair_changes:
lines.append(f"- `{change['operation']}`: {json.dumps(change, ensure_ascii=False, sort_keys=True)}")
path.write_text("\n".join(lines).rstrip() + "\n", encoding="utf-8")
@@ -527,6 +675,7 @@ def main() -> int:
raw_response_path = args.output_dir / "raw_model_response.txt"
output_path = args.output_dir / "consolidated_extractions.json"
report_path = args.output_dir / "report.md"
repair_metadata_path = args.output_dir / "repair_metadata.json"
try:
canonicalized = load_json_object(args.canonicalized_input)
@@ -534,9 +683,16 @@ def main() -> int:
prompt = build_consolidation_prompt(facts)
prompt_chars = len(prompt)
prompt_token_estimate = (prompt_chars + 3) // 4
num_predict = resolve_num_predict(
requested_num_predict=args.num_predict,
facts=facts,
prompt_token_estimate=prompt_token_estimate,
num_ctx=args.num_ctx,
)
print(f"Fact item count: {len(facts)}")
print(f"Estimated prompt size chars: {prompt_chars}")
print(f"Estimated prompt tokens: {prompt_token_estimate}")
print(f"Resolved num_predict: {num_predict}")
print("Expected LLM call count: 1")
print("Expected runtime: 5-10 minutes on current local benchmark basis")
@@ -546,12 +702,23 @@ def main() -> int:
prompt=prompt,
timeout=args.timeout,
num_ctx=args.num_ctx,
num_predict=args.num_predict,
num_predict=num_predict,
think=args.think,
progress_interval=args.progress_interval,
)
raw_response_path.write_text(raw_text + "\n", encoding="utf-8")
model_output = parse_model_json(raw_text)
model_output, repair_changes = repair_model_group_coverage(model_output, facts)
if repair_changes:
write_json(
repair_metadata_path,
{
"scope": "semantic_consolidator_v0_source_coverage",
"llm_used": False,
"repair_count": len(repair_changes),
"repairs": repair_changes,
},
)
expected_fact_ids = {item["item_id"] for item in facts}
groups = validate_model_groups(model_output, expected_fact_ids)
validate_group_shapes(groups)
@@ -564,9 +731,11 @@ def main() -> int:
runtime=runtime,
prompt_chars=prompt_chars,
prompt_token_estimate=prompt_token_estimate,
num_predict=num_predict,
fact_count=len(facts),
groups=groups,
output_path=output_path,
repair_changes=repair_changes,
)
except requests.ConnectionError as exc:
print(f"Error: Ollama is not reachable at {args.endpoint}: {exc}", file=sys.stderr)