Stabilize Meeting Lab pipeline for RC1 evaluation
This commit significantly improves the robustness and determinism of the Meeting Lab processing pipeline and establishes the first Release Candidate baseline for end-to-end evaluation. Highlights - BUG-009 - Implement deterministic responsible-party validation - Normalize participant aliases using Meeting Context - Reject invalid responsible values (dates, locations, technical terms, projects, products, unknown entities) - Record structured responsibility validation metadata - Add focused regression tests - BUG-010 - Implement adaptive num_predict estimation for Semantic Consolidator - Eliminate JSON truncation caused by fixed output limits - Add deterministic source coverage repair - Preserve strict post-repair validation - Add regression tests - BUG-011 - Implement Working Protocol V2 renderer contract enforcement - Preserve raw renderer responses - Reject invalid protocol output instead of accepting malformed documents - Add deterministic cleanup for harmless formatting deviations - Add focused renderer regression tests - Meeting Context - Validate Meeting Context V1 - Integrate authoritative participant alias normalization - Documentation - Update architecture documentation - Update output documentation - Update regression bug tracker The pipeline now fails safely instead of silently accepting invalid intermediate or final artifacts. Remaining work focuses primarily on extraction quality and semantic classification (decisions, action items, protocol faithfulness), rather than pipeline robustness.
This commit is contained in:
@@ -8,9 +8,12 @@ import json
|
||||
import re
|
||||
import sys
|
||||
from collections import Counter
|
||||
from dataclasses import dataclass
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
from src.meeting_lab.models.meeting_context import load_meeting_context
|
||||
|
||||
|
||||
SCHEMA_VERSION = "1"
|
||||
|
||||
@@ -36,6 +39,51 @@ CANONICAL_CATEGORIES = tuple(CATEGORY_NAMES.values())
|
||||
|
||||
CHUNK_EXTRACTION_RE = re.compile(r"^chunk_(\d+)_extraction\.json$")
|
||||
WHITESPACE_RE = re.compile(r"\s+")
|
||||
RESPONSIBLE_KEY_RE = re.compile(r"\s+")
|
||||
NUMERIC_DATE_RE = re.compile(r"(?i)\b\d{1,2}\s*[./]\s*(?:\d{1,2}|[a-zäöü]+)?\b")
|
||||
MONTH_DATE_RE = re.compile(
|
||||
r"(?i)\b\d{1,2}\.?\s*(?:oder\s+\d{1,2}\.?\s*)?"
|
||||
r"(januar|februar|märz|maerz|april|mai|juni|juli|august|september|oktober|november|dezember)\b"
|
||||
)
|
||||
TIME_RE = re.compile(r"(?i)\b\d{1,2}[:.]\d{2}\s*(?:uhr)?\b|\b\d{1,2}\s*uhr\b")
|
||||
DATE_FRAGMENT_RE = re.compile(r"(?i)\b(?:am|zum|bis|vor|nach)\s+\d{1,2}\.?\b")
|
||||
|
||||
DATE_WORDS = {
|
||||
"heute",
|
||||
"morgen",
|
||||
"übermorgen",
|
||||
"uebermorgen",
|
||||
"gestern",
|
||||
"vorgestern",
|
||||
}
|
||||
WEEKDAY_WORDS = {
|
||||
"montag",
|
||||
"dienstag",
|
||||
"mittwoch",
|
||||
"donnerstag",
|
||||
"freitag",
|
||||
"samstag",
|
||||
"sonntag",
|
||||
}
|
||||
RELATIVE_DATE_PHRASES = {
|
||||
"nächste woche",
|
||||
"naechste woche",
|
||||
"diese woche",
|
||||
"kommende woche",
|
||||
"nächsten monat",
|
||||
"naechsten monat",
|
||||
}
|
||||
GENERIC_RESPONSIBLE_WORDS = {
|
||||
"team",
|
||||
"projektteam",
|
||||
"projektleitung",
|
||||
"logistik",
|
||||
"alle",
|
||||
"autor",
|
||||
"labor",
|
||||
"partner",
|
||||
"gruppe",
|
||||
}
|
||||
|
||||
|
||||
def parse_args() -> argparse.Namespace:
|
||||
@@ -59,9 +107,110 @@ def parse_args() -> argparse.Namespace:
|
||||
action="store_true",
|
||||
help="Preserve exact duplicate items instead of merging them.",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--meeting-context",
|
||||
type=Path,
|
||||
help=(
|
||||
"Optional Meeting Context V1 YAML used to validate and normalize "
|
||||
"action-item responsible fields. If omitted, canonicalizer uses "
|
||||
"context provenance from extraction JSON when available."
|
||||
),
|
||||
)
|
||||
return parser.parse_args()
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class ResponsiblePartyNormalizer:
|
||||
allowed_names: dict[str, str]
|
||||
invalid_values: dict[str, str]
|
||||
|
||||
@classmethod
|
||||
def from_meeting_context_data(cls, data: dict[str, Any]) -> "ResponsiblePartyNormalizer":
|
||||
allowed: dict[str, str] = {}
|
||||
invalid: dict[str, str] = {}
|
||||
|
||||
for collection, id_key in (
|
||||
("participants", "participant_id"),
|
||||
("mentioned_people", "person_id"),
|
||||
):
|
||||
for person in data.get(collection, []) or []:
|
||||
if not isinstance(person, dict):
|
||||
continue
|
||||
display_name = clean_text(person.get("display_name"))
|
||||
if not display_name:
|
||||
continue
|
||||
for value in [display_name, *(person.get("aliases") or [])]:
|
||||
text = clean_text(value)
|
||||
if text:
|
||||
allowed[responsible_key(text)] = display_name
|
||||
|
||||
organization = data.get("organization")
|
||||
if isinstance(organization, dict):
|
||||
organization_name = clean_text(organization.get("name"))
|
||||
if organization_name:
|
||||
allowed[responsible_key(organization_name)] = organization_name
|
||||
|
||||
for department in organization.get("departments") or []:
|
||||
if not isinstance(department, dict):
|
||||
continue
|
||||
department_name = clean_text(department.get("name"))
|
||||
if department_name:
|
||||
allowed[responsible_key(department_name)] = department_name
|
||||
for alias in department.get("aliases") or []:
|
||||
text = clean_text(alias)
|
||||
if text and department_name:
|
||||
allowed[responsible_key(text)] = department_name
|
||||
|
||||
known_entities = data.get("known_entities")
|
||||
if isinstance(known_entities, dict):
|
||||
for collection in (
|
||||
"projects",
|
||||
"products",
|
||||
"systems",
|
||||
"locations",
|
||||
"technical_terms",
|
||||
):
|
||||
for value in known_entities.get(collection) or []:
|
||||
text = clean_text(value)
|
||||
if text:
|
||||
invalid[responsible_key(text)] = collection
|
||||
|
||||
return cls(allowed_names=allowed, invalid_values=invalid)
|
||||
|
||||
def normalize(
|
||||
self,
|
||||
value: str | None,
|
||||
item_id: str,
|
||||
) -> tuple[str | None, dict[str, Any] | None]:
|
||||
original = clean_text(value)
|
||||
if original is None:
|
||||
return None, None
|
||||
|
||||
key = responsible_key(original)
|
||||
canonical = self.allowed_names.get(key)
|
||||
if canonical is not None:
|
||||
return canonical, {
|
||||
"action_item_id": item_id,
|
||||
"field": "responsible",
|
||||
"original_value": original,
|
||||
"normalized_value": canonical,
|
||||
"status": "accepted",
|
||||
"reason": "resolved_to_meeting_context_entity",
|
||||
"cleared_to_null": False,
|
||||
}
|
||||
|
||||
reason = invalid_responsible_reason(original, self.invalid_values)
|
||||
return None, {
|
||||
"action_item_id": item_id,
|
||||
"field": "responsible",
|
||||
"original_value": original,
|
||||
"normalized_value": None,
|
||||
"status": "rejected",
|
||||
"reason": reason,
|
||||
"cleared_to_null": True,
|
||||
}
|
||||
|
||||
|
||||
def chunk_sort_key(path: Path) -> tuple[int, str]:
|
||||
match = CHUNK_EXTRACTION_RE.match(path.name)
|
||||
if not match:
|
||||
@@ -109,6 +258,30 @@ def clean_text(value: Any) -> str | None:
|
||||
return text if text else None
|
||||
|
||||
|
||||
def responsible_key(value: str) -> str:
|
||||
return RESPONSIBLE_KEY_RE.sub(" ", value.casefold()).strip(" .,:;")
|
||||
|
||||
|
||||
def invalid_responsible_reason(
|
||||
value: str,
|
||||
invalid_values: dict[str, str],
|
||||
) -> str:
|
||||
key = responsible_key(value)
|
||||
if key in invalid_values:
|
||||
return f"known_{invalid_values[key]}_not_responsible_entity"
|
||||
if key in GENERIC_RESPONSIBLE_WORDS:
|
||||
return "generic_process_word"
|
||||
if key in DATE_WORDS or key in WEEKDAY_WORDS or key in RELATIVE_DATE_PHRASES:
|
||||
return "date_or_relative_date"
|
||||
if NUMERIC_DATE_RE.search(value) or MONTH_DATE_RE.search(value):
|
||||
return "date_or_date_range"
|
||||
if TIME_RE.search(value):
|
||||
return "clock_time"
|
||||
if DATE_FRAGMENT_RE.search(value):
|
||||
return "date_fragment"
|
||||
return "unknown_responsible_entity"
|
||||
|
||||
|
||||
def split_legacy_string(value: str) -> list[str]:
|
||||
return [part.strip() for part in value.split("|")]
|
||||
|
||||
@@ -324,6 +497,7 @@ def canonicalize_value(
|
||||
source_file: str,
|
||||
source_index: int,
|
||||
counts: Counter[str],
|
||||
responsible_normalizer: ResponsiblePartyNormalizer | None = None,
|
||||
) -> dict[str, Any]:
|
||||
parsed = PARSERS[category](value)
|
||||
item: dict[str, Any] = {
|
||||
@@ -337,6 +511,14 @@ def canonicalize_value(
|
||||
}
|
||||
for key, parsed_value in parsed.items():
|
||||
item[key] = parsed_value
|
||||
if category == "action_item" and responsible_normalizer is not None:
|
||||
normalized, validation = responsible_normalizer.normalize(
|
||||
item.get("responsible"),
|
||||
item["item_id"],
|
||||
)
|
||||
item["responsible"] = normalized
|
||||
if validation is not None:
|
||||
item["responsibility_validation"] = validation
|
||||
item["source_references"] = [
|
||||
source_reference(source_file, source_index, value, item["evidence"])
|
||||
]
|
||||
@@ -367,6 +549,7 @@ def merge_exact_duplicates(items: list[dict[str, Any]]) -> tuple[list[dict[str,
|
||||
def canonicalize_extractions(
|
||||
input_dir: Path,
|
||||
merge_duplicates: bool = True,
|
||||
meeting_context_path: Path | None = None,
|
||||
) -> dict[str, Any]:
|
||||
files = find_extraction_files(input_dir)
|
||||
if not files:
|
||||
@@ -375,9 +558,22 @@ def canonicalize_extractions(
|
||||
counts: Counter[str] = Counter()
|
||||
input_counts: Counter[str] = Counter()
|
||||
items: list[dict[str, Any]] = []
|
||||
responsible_normalizer: ResponsiblePartyNormalizer | None = None
|
||||
responsibility_validations: list[dict[str, Any]] = []
|
||||
|
||||
loaded_files: list[tuple[Path, dict[str, Any]]] = []
|
||||
for path in files:
|
||||
data = load_json_object(path)
|
||||
loaded_files.append((path, data))
|
||||
|
||||
context_path = meeting_context_path or infer_meeting_context_path(loaded_files)
|
||||
if context_path is not None:
|
||||
meeting_context = load_meeting_context(context_path)
|
||||
responsible_normalizer = ResponsiblePartyNormalizer.from_meeting_context_data(
|
||||
meeting_context.data
|
||||
)
|
||||
|
||||
for path, data in loaded_files:
|
||||
validate_required_categories(data, path)
|
||||
for raw_category in REQUIRED_CATEGORIES:
|
||||
category = CATEGORY_NAMES[raw_category]
|
||||
@@ -391,8 +587,16 @@ def canonicalize_extractions(
|
||||
source_file=path.name,
|
||||
source_index=source_index,
|
||||
counts=counts,
|
||||
responsible_normalizer=responsible_normalizer,
|
||||
)
|
||||
)
|
||||
if (
|
||||
items[-1].get("category") == "action_item"
|
||||
and "responsibility_validation" in items[-1]
|
||||
):
|
||||
responsibility_validations.append(
|
||||
items[-1]["responsibility_validation"]
|
||||
)
|
||||
|
||||
exact_duplicates = 0
|
||||
if merge_duplicates:
|
||||
@@ -415,11 +619,40 @@ def canonicalize_extractions(
|
||||
"input_item_count_by_category": input_counts_by_category,
|
||||
"output_item_count_by_category": output_counts_by_category,
|
||||
"exact_duplicates_merged": exact_duplicates,
|
||||
"responsibility_validation_count": len(responsibility_validations),
|
||||
"responsibility_rejection_count": sum(
|
||||
1
|
||||
for validation in responsibility_validations
|
||||
if validation.get("status") == "rejected"
|
||||
),
|
||||
"responsibility_normalization_count": sum(
|
||||
1
|
||||
for validation in responsibility_validations
|
||||
if validation.get("status") == "accepted"
|
||||
and validation.get("original_value") != validation.get("normalized_value")
|
||||
),
|
||||
},
|
||||
"responsibility_validations": responsibility_validations,
|
||||
"items": items,
|
||||
}
|
||||
|
||||
|
||||
def infer_meeting_context_path(
|
||||
loaded_files: list[tuple[Path, dict[str, Any]]],
|
||||
) -> Path | None:
|
||||
paths: set[str] = set()
|
||||
for _path, data in loaded_files:
|
||||
context = data.get("context")
|
||||
if not isinstance(context, dict):
|
||||
continue
|
||||
source_file = clean_text(context.get("source_file"))
|
||||
if source_file:
|
||||
paths.add(source_file)
|
||||
if len(paths) != 1:
|
||||
return None
|
||||
return Path(next(iter(paths)))
|
||||
|
||||
|
||||
def write_canonicalized(output: dict[str, Any], output_path: Path) -> Path:
|
||||
output_path.parent.mkdir(parents=True, exist_ok=True)
|
||||
output_path.write_text(
|
||||
@@ -435,6 +668,7 @@ def main() -> int:
|
||||
output = canonicalize_extractions(
|
||||
args.input_dir,
|
||||
merge_duplicates=not args.no_merge_exact_duplicates,
|
||||
meeting_context_path=args.meeting_context,
|
||||
)
|
||||
output_path = write_canonicalized(output, args.output)
|
||||
except (OSError, UnicodeError, ValueError) as exc:
|
||||
|
||||
@@ -4,6 +4,7 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import copy
|
||||
import json
|
||||
import sys
|
||||
import time
|
||||
@@ -25,8 +26,13 @@ except ModuleNotFoundError: # pragma: no cover - used by repository-root tests.
|
||||
DEFAULT_MODEL = "qwen3.5:9b"
|
||||
DEFAULT_ENDPOINT = "http://127.0.0.1:11434/api/generate"
|
||||
DEFAULT_NUM_CTX = 32768
|
||||
DEFAULT_NUM_PREDICT = 4096
|
||||
DEFAULT_MIN_NUM_PREDICT = 4096
|
||||
DEFAULT_NUM_PREDICT = DEFAULT_MIN_NUM_PREDICT
|
||||
DEFAULT_PROGRESS_INTERVAL = 30
|
||||
OUTPUT_CONTEXT_RESERVE_TOKENS = 1024
|
||||
OUTPUT_TOKEN_ESTIMATE_CHARS = 4
|
||||
OUTPUT_GROUP_OVERHEAD_CHARS = 320
|
||||
OUTPUT_SAFETY_MARGIN = 1.35
|
||||
PROMPT_NAME = "consolidate_facts.md"
|
||||
|
||||
|
||||
@@ -75,10 +81,11 @@ def parse_args() -> argparse.Namespace:
|
||||
parser.add_argument(
|
||||
"--num-predict",
|
||||
type=int,
|
||||
default=DEFAULT_NUM_PREDICT,
|
||||
default=None,
|
||||
help=(
|
||||
"Maximum generated tokens. The default is bounded for the expected "
|
||||
f"fact-group JSON while leaving truncation headroom (default: {DEFAULT_NUM_PREDICT})."
|
||||
"Maximum generated tokens. By default this is estimated from the "
|
||||
"fact payload size and bounded by the context window. Explicit "
|
||||
"values preserve the previous fixed-budget behavior."
|
||||
),
|
||||
)
|
||||
thinking = parser.add_mutually_exclusive_group()
|
||||
@@ -158,6 +165,49 @@ def build_consolidation_prompt(facts: list[dict[str, Any]]) -> str:
|
||||
return f"{task_prompt}\n\nFACT ITEMS:\n{payload}\n"
|
||||
|
||||
|
||||
def estimate_response_tokens(facts: list[dict[str, Any]]) -> int:
|
||||
"""
|
||||
Estimate the token budget needed for the model's grouping JSON.
|
||||
|
||||
Semantic Consolidator V0 asks the model to return one group per source fact
|
||||
unless it finds a conservative duplicate. The response therefore scales with
|
||||
the number and text size of fact items. The estimate intentionally includes
|
||||
per-group JSON overhead and a safety margin; strict validation still decides
|
||||
whether the actual response is usable.
|
||||
"""
|
||||
text_chars = 0
|
||||
for item in facts:
|
||||
text_chars += len(str(item.get("text", "")))
|
||||
text_chars += len(str(item.get("evidence", "")))
|
||||
|
||||
estimated_chars = int(
|
||||
(text_chars + len(facts) * OUTPUT_GROUP_OVERHEAD_CHARS)
|
||||
* OUTPUT_SAFETY_MARGIN
|
||||
)
|
||||
return max(
|
||||
DEFAULT_MIN_NUM_PREDICT,
|
||||
(estimated_chars + OUTPUT_TOKEN_ESTIMATE_CHARS - 1)
|
||||
// OUTPUT_TOKEN_ESTIMATE_CHARS,
|
||||
)
|
||||
|
||||
|
||||
def resolve_num_predict(
|
||||
requested_num_predict: int | None,
|
||||
facts: list[dict[str, Any]],
|
||||
prompt_token_estimate: int,
|
||||
num_ctx: int,
|
||||
) -> int:
|
||||
if requested_num_predict is not None:
|
||||
return requested_num_predict
|
||||
|
||||
estimated = estimate_response_tokens(facts)
|
||||
max_available = max(
|
||||
DEFAULT_MIN_NUM_PREDICT,
|
||||
num_ctx - prompt_token_estimate - OUTPUT_CONTEXT_RESERVE_TOKENS,
|
||||
)
|
||||
return min(estimated, max_available)
|
||||
|
||||
|
||||
def response_text_from_ollama_data(data: dict[str, Any]) -> str | None:
|
||||
text = data.get("response")
|
||||
if isinstance(text, str) and text.strip():
|
||||
@@ -370,6 +420,96 @@ def validate_group_shapes(groups: list[dict[str, Any]]) -> None:
|
||||
raise ConsolidationValidationError("Merged groups need at least two IDs.")
|
||||
|
||||
|
||||
def repair_model_group_coverage(
|
||||
model_output: dict[str, Any],
|
||||
facts: list[dict[str, Any]],
|
||||
) -> tuple[dict[str, Any], list[dict[str, Any]]]:
|
||||
"""
|
||||
Apply deterministic source-coverage repairs to model grouping JSON.
|
||||
|
||||
The repair is intentionally conservative. Repeated source IDs are removed
|
||||
after their first occurrence, empty groups created by that removal are
|
||||
dropped, and missing facts are restored as singleton groups using the
|
||||
original canonicalized fact text. No existing group text, merge reason or
|
||||
semantic merge is rewritten.
|
||||
"""
|
||||
groups = model_output.get("groups")
|
||||
if not isinstance(groups, list):
|
||||
return model_output, []
|
||||
|
||||
repaired = copy.deepcopy(model_output)
|
||||
repaired_groups = repaired["groups"]
|
||||
facts_by_id = {str(item.get("item_id")): item for item in facts}
|
||||
expected_ids = set(facts_by_id)
|
||||
seen: set[str] = set()
|
||||
changes: list[dict[str, Any]] = []
|
||||
|
||||
for group_index, group in enumerate(repaired_groups):
|
||||
if not isinstance(group, dict):
|
||||
continue
|
||||
source_ids = group.get("source_item_ids")
|
||||
if not isinstance(source_ids, list):
|
||||
continue
|
||||
kept_ids: list[str] = []
|
||||
for id_index, item_id in enumerate(source_ids):
|
||||
if not isinstance(item_id, str) or item_id not in expected_ids:
|
||||
kept_ids.append(item_id)
|
||||
continue
|
||||
if item_id in seen:
|
||||
changes.append(
|
||||
{
|
||||
"operation": "remove_duplicate_source_id",
|
||||
"id": item_id,
|
||||
"group_index": group_index,
|
||||
"id_index": id_index,
|
||||
}
|
||||
)
|
||||
continue
|
||||
seen.add(item_id)
|
||||
kept_ids.append(item_id)
|
||||
group["source_item_ids"] = kept_ids
|
||||
|
||||
non_empty_groups: list[dict[str, Any]] = []
|
||||
for group_index, group in enumerate(repaired_groups):
|
||||
if (
|
||||
isinstance(group, dict)
|
||||
and isinstance(group.get("source_item_ids"), list)
|
||||
and len(group["source_item_ids"]) == 0
|
||||
):
|
||||
changes.append(
|
||||
{
|
||||
"operation": "remove_empty_group",
|
||||
"group_index": group_index,
|
||||
"canonical_text": group.get("canonical_text"),
|
||||
}
|
||||
)
|
||||
continue
|
||||
non_empty_groups.append(group)
|
||||
repaired["groups"] = non_empty_groups
|
||||
|
||||
missing_ids = sorted(expected_ids - seen)
|
||||
for item_id in missing_ids:
|
||||
fact = facts_by_id[item_id]
|
||||
repaired["groups"].append(
|
||||
{
|
||||
"canonical_text": str(fact.get("text", "")).strip(),
|
||||
"source_item_ids": [item_id],
|
||||
"merge_reason": (
|
||||
"Deterministic coverage repair: source fact was missing "
|
||||
"from the model grouping and is preserved as a singleton."
|
||||
),
|
||||
}
|
||||
)
|
||||
changes.append(
|
||||
{
|
||||
"operation": "restore_missing_source_id_as_singleton",
|
||||
"id": item_id,
|
||||
}
|
||||
)
|
||||
|
||||
return repaired, changes
|
||||
|
||||
|
||||
def build_consolidated_fact_item(
|
||||
group: dict[str, Any],
|
||||
fact_by_id: dict[str, dict[str, Any]],
|
||||
@@ -481,9 +621,11 @@ def write_report(
|
||||
runtime: float,
|
||||
prompt_chars: int,
|
||||
prompt_token_estimate: int,
|
||||
num_predict: int,
|
||||
fact_count: int,
|
||||
groups: list[dict[str, Any]],
|
||||
output_path: Path,
|
||||
repair_changes: list[dict[str, Any]] | None = None,
|
||||
) -> None:
|
||||
merged = [group for group in groups if len(group["source_item_ids"]) > 1]
|
||||
singletons = [group for group in groups if len(group["source_item_ids"]) == 1]
|
||||
@@ -497,9 +639,11 @@ def write_report(
|
||||
f"- Fact item count: {fact_count}",
|
||||
f"- Prompt characters: {prompt_chars}",
|
||||
f"- Estimated prompt tokens: {prompt_token_estimate}",
|
||||
f"- num_predict: {num_predict}",
|
||||
f"- Merged fact groups: {len(merged)}",
|
||||
f"- Source facts involved in merges: {sum(len(group['source_item_ids']) for group in merged)}",
|
||||
f"- Singleton fact groups: {len(singletons)}",
|
||||
f"- Deterministic repair changes: {len(repair_changes or [])}",
|
||||
f"- Output path: `{output_path}`",
|
||||
"",
|
||||
"## Actual Merges",
|
||||
@@ -518,6 +662,10 @@ def write_report(
|
||||
"",
|
||||
]
|
||||
)
|
||||
if repair_changes:
|
||||
lines.extend(["", "## Deterministic Coverage Repairs", ""])
|
||||
for change in repair_changes:
|
||||
lines.append(f"- `{change['operation']}`: {json.dumps(change, ensure_ascii=False, sort_keys=True)}")
|
||||
path.write_text("\n".join(lines).rstrip() + "\n", encoding="utf-8")
|
||||
|
||||
|
||||
@@ -527,6 +675,7 @@ def main() -> int:
|
||||
raw_response_path = args.output_dir / "raw_model_response.txt"
|
||||
output_path = args.output_dir / "consolidated_extractions.json"
|
||||
report_path = args.output_dir / "report.md"
|
||||
repair_metadata_path = args.output_dir / "repair_metadata.json"
|
||||
|
||||
try:
|
||||
canonicalized = load_json_object(args.canonicalized_input)
|
||||
@@ -534,9 +683,16 @@ def main() -> int:
|
||||
prompt = build_consolidation_prompt(facts)
|
||||
prompt_chars = len(prompt)
|
||||
prompt_token_estimate = (prompt_chars + 3) // 4
|
||||
num_predict = resolve_num_predict(
|
||||
requested_num_predict=args.num_predict,
|
||||
facts=facts,
|
||||
prompt_token_estimate=prompt_token_estimate,
|
||||
num_ctx=args.num_ctx,
|
||||
)
|
||||
print(f"Fact item count: {len(facts)}")
|
||||
print(f"Estimated prompt size chars: {prompt_chars}")
|
||||
print(f"Estimated prompt tokens: {prompt_token_estimate}")
|
||||
print(f"Resolved num_predict: {num_predict}")
|
||||
print("Expected LLM call count: 1")
|
||||
print("Expected runtime: 5-10 minutes on current local benchmark basis")
|
||||
|
||||
@@ -546,12 +702,23 @@ def main() -> int:
|
||||
prompt=prompt,
|
||||
timeout=args.timeout,
|
||||
num_ctx=args.num_ctx,
|
||||
num_predict=args.num_predict,
|
||||
num_predict=num_predict,
|
||||
think=args.think,
|
||||
progress_interval=args.progress_interval,
|
||||
)
|
||||
raw_response_path.write_text(raw_text + "\n", encoding="utf-8")
|
||||
model_output = parse_model_json(raw_text)
|
||||
model_output, repair_changes = repair_model_group_coverage(model_output, facts)
|
||||
if repair_changes:
|
||||
write_json(
|
||||
repair_metadata_path,
|
||||
{
|
||||
"scope": "semantic_consolidator_v0_source_coverage",
|
||||
"llm_used": False,
|
||||
"repair_count": len(repair_changes),
|
||||
"repairs": repair_changes,
|
||||
},
|
||||
)
|
||||
expected_fact_ids = {item["item_id"] for item in facts}
|
||||
groups = validate_model_groups(model_output, expected_fact_ids)
|
||||
validate_group_shapes(groups)
|
||||
@@ -564,9 +731,11 @@ def main() -> int:
|
||||
runtime=runtime,
|
||||
prompt_chars=prompt_chars,
|
||||
prompt_token_estimate=prompt_token_estimate,
|
||||
num_predict=num_predict,
|
||||
fact_count=len(facts),
|
||||
groups=groups,
|
||||
output_path=output_path,
|
||||
repair_changes=repair_changes,
|
||||
)
|
||||
except requests.ConnectionError as exc:
|
||||
print(f"Error: Ollama is not reachable at {args.endpoint}: {exc}", file=sys.stderr)
|
||||
|
||||
Reference in New Issue
Block a user