From 0b24351127b6d8767a8f76c4947b2985ff2cf9ba Mon Sep 17 00:00:00 2001 From: Martin Date: Sun, 9 Aug 2026 14:59:27 +0200 Subject: [PATCH] Repair hallucinated Semantic Consolidator source IDs --- docs/architecture.md | 9 +- docs/regression-bugs.md | 89 +++++++++++- .../consolidation/consolidate_facts.py | 23 +++- tests/test_consolidate_facts.py | 130 ++++++++++++++++++ 4 files changed, 240 insertions(+), 11 deletions(-) diff --git a/docs/architecture.md b/docs/architecture.md index 2713a00..343d946 100644 --- a/docs/architecture.md +++ b/docs/architecture.md @@ -379,8 +379,13 @@ Views documented in `output-views.md`. Semantic Consolidator V0 preserves every raw model response before parsing. Its normal path accepts parseable grouping JSON and leaves duplicate source-ID -and missing-source-ID correction to the existing deterministic coverage -repair. +unknown-source-ID and missing-source-ID correction to the deterministic +coverage repair. Unknown source IDs are removed without attempting numeric or +semantic remapping. A group retains its valid IDs and remains present when at +least one valid ID survives. A group containing only unknown IDs is removed +after it becomes empty. Existing coverage repair then restores every genuinely +missing canonical fact as a singleton. Strict validation runs against the +repaired result and still rejects any unknown ID that survives this process. An invalid or truncated response is not retried by default. One controlled retry is allowed only when deterministic inspection finds at least three diff --git a/docs/regression-bugs.md b/docs/regression-bugs.md index 4a2c664..13c1af2 100644 --- a/docs/regression-bugs.md +++ b/docs/regression-bugs.md @@ -1038,6 +1038,89 @@ invalid only because group 89 contained unknown `fact_0114`. The stage therefore failed cleanly after the single retry and preserved both attempts. No third LLM call occurred. The remaining unknown-ID failure is not a -repetitive generation loop and is not silently repaired by the existing -coverage repair. This verification does not claim that general Semantic -Consolidator output quality is solved. +repetitive generation loop; its deterministic repair is tracked separately as +BUG-014. This verification does not claim that general Semantic Consolidator +output quality is solved. + +## BUG-014 + +ID: BUG-014 + +Title: Semantic Consolidator hallucinates unknown source item IDs + +Pipeline stage: Semantic Consolidator deterministic coverage repair + +Severity: High + +Status: Verified + +Date discovered: 2026-08-09 + +Version first observed: BUG-013 consolidator-only regression + +Description: + +The parseable BUG-013 retry response contained a semantic group at index 104 +with `source_item_ids` equal to `fact_0113` and unknown `fact_0114`. The +canonicalized input contains 113 facts ending at `fact_0113`; no `fact_0114` +exists. + +The group canonical text, "Ich habe ihn heute Morgen im Büro angetroffen.", is +directly supported by valid `fact_0113`. The unknown ID adds no identifiable +canonical source content and appears to be an additional hallucinated ID, not +a safely correctable typo or a reference that can be mapped to another input +fact. + +Invariant: + +Every `source_item_id` emitted by the Semantic Consolidator must refer to an +existing canonical input fact. Strict validation must continue to reject any +unknown ID that survives deterministic repair. + +Deterministic repair rule: + +- Remove every string source ID that is not present in the canonical fact-ID + set. Never map it to a numerically nearby or semantically guessed ID. +- Preserve all valid source IDs in the group. +- Preserve the group when at least one valid source ID remains. +- Remove the group when unknown-ID removal leaves it empty. +- Apply existing duplicate-ID removal to surviving valid IDs. +- Restore genuinely missing canonical facts as singletons through the existing + coverage repair. +- Preserve the original group text and merge reason; do not rewrite semantics. + +Related files: + +- `/tmp/meeting-lab-bug013-regression/raw_model_response_retry.txt` +- `samples/benchmarks/progeo_north_linux_20260809_124954/canonicalizer/canonicalized_extractions.json` +- `src/meeting_lab/consolidation/consolidate_facts.py` +- `tests/test_consolidate_facts.py` + +Regression test available (yes/no): yes + +Current status: + +Verified with a consolidator-only regression using the preserved canonicalized +input. Extraction, canonicalization, full benchmark orchestration and rendering +were not run. + +The first attempt reproduced BUG-013: `eval_count=18654`, +`done_reason=length`, invalid JSON and a detected 192-group consecutive +repetition run. This triggered one controlled retry. The retry returned 105 +parseable groups with 154 source-ID occurrences, `eval_count=7362`, +`done_reason=stop`, and no repetition loop. + +Strict source-coverage inspection before repair found one unknown ID +(`fact_0114` in group 104), duplicate IDs and four missing canonical IDs. +Deterministic repair made 64 changes: + +- 44 `remove_duplicate_source_id` +- 1 `remove_unknown_source_id` +- 15 `remove_empty_group` +- 4 `restore_missing_source_id_as_singleton` + +Strict validation after repair passed with 94 groups containing exactly 113 +source-ID occurrences and 113 unique canonical source IDs. There were no +missing IDs, unknown IDs or duplicate occurrences. BUG-014 is independent of +BUG-013: BUG-013 detects invalid JSON caused by runaway repeated groups, +whereas BUG-014 repairs unknown source IDs in parseable model grouping JSON. diff --git a/src/meeting_lab/consolidation/consolidate_facts.py b/src/meeting_lab/consolidation/consolidate_facts.py index 07c0412..ead2f33 100644 --- a/src/meeting_lab/consolidation/consolidate_facts.py +++ b/src/meeting_lab/consolidation/consolidate_facts.py @@ -600,11 +600,12 @@ def repair_model_group_coverage( """ Apply deterministic source-coverage repairs to model grouping JSON. - The repair is intentionally conservative. Repeated source IDs are removed - after their first occurrence, empty groups created by that removal are - dropped, and missing facts are restored as singleton groups using the - original canonicalized fact text. No existing group text, merge reason or - semantic merge is rewritten. + The repair is intentionally conservative. Unknown source IDs are removed + without guessing a replacement. Repeated valid source IDs are removed after + their first occurrence, empty groups created by either removal are dropped, + and missing facts are restored as singleton groups using the original + canonicalized fact text. No existing group text, merge reason or semantic + merge is rewritten. """ groups = model_output.get("groups") if not isinstance(groups, list): @@ -625,9 +626,19 @@ def repair_model_group_coverage( continue kept_ids: list[str] = [] for id_index, item_id in enumerate(source_ids): - if not isinstance(item_id, str) or item_id not in expected_ids: + if not isinstance(item_id, str): kept_ids.append(item_id) continue + if item_id not in expected_ids: + changes.append( + { + "operation": "remove_unknown_source_id", + "id": item_id, + "group_index": group_index, + "id_index": id_index, + } + ) + continue if item_id in seen: changes.append( { diff --git a/tests/test_consolidate_facts.py b/tests/test_consolidate_facts.py index afa3737..fb45686 100644 --- a/tests/test_consolidate_facts.py +++ b/tests/test_consolidate_facts.py @@ -415,6 +415,136 @@ class ConsolidateFactsTests(unittest.TestCase): ["remove_duplicate_source_id", "remove_empty_group"], ) + def test_source_coverage_repair_keeps_valid_ids_unchanged(self): + facts = fact_items(canonicalized_fixture()) + model_output = { + "groups": [self.group("Merged", ["fact_0001", "fact_0002"], "Same.")] + } + + repaired, changes = repair_model_group_coverage(model_output, facts) + + self.assertEqual(repaired, model_output) + self.assertEqual(changes, []) + validate_model_groups(repaired, {"fact_0001", "fact_0002"}) + + def test_source_coverage_repair_removes_unknown_id_but_keeps_valid_id(self): + facts = fact_items(canonicalized_fixture()) + repaired, changes = repair_model_group_coverage( + { + "groups": [ + self.group("Supported", ["fact_0001", "fact_9999"], "Partial.") + ] + }, + facts, + ) + + self.assertEqual(repaired["groups"][0]["source_item_ids"], ["fact_0001"]) + self.assertEqual(changes[0]["operation"], "remove_unknown_source_id") + self.assertEqual(changes[0]["id"], "fact_9999") + + def test_source_coverage_repair_removes_multiple_unknown_ids(self): + facts = fact_items(canonicalized_fixture()) + repaired, changes = repair_model_group_coverage( + { + "groups": [ + self.group( + "Supported", + ["unknown_a", "fact_0001", "unknown_b", "fact_0002"], + "Partial.", + ) + ] + }, + facts, + ) + + self.assertEqual( + repaired["groups"][0]["source_item_ids"], + ["fact_0001", "fact_0002"], + ) + self.assertEqual( + [change["id"] for change in changes], + ["unknown_a", "unknown_b"], + ) + + def test_source_coverage_repair_drops_group_with_only_unknown_ids(self): + facts = fact_items(canonicalized_fixture()) + repaired, changes = repair_model_group_coverage( + {"groups": [self.group("Unsupported", ["unknown_a"], "Unsupported.")]}, + facts, + ) + + self.assertNotIn("Unsupported", [group["canonical_text"] for group in repaired["groups"]]) + self.assertEqual( + [change["operation"] for change in changes[:2]], + ["remove_unknown_source_id", "remove_empty_group"], + ) + validate_model_groups(repaired, {"fact_0001", "fact_0002"}) + + def test_unknown_id_removal_that_empties_group_is_recorded(self): + facts = fact_items(canonicalized_fixture()) + _repaired, changes = repair_model_group_coverage( + { + "groups": [ + self.group("Unsupported", ["unknown_a", "unknown_b"], "Unsupported.") + ] + }, + facts, + ) + + self.assertEqual( + [change["operation"] for change in changes[:3]], + [ + "remove_unknown_source_id", + "remove_unknown_source_id", + "remove_empty_group", + ], + ) + + def test_unknown_id_removal_interacts_with_duplicate_repair(self): + facts = fact_items(canonicalized_fixture()) + repaired, changes = repair_model_group_coverage( + { + "groups": [ + self.group("First", ["fact_0001", "unknown_a"], "First."), + self.group("Second", ["fact_0001", "fact_0002"], "Second."), + ] + }, + facts, + ) + + self.assertEqual(repaired["groups"][0]["source_item_ids"], ["fact_0001"]) + self.assertEqual(repaired["groups"][1]["source_item_ids"], ["fact_0002"]) + self.assertEqual( + [change["operation"] for change in changes], + ["remove_unknown_source_id", "remove_duplicate_source_id"], + ) + validate_model_groups(repaired, {"fact_0001", "fact_0002"}) + + def test_unknown_removal_and_missing_id_singleton_restoration(self): + facts = fact_items(canonicalized_fixture()) + repaired, changes = repair_model_group_coverage( + { + "groups": [ + self.group("Supported", ["fact_0001", "unknown_a"], "Partial.") + ] + }, + facts, + ) + + self.assertEqual(repaired["groups"][-1]["source_item_ids"], ["fact_0002"]) + self.assertEqual( + [change["operation"] for change in changes], + ["remove_unknown_source_id", "restore_missing_source_id_as_singleton"], + ) + validate_model_groups(repaired, {"fact_0001", "fact_0002"}) + + def test_final_validator_still_rejects_unknown_source_id(self): + with self.assertRaisesRegex(ConsolidationValidationError, "unknown"): + validate_model_groups( + {"groups": [self.group("Unsupported", ["fact_9999"], "Bad ID.")]}, + {"fact_0001"}, + ) + def test_merged_group_validation(self): groups = validate_model_groups( {