diff --git a/CHANGELOG.md b/CHANGELOG.md index 4bf65e4..99865aa 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -4,6 +4,8 @@ ### Added +- Versioned UTF-8 YAML import/export for the SQLite terminology glossary, with + stable entry IDs, complete validation, and atomic replacement semantics. - First Streamlit MVP for audio upload, Meeting Context entry, participant management, processing progress and protocol editing. - Application service and Meeting Lab adapter with environment-based runtime diff --git a/PROJECT_KNOWLEDGE.md b/PROJECT_KNOWLEDGE.md index 63a7d7f..c7bcc1f 100644 --- a/PROJECT_KNOWLEDGE.md +++ b/PROJECT_KNOWLEDGE.md @@ -57,6 +57,8 @@ Meeting Assistant owns user interaction and product workflow: - a global SQLite terminology glossary whose active canonical core terms and recognition aliases are rendered through Meeting Context into direct protocol prompts +- versioned UTF-8 YAML glossary interchange (`version: 1`) that preserves entry + IDs and atomically replaces SQLite state only after complete validation - export and presentation of protocol versions Processing logic must not be duplicated in the application. diff --git a/README.md b/README.md index 1a02aa5..7f32915 100644 --- a/README.md +++ b/README.md @@ -143,6 +143,27 @@ The database defaults to `data/database/glossary.sqlite3`. It is created and bootstrapped automatically and can be moved with `MKA_GLOSSARY_DATABASE`. Glossary integration does not rewrite raw or diarized transcript artifacts. +Use **Export glossary** to download the complete SQLite-backed glossary as +UTF-8 YAML, and **Import glossary** to replace it from a previously exported +file. Imports are fully validated before a single SQLite transaction replaces +the current glossary, so malformed or conflicting files leave existing data +unchanged. Schema version 1 is: + +```yaml +version: 1 +glossary: + - id: 1 + canonical_term: Secugrid HS + aliases: + - Sikirgut + category: product + description: Canonical product spelling + active: true +``` + +`id` is the stable SQLite glossary-entry identifier. Categories are `product`, +`material`, `organization`, `technical_term`, `acronym`, or `other`. + ## Input configuration import and export Use **Export inputs** to save the current Meeting Assistant run-input form as a diff --git a/src/mka/application/glossary.py b/src/mka/application/glossary.py index efd5f6b..03fb398 100644 --- a/src/mka/application/glossary.py +++ b/src/mka/application/glossary.py @@ -35,6 +35,18 @@ class GlossaryEntry: updated_at: str +@dataclass(frozen=True) +class GlossaryReplacementEntry: + """Validated values used to atomically replace the persisted glossary.""" + + id: int + canonical_term: str + category: str + description: str | None + is_active: bool + aliases: tuple[str, ...] + + class GlossaryRepository: """Small data-access boundary for the local glossary database.""" @@ -194,6 +206,64 @@ class GlossaryRepository: if cursor.rowcount == 0: raise KeyError(f"Unknown glossary entry: {entry_id}") + def replace_all(self, entries: Iterable[GlossaryReplacementEntry]) -> None: + """Atomically replace every glossary entry, preserving supplied stable IDs.""" + replacements = tuple(entries) + validated: list[GlossaryReplacementEntry] = [] + seen_ids: set[int] = set() + seen_terms: set[str] = set() + for entry in replacements: + if type(entry.id) is not int or entry.id <= 0: + raise ValueError("Glossary entry IDs must be positive integers.") + if entry.id in seen_ids: + raise GlossaryConflictError(f"Duplicate glossary entry ID: {entry.id}.") + canonical, aliases = self._validate_values( + entry.canonical_term, entry.category, entry.aliases + ) + folded_terms = {canonical.casefold(), *(alias.casefold() for alias in aliases)} + if seen_terms.intersection(folded_terms): + raise GlossaryConflictError("Canonical term or alias already exists.") + seen_ids.add(entry.id) + seen_terms.update(folded_terms) + validated.append( + GlossaryReplacementEntry( + id=entry.id, + canonical_term=canonical, + category=entry.category, + description=_optional_text(entry.description), + is_active=entry.is_active, + aliases=aliases, + ) + ) + + now = _timestamp() + try: + with self._connect() as connection: + connection.execute("DELETE FROM glossary_entries") + connection.executemany( + """INSERT INTO glossary_entries + (id, canonical_term, category, description, is_active, created_at, updated_at) + VALUES (?, ?, ?, ?, ?, ?, ?)""", + ( + ( + entry.id, + entry.canonical_term, + entry.category, + entry.description, + entry.is_active, + now, + now, + ) + for entry in validated + ), + ) + connection.executemany( + "INSERT INTO glossary_aliases (entry_id, alias, created_at) VALUES (?, ?, ?)", + ((entry.id, alias, now) for entry in validated for alias in entry.aliases), + ) + except sqlite3.IntegrityError as exc: + raise GlossaryConflictError("Canonical term or alias already exists.") from exc + @contextmanager def _connect(self) -> Iterator[sqlite3.Connection]: connection = sqlite3.connect(self.database_path, timeout=5) diff --git a/src/mka/application/glossary_yaml.py b/src/mka/application/glossary_yaml.py new file mode 100644 index 0000000..c12f3c6 --- /dev/null +++ b/src/mka/application/glossary_yaml.py @@ -0,0 +1,122 @@ +"""Versioned UTF-8 YAML import and export for the terminology glossary.""" + +from __future__ import annotations + +from collections.abc import Sequence + +import yaml + +from mka.application.glossary import ( + GLOSSARY_CATEGORIES, + GlossaryEntry, + GlossaryReplacementEntry, +) + +GLOSSARY_YAML_VERSION = 1 + + +class GlossaryYamlError(ValueError): + """Raised when a glossary YAML document is malformed or invalid.""" + + +def export_glossary_yaml(entries: Sequence[GlossaryEntry]) -> str: + """Serialize all glossary state in a deterministic, versioned document.""" + document = { + "version": GLOSSARY_YAML_VERSION, + "glossary": [ + { + "id": entry.id, + "canonical_term": entry.canonical_term, + "aliases": list(entry.aliases), + "category": entry.category, + "description": entry.description, + "active": entry.is_active, + } + for entry in entries + ], + } + return yaml.safe_dump(document, allow_unicode=True, sort_keys=False, default_flow_style=False) + + +def import_glossary_yaml(content: str | bytes) -> list[GlossaryReplacementEntry]: + """Parse and completely validate a glossary replacement document.""" + try: + if isinstance(content, bytes): + content = content.decode("utf-8") + document = yaml.safe_load(content) + except (yaml.YAMLError, UnicodeDecodeError) as exc: + raise GlossaryYamlError(f"Malformed glossary YAML: {exc}") from exc + + if not isinstance(document, dict): + raise GlossaryYamlError("Glossary YAML must contain a top-level mapping.") + version = document.get("version") + if type(version) is not int or version != GLOSSARY_YAML_VERSION: + raise GlossaryYamlError( + f"Unsupported glossary YAML version {version!r}; expected {GLOSSARY_YAML_VERSION}." + ) + raw_entries = document.get("glossary") + if not isinstance(raw_entries, list): + raise GlossaryYamlError("Glossary YAML must contain a top-level 'glossary' list.") + + result: list[GlossaryReplacementEntry] = [] + seen_ids: set[int] = set() + seen_terms: set[str] = set() + for index, raw in enumerate(raw_entries, start=1): + if not isinstance(raw, dict): + raise GlossaryYamlError(f"Glossary entry {index} must be a mapping.") + entry_id = raw.get("id") + if type(entry_id) is not int or entry_id <= 0: + raise GlossaryYamlError(f"Glossary entry {index} requires a positive integer id.") + if entry_id in seen_ids: + raise GlossaryYamlError(f"Duplicate glossary entry id: {entry_id}.") + seen_ids.add(entry_id) + + canonical = _required_text(raw, "canonical_term", index).strip() + category = raw.get("category") + if category not in GLOSSARY_CATEGORIES: + allowed = ", ".join(GLOSSARY_CATEGORIES) + raise GlossaryYamlError( + f"Glossary entry {index} has invalid category; expected one of: {allowed}." + ) + aliases_value = raw.get("aliases") + if not isinstance(aliases_value, list) or any( + not isinstance(alias, str) or not alias.strip() for alias in aliases_value + ): + raise GlossaryYamlError(f"Glossary entry {index} aliases must be a list of text.") + aliases = tuple(alias.strip() for alias in aliases_value) + folded = [canonical.casefold(), *(alias.casefold() for alias in aliases)] + if len(folded) != len(set(folded)): + raise GlossaryYamlError( + f"Glossary entry {index} aliases must be unique and differ from its canonical term." + ) + duplicate = next((term for term in folded if term in seen_terms), None) + if duplicate is not None: + raise GlossaryYamlError( + f"Glossary entry {index} contains a duplicate canonical term or alias." + ) + seen_terms.update(folded) + + description = raw.get("description") + if description is not None and not isinstance(description, str): + raise GlossaryYamlError(f"Glossary entry {index} description must be text or null.") + active = raw.get("active") + if type(active) is not bool: + raise GlossaryYamlError(f"Glossary entry {index} active must be true or false.") + result.append( + GlossaryReplacementEntry( + id=entry_id, + canonical_term=canonical, + aliases=aliases, + category=category, + description=description, + is_active=active, + ) + ) + return result + + +def _required_text(entry: dict[object, object], field: str, index: int) -> str: + value = entry.get(field) + if not isinstance(value, str) or not value.strip(): + raise GlossaryYamlError(f"Glossary entry {index} requires a non-empty {field}.") + return value diff --git a/src/mka/ui/streamlit_app.py b/src/mka/ui/streamlit_app.py index 5e4724f..4fadadb 100644 --- a/src/mka/ui/streamlit_app.py +++ b/src/mka/ui/streamlit_app.py @@ -14,6 +14,11 @@ from mka.application.glossary import ( GlossaryConflictError, GlossaryRepository, ) +from mka.application.glossary_yaml import ( + GlossaryYamlError, + export_glossary_yaml, + import_glossary_yaml, +) from mka.application.meeting_service import ( STAGES, AppProgressEvent, @@ -63,6 +68,33 @@ def _render_glossary(repository: GlossaryRepository) -> None: "Store canonical core terms and recognition aliases. Compound phrases are " "composed from meeting context during protocol generation." ) + import_file = st.file_uploader( + "Import glossary", + type=["yaml", "yml"], + key="glossary_import_file", + help="Validate and atomically replace the SQLite glossary from versioned YAML.", + ) + action_columns = st.columns(2) + if action_columns[0].button("Import glossary", disabled=import_file is None): + try: + imported = import_glossary_yaml(import_file.getvalue()) + repository.replace_all(imported) + except (GlossaryYamlError, GlossaryConflictError, ValueError) as exc: + st.error(str(exc)) + else: + st.session_state.glossary_import_message = ( + f"Imported {len(imported)} glossary entries." + ) + st.rerun() + action_columns[1].download_button( + "Export glossary", + data=export_glossary_yaml(repository.list()).encode("utf-8"), + file_name="terminology-glossary.yaml", + mime="application/yaml", + ) + if message := st.session_state.pop("glossary_import_message", None): + st.success(message) + with st.form("glossary_add"): columns = st.columns(2) canonical = columns[0].text_input("Canonical term") diff --git a/tests/test_glossary_yaml.py b/tests/test_glossary_yaml.py new file mode 100644 index 0000000..1499614 --- /dev/null +++ b/tests/test_glossary_yaml.py @@ -0,0 +1,147 @@ +from pathlib import Path + +import pytest + +from mka.application.glossary import GlossaryRepository +from mka.application.glossary_yaml import ( + GlossaryYamlError, + export_glossary_yaml, + import_glossary_yaml, +) + + +def repository(tmp_path: Path) -> GlossaryRepository: + result = GlossaryRepository(tmp_path / "glossary.sqlite3") + result.initialize() + return result + + +def test_round_trip_preserves_ids_aliases_and_active_state(tmp_path: Path) -> None: + source = repository(tmp_path / "source") + active = source.create( + "Größenmaß", + "technical_term", + aliases=("Groessenmass", "Größen-Maß"), + description="Canonical UTF-8 spelling", + ) + inactive = source.create("Legacy Product", "product", is_active=False) + + encoded = export_glossary_yaml(source.list()).encode("utf-8") + target = repository(tmp_path / "target") + target.create("Will be replaced", "other") + target.replace_all(import_glossary_yaml(encoded)) + + entries = target.list() + assert [entry.id for entry in entries] == [active.id, inactive.id] + assert entries[0].canonical_term == "Größenmaß" + assert entries[0].aliases == ("Groessenmass", "Größen-Maß") + assert entries[0].description == "Canonical UTF-8 spelling" + assert entries[0].is_active is True + assert entries[1].is_active is False + + +def test_malformed_yaml_does_not_replace_existing_glossary(tmp_path: Path) -> None: + glossary = repository(tmp_path) + existing = glossary.create("PBAT", "acronym") + + with pytest.raises(GlossaryYamlError, match="Malformed glossary YAML"): + import_glossary_yaml(b"version: 1\nglossary: [") + + assert glossary.list() == [existing] + + +@pytest.mark.parametrize( + "entries, message", + [ + ( + [ + { + "id": 1, + "canonical_term": "PBAT", + "aliases": [], + "category": "acronym", + "description": None, + "active": True, + }, + { + "id": 2, + "canonical_term": "pbat", + "aliases": [], + "category": "material", + "description": None, + "active": True, + }, + ], + "duplicate canonical term or alias", + ), + ( + [ + { + "id": 1, + "canonical_term": "Secugrid", + "aliases": ["Sikirgut", "SIKIRGUT"], + "category": "product", + "description": None, + "active": True, + }, + ], + "aliases must be unique", + ), + ( + [ + { + "id": 1, + "canonical_term": "Secugrid", + "aliases": ["PBAT"], + "category": "product", + "description": None, + "active": True, + }, + { + "id": 2, + "canonical_term": "PBAT", + "aliases": [], + "category": "acronym", + "description": None, + "active": False, + }, + ], + "duplicate canonical term or alias", + ), + ], +) +def test_duplicate_terms_and_aliases_are_rejected( + entries: list[dict[str, object]], message: str +) -> None: + import yaml + + content = yaml.safe_dump({"version": 1, "glossary": entries}) + + with pytest.raises(GlossaryYamlError, match=message): + import_glossary_yaml(content) + + +def test_validation_failure_leaves_database_unchanged(tmp_path: Path) -> None: + glossary = repository(tmp_path) + existing = glossary.create("Existing", "other", aliases=("Existing alias",)) + invalid = b"""version: 1 +glossary: + - id: 10 + canonical_term: Duplicate + aliases: [] + category: other + description: null + active: true + - id: 11 + canonical_term: duplicate + aliases: [] + category: other + description: null + active: false +""" + + with pytest.raises(GlossaryYamlError): + imported = import_glossary_yaml(invalid) + glossary.replace_all(imported) + + assert glossary.list() == [existing]