Files
meeting-assistant/src/mka/application/glossary.py
T

370 lines
15 KiB
Python

"""SQLite-backed global terminology glossary."""
from __future__ import annotations
import sqlite3
from collections.abc import Iterable, Iterator
from contextlib import contextmanager
from dataclasses import dataclass
from datetime import UTC, datetime
from pathlib import Path
GLOSSARY_CATEGORIES = (
"product",
"material",
"organization",
"technical_term",
"acronym",
"other",
)
class GlossaryConflictError(ValueError):
"""Raised when a canonical term or alias conflicts with existing terminology."""
@dataclass(frozen=True)
class GlossaryEntry:
id: int
canonical_term: str
category: str
description: str | None
is_active: bool
aliases: tuple[str, ...]
created_at: str
updated_at: str
@dataclass(frozen=True)
class GlossaryReplacementEntry:
"""Validated values used to atomically replace the persisted glossary."""
id: int
canonical_term: str
category: str
description: str | None
is_active: bool
aliases: tuple[str, ...]
class GlossaryRepository:
"""Small data-access boundary for the local glossary database."""
def __init__(self, database_path: Path) -> None:
self.database_path = Path(database_path)
def initialize(self) -> None:
"""Create the database and current schema when absent."""
self.database_path.parent.mkdir(parents=True, exist_ok=True)
with self._connect() as connection:
connection.executescript(
"""
CREATE TABLE IF NOT EXISTS glossary_entries (
id INTEGER PRIMARY KEY,
canonical_term TEXT NOT NULL COLLATE NOCASE UNIQUE,
category TEXT NOT NULL CHECK (category IN (
'product', 'material', 'organization',
'technical_term', 'acronym', 'other'
)),
description TEXT,
is_active INTEGER NOT NULL DEFAULT 1 CHECK (is_active IN (0, 1)),
created_at TEXT NOT NULL,
updated_at TEXT NOT NULL
);
CREATE TABLE IF NOT EXISTS glossary_aliases (
id INTEGER PRIMARY KEY,
entry_id INTEGER NOT NULL REFERENCES glossary_entries(id)
ON DELETE CASCADE,
alias TEXT NOT NULL COLLATE NOCASE UNIQUE,
created_at TEXT NOT NULL
);
CREATE INDEX IF NOT EXISTS idx_glossary_aliases_entry_id
ON glossary_aliases(entry_id);
PRAGMA user_version = 1;
"""
)
def create(
self,
canonical_term: str,
category: str,
*,
aliases: Iterable[str] = (),
description: str | None = None,
is_active: bool = True,
) -> GlossaryEntry:
canonical, normalized_aliases = self._validate_values(canonical_term, category, aliases)
now = _timestamp()
try:
with self._connect() as connection:
self._ensure_terms_available(connection, canonical, normalized_aliases)
cursor = connection.execute(
"""INSERT INTO glossary_entries
(canonical_term, category, description, is_active, created_at, updated_at)
VALUES (?, ?, ?, ?, ?, ?)""",
(canonical, category, _optional_text(description), is_active, now, now),
)
entry_id = int(cursor.lastrowid)
connection.executemany(
"INSERT INTO glossary_aliases (entry_id, alias, created_at) VALUES (?, ?, ?)",
((entry_id, alias, now) for alias in normalized_aliases),
)
except sqlite3.IntegrityError as exc:
raise GlossaryConflictError("Canonical term or alias already exists.") from exc
return self.get(entry_id)
def get(self, entry_id: int) -> GlossaryEntry:
with self._connect() as connection:
row = connection.execute(
"SELECT * FROM glossary_entries WHERE id = ?", (entry_id,)
).fetchone()
if row is None:
raise KeyError(f"Unknown glossary entry: {entry_id}")
return self._to_entry(connection, row)
def list(self, search: str = "", *, active_only: bool = False) -> list[GlossaryEntry]:
clauses: list[str] = []
parameters: list[object] = []
if active_only:
clauses.append("entry.is_active = 1")
if search.strip():
clauses.append(
"(entry.canonical_term LIKE ? COLLATE NOCASE "
"OR entry.category LIKE ? COLLATE NOCASE "
"OR entry.description LIKE ? COLLATE NOCASE "
"OR alias.alias LIKE ? COLLATE NOCASE)"
)
pattern = f"%{search.strip()}%"
parameters.extend([pattern] * 4)
where = f"WHERE {' AND '.join(clauses)}" if clauses else ""
with self._connect() as connection:
rows = connection.execute(
f"""SELECT DISTINCT entry.* FROM glossary_entries AS entry
LEFT JOIN glossary_aliases AS alias ON alias.entry_id = entry.id
{where} ORDER BY entry.canonical_term COLLATE NOCASE""", # noqa: S608
parameters,
).fetchall()
return [self._to_entry(connection, row) for row in rows]
def update(
self,
entry_id: int,
canonical_term: str,
category: str,
*,
aliases: Iterable[str] = (),
description: str | None = None,
is_active: bool = True,
) -> GlossaryEntry:
canonical, normalized_aliases = self._validate_values(canonical_term, category, aliases)
try:
with self._connect() as connection:
exists = connection.execute(
"SELECT 1 FROM glossary_entries WHERE id = ?", (entry_id,)
).fetchone()
if exists is None:
raise KeyError(f"Unknown glossary entry: {entry_id}")
self._ensure_terms_available(
connection, canonical, normalized_aliases, excluding_entry_id=entry_id
)
now = _timestamp()
connection.execute(
"""UPDATE glossary_entries SET canonical_term = ?, category = ?,
description = ?, is_active = ?, updated_at = ? WHERE id = ?""",
(
canonical,
category,
_optional_text(description),
is_active,
now,
entry_id,
),
)
connection.execute("DELETE FROM glossary_aliases WHERE entry_id = ?", (entry_id,))
connection.executemany(
"INSERT INTO glossary_aliases (entry_id, alias, created_at) VALUES (?, ?, ?)",
((entry_id, alias, now) for alias in normalized_aliases),
)
except sqlite3.IntegrityError as exc:
raise GlossaryConflictError("Canonical term or alias already exists.") from exc
return self.get(entry_id)
def set_active(self, entry_id: int, is_active: bool) -> GlossaryEntry:
entry = self.get(entry_id)
return self.update(
entry.id,
entry.canonical_term,
entry.category,
aliases=entry.aliases,
description=entry.description,
is_active=is_active,
)
def delete(self, entry_id: int) -> None:
with self._connect() as connection:
cursor = connection.execute("DELETE FROM glossary_entries WHERE id = ?", (entry_id,))
if cursor.rowcount == 0:
raise KeyError(f"Unknown glossary entry: {entry_id}")
def replace_all(self, entries: Iterable[GlossaryReplacementEntry]) -> None:
"""Atomically replace every glossary entry, preserving supplied stable IDs."""
replacements = tuple(entries)
validated: list[GlossaryReplacementEntry] = []
seen_ids: set[int] = set()
seen_terms: set[str] = set()
for entry in replacements:
if type(entry.id) is not int or entry.id <= 0:
raise ValueError("Glossary entry IDs must be positive integers.")
if entry.id in seen_ids:
raise GlossaryConflictError(f"Duplicate glossary entry ID: {entry.id}.")
canonical, aliases = self._validate_values(
entry.canonical_term, entry.category, entry.aliases
)
folded_terms = {canonical.casefold(), *(alias.casefold() for alias in aliases)}
if seen_terms.intersection(folded_terms):
raise GlossaryConflictError("Canonical term or alias already exists.")
seen_ids.add(entry.id)
seen_terms.update(folded_terms)
validated.append(
GlossaryReplacementEntry(
id=entry.id,
canonical_term=canonical,
category=entry.category,
description=_optional_text(entry.description),
is_active=entry.is_active,
aliases=aliases,
)
)
now = _timestamp()
try:
with self._connect() as connection:
connection.execute("DELETE FROM glossary_entries")
connection.executemany(
"""INSERT INTO glossary_entries
(id, canonical_term, category, description, is_active, created_at, updated_at)
VALUES (?, ?, ?, ?, ?, ?, ?)""",
(
(
entry.id,
entry.canonical_term,
entry.category,
entry.description,
entry.is_active,
now,
now,
)
for entry in validated
),
)
connection.executemany(
"INSERT INTO glossary_aliases (entry_id, alias, created_at) VALUES (?, ?, ?)",
((entry.id, alias, now) for entry in validated for alias in entry.aliases),
)
except sqlite3.IntegrityError as exc:
raise GlossaryConflictError("Canonical term or alias already exists.") from exc
@contextmanager
def _connect(self) -> Iterator[sqlite3.Connection]:
connection = sqlite3.connect(self.database_path, timeout=5)
try:
connection.row_factory = sqlite3.Row
connection.execute("PRAGMA foreign_keys = ON")
connection.execute("PRAGMA journal_mode = WAL")
connection.execute("PRAGMA busy_timeout = 5000")
with connection:
yield connection
finally:
connection.close()
@staticmethod
def _validate_values(
canonical_term: str, category: str, aliases: Iterable[str]
) -> tuple[str, tuple[str, ...]]:
canonical = canonical_term.strip()
if not canonical:
raise ValueError("Canonical term is required.")
if category not in GLOSSARY_CATEGORIES:
raise ValueError(f"Unsupported glossary category: {category}")
normalized_aliases = tuple(
dict.fromkeys(alias.strip() for alias in aliases if alias.strip())
)
folded = [alias.casefold() for alias in normalized_aliases]
if len(folded) != len(set(folded)) or canonical.casefold() in folded:
raise GlossaryConflictError(
"Aliases must be unique and differ from the canonical term."
)
return canonical, normalized_aliases
@staticmethod
def _ensure_terms_available(
connection: sqlite3.Connection,
canonical: str,
aliases: tuple[str, ...],
*,
excluding_entry_id: int | None = None,
) -> None:
terms = (canonical, *aliases)
placeholders = ", ".join("?" for _ in terms)
exclusion = "AND entry_id != ?" if excluding_entry_id is not None else ""
alias_parameters: list[object] = [*terms]
if excluding_entry_id is not None:
alias_parameters.append(excluding_entry_id)
alias_conflict = connection.execute(
f"SELECT 1 FROM glossary_aliases WHERE alias IN ({placeholders}) {exclusion} LIMIT 1", # noqa: S608
alias_parameters,
).fetchone()
entry_exclusion = "AND id != ?" if excluding_entry_id is not None else ""
entry_parameters: list[object] = [*terms]
if excluding_entry_id is not None:
entry_parameters.append(excluding_entry_id)
canonical_conflict = connection.execute(
f"SELECT 1 FROM glossary_entries WHERE canonical_term IN ({placeholders}) " # noqa: S608
f"{entry_exclusion} LIMIT 1",
entry_parameters,
).fetchone()
if alias_conflict or canonical_conflict:
raise GlossaryConflictError("Canonical term or alias already exists.")
@staticmethod
def _to_entry(connection: sqlite3.Connection, row: sqlite3.Row) -> GlossaryEntry:
aliases = connection.execute(
"SELECT alias FROM glossary_aliases WHERE entry_id = ? ORDER BY alias COLLATE NOCASE",
(row["id"],),
).fetchall()
return GlossaryEntry(
id=row["id"],
canonical_term=row["canonical_term"],
category=row["category"],
description=row["description"],
is_active=bool(row["is_active"]),
aliases=tuple(alias["alias"] for alias in aliases),
created_at=row["created_at"],
updated_at=row["updated_at"],
)
def render_glossary_terms(entries: Iterable[GlossaryEntry]) -> list[str]:
"""Render concise canonical terms with recognition aliases for Meeting Context."""
rendered = []
for entry in entries:
item = entry.canonical_term
if entry.aliases:
item += f" (aliases: {', '.join(entry.aliases)})"
rendered.append(item)
return rendered
def glossary_alias_mapping(entries: Iterable[GlossaryEntry]) -> dict[str, str]:
"""Return explicit alias configuration for provenance, not substitution."""
return {alias: entry.canonical_term for entry in entries for alias in entry.aliases}
def _timestamp() -> str:
return datetime.now(UTC).isoformat(timespec="seconds")
def _optional_text(value: str | None) -> str | None:
stripped = value.strip() if value else ""
return stripped or None