Add deterministic article search to MCP
This commit is contained in:
@@ -251,6 +251,56 @@ class ArticleRepository:
|
||||
|
||||
return {"status": "not_requested", "article": None}
|
||||
|
||||
def search(self, query: str, *, limit: int = 20) -> dict[str, Any]:
|
||||
"""Return deterministic article-discovery candidates without resolving one."""
|
||||
if not isinstance(query, str):
|
||||
return {
|
||||
"status": "invalid_query",
|
||||
"query": None,
|
||||
"total_matches": 0,
|
||||
"candidates": [],
|
||||
}
|
||||
if isinstance(limit, bool) or not isinstance(limit, int) or limit <= 0:
|
||||
raise ValueError("limit must be a positive integer")
|
||||
|
||||
query = query.strip()
|
||||
query_key = _canonical_name(query)
|
||||
if not query_key:
|
||||
return {
|
||||
"status": "search_not_found",
|
||||
"query": query,
|
||||
"total_matches": 0,
|
||||
"candidates": [],
|
||||
}
|
||||
|
||||
query_tokens = _token_counts(query_key)
|
||||
matches: list[tuple[bool, int, dict[str, Any]]] = []
|
||||
for article in self.articles:
|
||||
name_key = _canonical_name(str(article.get("name", "")))
|
||||
name_tokens = _token_counts(name_key)
|
||||
if not _tokens_contained(query_tokens, name_tokens):
|
||||
continue
|
||||
unmatched_tokens = _unmatched_token_count(name_tokens, query_tokens)
|
||||
matches.append((name_key == query_key, unmatched_tokens, article))
|
||||
|
||||
matches.sort(
|
||||
key=lambda item: (
|
||||
not item[0],
|
||||
item[1],
|
||||
str(item[2].get("nr", "")),
|
||||
str(item[2].get("name", "")),
|
||||
)
|
||||
)
|
||||
return {
|
||||
"status": "search_results" if matches else "search_not_found",
|
||||
"query": query,
|
||||
"total_matches": len(matches),
|
||||
"candidates": [
|
||||
_article_search_candidate(article)
|
||||
for _, _, article in matches[:limit]
|
||||
],
|
||||
}
|
||||
|
||||
def _matching_names(self, article_name_hint: str) -> list[dict[str, Any]]:
|
||||
hint_keys = _article_name_keys(article_name_hint)
|
||||
if not hint_keys:
|
||||
@@ -351,6 +401,31 @@ def _canonical_name(value: str) -> str:
|
||||
return " ".join(compacted)
|
||||
|
||||
|
||||
def _token_counts(value: str) -> dict[str, int]:
|
||||
counts: dict[str, int] = {}
|
||||
for token in value.split():
|
||||
counts[token] = counts.get(token, 0) + 1
|
||||
return counts
|
||||
|
||||
|
||||
def _tokens_contained(
|
||||
query_tokens: dict[str, int], candidate_tokens: dict[str, int]
|
||||
) -> bool:
|
||||
return all(
|
||||
candidate_tokens.get(token, 0) >= count
|
||||
for token, count in query_tokens.items()
|
||||
)
|
||||
|
||||
|
||||
def _unmatched_token_count(
|
||||
candidate_tokens: dict[str, int], query_tokens: dict[str, int]
|
||||
) -> int:
|
||||
return sum(
|
||||
max(0, count - query_tokens.get(token, 0))
|
||||
for token, count in candidate_tokens.items()
|
||||
)
|
||||
|
||||
|
||||
def _article_name_keys(value: str) -> set[str]:
|
||||
keys = {_canonical_name(value)}
|
||||
without_dimensions = re.sub(
|
||||
@@ -391,6 +466,22 @@ def _article_candidate(article: dict[str, Any]) -> dict[str, str]:
|
||||
}
|
||||
|
||||
|
||||
def _article_search_candidate(article: dict[str, Any]) -> dict[str, Any]:
|
||||
name = str(article.get("name", ""))
|
||||
number = str(article.get("nr", ""))
|
||||
return {
|
||||
"article_number": number,
|
||||
"name": " ".join(name.split()),
|
||||
"width_m": _article_width(name),
|
||||
"production_site": article_production_site(number),
|
||||
}
|
||||
|
||||
|
||||
def article_production_site(article_number: str) -> str | None:
|
||||
"""Return production-site metadata established by article-number rules."""
|
||||
return "Malaysia" if article_number.startswith("8") else None
|
||||
|
||||
|
||||
def _article_width(name: str) -> float | None:
|
||||
patterns = (
|
||||
r"(?<!\d)(\d{1,2}[,.]\d{1,3})\s*[x×]\s*<?\s*\d+(?:[,.]\d+)?\s*m\b",
|
||||
|
||||
Reference in New Issue
Block a user