Add ERP explorer

This commit is contained in:
2026-07-29 14:03:06 +02:00
parent 720d789914
commit d92dc61277
6 changed files with 466 additions and 0 deletions
@@ -0,0 +1,277 @@
"""Read-only ERP CSV explorer for data profiling."""
from __future__ import annotations
import csv
from collections import Counter
from dataclasses import dataclass
from pathlib import Path
from article_data_manager.importers.rollcalc_json import load_rollcalc_articles
DEFAULT_ARTICLE_NUMBER_COLUMN = "SL_ITEM_NO"
DEFAULT_REPORT_PATH = Path("data/reports/erp_profile_report.txt")
@dataclass(frozen=True, slots=True)
class ColumnProfile:
"""Profile information for one CSV column."""
name: str
filled_count: int
empty_count: int
distinct_count: int
inferred_type: str
@dataclass(frozen=True, slots=True)
class DuplicateArticleNumber:
"""Repeated ERP article number and its exact occurrence count."""
nr: str
count: int
@dataclass(frozen=True, slots=True)
class ArticleNumberProfile:
"""Profile information for an ERP article number column."""
column_name: str
distinct_count: int
duplicate_count: int
duplicates: tuple[DuplicateArticleNumber, ...]
@dataclass(frozen=True, slots=True)
class RollCalcComparison:
"""Simple ERP/RollCalc article number coverage comparison."""
rollcalc_article_count: int
found_count: int
missing_count: int
missing_article_numbers: tuple[str, ...]
@dataclass(frozen=True, slots=True)
class ErpProfile:
"""Complete read-only profile of one ERP CSV export."""
file_name: str
row_count: int
column_count: int
columns: tuple[ColumnProfile, ...]
article_numbers: ArticleNumberProfile
rollcalc_comparison: RollCalcComparison | None = None
def profile_erp_csv(
csv_path: Path,
*,
article_number_column: str = DEFAULT_ARTICLE_NUMBER_COLUMN,
rollcalc_path: Path | None = None,
) -> ErpProfile:
"""Profile an ERP CSV export without interpreting or changing its source data."""
source_path = Path(csv_path)
rows, fieldnames = _read_csv(source_path)
column_profiles = tuple(
_profile_column(name, [row.get(name, "") for row in rows]) for name in fieldnames
)
article_number_profile = _profile_article_numbers(rows, article_number_column)
erp_article_numbers = {row.get(article_number_column, "") for row in rows}
comparison = (
_compare_rollcalc_articles(rollcalc_path, erp_article_numbers)
if rollcalc_path is not None
else None
)
return ErpProfile(
file_name=source_path.name,
row_count=len(rows),
column_count=len(fieldnames),
columns=column_profiles,
article_numbers=article_number_profile,
rollcalc_comparison=comparison,
)
def write_erp_profile_report(
csv_path: Path,
*,
report_path: Path = DEFAULT_REPORT_PATH,
article_number_column: str = DEFAULT_ARTICLE_NUMBER_COLUMN,
rollcalc_path: Path | None = None,
) -> ErpProfile:
"""Create a text profile report under the requested report path.
The ERP CSV and optional RollCalc JSON are read only. The only write performed
by this function is the requested text report.
"""
profile = profile_erp_csv(
csv_path,
article_number_column=article_number_column,
rollcalc_path=rollcalc_path,
)
destination = Path(report_path)
destination.parent.mkdir(parents=True, exist_ok=True)
destination.write_text(render_erp_profile_report(profile), encoding="utf-8")
return profile
def render_erp_profile_report(profile: ErpProfile) -> str:
"""Render an ERP profile as deterministic plain text."""
lines = [
"ERP Profile Report",
f"File: {profile.file_name}",
f"Rows: {profile.row_count}",
f"Columns: {profile.column_count}",
"",
"Column profiles",
]
for column in profile.columns:
lines.extend(
[
f"Column: {column.name}",
f"Filled: {column.filled_count}",
f"Empty: {column.empty_count}",
f"Distinct: {column.distinct_count}",
f"Type: {column.inferred_type}",
"",
]
)
lines.extend(
[
"Article numbers",
f"Column: {profile.article_numbers.column_name}",
f"Distinct: {profile.article_numbers.distinct_count}",
f"Duplicates: {profile.article_numbers.duplicate_count}",
"",
"Duplicate article numbers",
]
)
for duplicate in profile.article_numbers.duplicates:
lines.append(f"{duplicate.nr}\t{duplicate.count}")
if profile.rollcalc_comparison is not None:
comparison = profile.rollcalc_comparison
lines.extend(
[
"",
"RollCalc comparison",
f"RollCalc articles: {comparison.rollcalc_article_count}",
f"Found: {comparison.found_count}",
f"Missing: {comparison.missing_count}",
"Missing article numbers",
]
)
lines.extend(comparison.missing_article_numbers)
return "\n".join(lines) + "\n"
def _read_csv(path: Path) -> tuple[list[dict[str, str]], list[str]]:
with path.open("r", encoding="utf-8-sig", newline="") as csv_file:
sample = csv_file.read(4096)
csv_file.seek(0)
dialect = csv.Sniffer().sniff(sample, delimiters=",;\t")
reader = csv.DictReader(csv_file, dialect=dialect)
fieldnames = list(reader.fieldnames or [])
return list(reader), fieldnames
def _profile_column(name: str, values: list[str]) -> ColumnProfile:
filled_values = [value for value in values if not _is_empty(value)]
empty_count = len(values) - len(filled_values)
return ColumnProfile(
name=name,
filled_count=len(filled_values),
empty_count=empty_count,
distinct_count=len(set(filled_values)),
inferred_type=infer_value_type(filled_values),
)
def infer_value_type(values: list[str]) -> str:
"""Infer a simple data type for already-filled CSV values."""
if not values:
return "empty"
value_types = {_infer_single_value_type(value) for value in values}
if value_types == {"integer"}:
return "integer"
if value_types <= {"integer", "decimal"}:
return "decimal"
if value_types == {"text"}:
return "text"
return "mixed"
def _infer_single_value_type(value: str) -> str:
normalized = value.strip()
if _is_integer(normalized):
return "integer"
if _is_decimal(normalized):
return "decimal"
return "text"
def _is_integer(value: str) -> bool:
if value.startswith(("+", "-")):
value = value[1:]
return value.isdecimal()
def _is_decimal(value: str) -> bool:
if value.count(",") + value.count(".") != 1:
return False
separator = "," if "," in value else "."
left, right = value.split(separator, 1)
if left.startswith(("+", "-")):
left = left[1:]
return left.isdecimal() and right.isdecimal()
def _profile_article_numbers(
rows: list[dict[str, str]],
article_number_column: str,
) -> ArticleNumberProfile:
values = []
for row in rows:
value = row.get(article_number_column, "")
if not _is_empty(value):
values.append(value)
counts = Counter(values)
duplicates = tuple(
DuplicateArticleNumber(nr=nr, count=count)
for nr, count in sorted(counts.items())
if count > 1
)
return ArticleNumberProfile(
column_name=article_number_column,
distinct_count=len(counts),
duplicate_count=len(duplicates),
duplicates=duplicates,
)
def _compare_rollcalc_articles(
rollcalc_path: Path,
erp_article_numbers: set[str],
) -> RollCalcComparison:
rollcalc_articles = load_rollcalc_articles(Path(rollcalc_path))
rollcalc_numbers = tuple(article.nr for article in rollcalc_articles)
missing_article_numbers = tuple(nr for nr in rollcalc_numbers if nr not in erp_article_numbers)
return RollCalcComparison(
rollcalc_article_count=len(rollcalc_numbers),
found_count=len(rollcalc_numbers) - len(missing_article_numbers),
missing_count=len(missing_article_numbers),
missing_article_numbers=missing_article_numbers,
)
def _is_empty(value: str | None) -> bool:
return value is None or value == ""