Add ERP explorer
This commit is contained in:
@@ -0,0 +1,277 @@
|
||||
"""Read-only ERP CSV explorer for data profiling."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import csv
|
||||
from collections import Counter
|
||||
from dataclasses import dataclass
|
||||
from pathlib import Path
|
||||
|
||||
from article_data_manager.importers.rollcalc_json import load_rollcalc_articles
|
||||
|
||||
DEFAULT_ARTICLE_NUMBER_COLUMN = "SL_ITEM_NO"
|
||||
DEFAULT_REPORT_PATH = Path("data/reports/erp_profile_report.txt")
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class ColumnProfile:
|
||||
"""Profile information for one CSV column."""
|
||||
|
||||
name: str
|
||||
filled_count: int
|
||||
empty_count: int
|
||||
distinct_count: int
|
||||
inferred_type: str
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class DuplicateArticleNumber:
|
||||
"""Repeated ERP article number and its exact occurrence count."""
|
||||
|
||||
nr: str
|
||||
count: int
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class ArticleNumberProfile:
|
||||
"""Profile information for an ERP article number column."""
|
||||
|
||||
column_name: str
|
||||
distinct_count: int
|
||||
duplicate_count: int
|
||||
duplicates: tuple[DuplicateArticleNumber, ...]
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class RollCalcComparison:
|
||||
"""Simple ERP/RollCalc article number coverage comparison."""
|
||||
|
||||
rollcalc_article_count: int
|
||||
found_count: int
|
||||
missing_count: int
|
||||
missing_article_numbers: tuple[str, ...]
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class ErpProfile:
|
||||
"""Complete read-only profile of one ERP CSV export."""
|
||||
|
||||
file_name: str
|
||||
row_count: int
|
||||
column_count: int
|
||||
columns: tuple[ColumnProfile, ...]
|
||||
article_numbers: ArticleNumberProfile
|
||||
rollcalc_comparison: RollCalcComparison | None = None
|
||||
|
||||
|
||||
def profile_erp_csv(
|
||||
csv_path: Path,
|
||||
*,
|
||||
article_number_column: str = DEFAULT_ARTICLE_NUMBER_COLUMN,
|
||||
rollcalc_path: Path | None = None,
|
||||
) -> ErpProfile:
|
||||
"""Profile an ERP CSV export without interpreting or changing its source data."""
|
||||
|
||||
source_path = Path(csv_path)
|
||||
rows, fieldnames = _read_csv(source_path)
|
||||
column_profiles = tuple(
|
||||
_profile_column(name, [row.get(name, "") for row in rows]) for name in fieldnames
|
||||
)
|
||||
article_number_profile = _profile_article_numbers(rows, article_number_column)
|
||||
erp_article_numbers = {row.get(article_number_column, "") for row in rows}
|
||||
comparison = (
|
||||
_compare_rollcalc_articles(rollcalc_path, erp_article_numbers)
|
||||
if rollcalc_path is not None
|
||||
else None
|
||||
)
|
||||
return ErpProfile(
|
||||
file_name=source_path.name,
|
||||
row_count=len(rows),
|
||||
column_count=len(fieldnames),
|
||||
columns=column_profiles,
|
||||
article_numbers=article_number_profile,
|
||||
rollcalc_comparison=comparison,
|
||||
)
|
||||
|
||||
|
||||
def write_erp_profile_report(
|
||||
csv_path: Path,
|
||||
*,
|
||||
report_path: Path = DEFAULT_REPORT_PATH,
|
||||
article_number_column: str = DEFAULT_ARTICLE_NUMBER_COLUMN,
|
||||
rollcalc_path: Path | None = None,
|
||||
) -> ErpProfile:
|
||||
"""Create a text profile report under the requested report path.
|
||||
|
||||
The ERP CSV and optional RollCalc JSON are read only. The only write performed
|
||||
by this function is the requested text report.
|
||||
"""
|
||||
|
||||
profile = profile_erp_csv(
|
||||
csv_path,
|
||||
article_number_column=article_number_column,
|
||||
rollcalc_path=rollcalc_path,
|
||||
)
|
||||
destination = Path(report_path)
|
||||
destination.parent.mkdir(parents=True, exist_ok=True)
|
||||
destination.write_text(render_erp_profile_report(profile), encoding="utf-8")
|
||||
return profile
|
||||
|
||||
|
||||
def render_erp_profile_report(profile: ErpProfile) -> str:
|
||||
"""Render an ERP profile as deterministic plain text."""
|
||||
|
||||
lines = [
|
||||
"ERP Profile Report",
|
||||
f"File: {profile.file_name}",
|
||||
f"Rows: {profile.row_count}",
|
||||
f"Columns: {profile.column_count}",
|
||||
"",
|
||||
"Column profiles",
|
||||
]
|
||||
for column in profile.columns:
|
||||
lines.extend(
|
||||
[
|
||||
f"Column: {column.name}",
|
||||
f"Filled: {column.filled_count}",
|
||||
f"Empty: {column.empty_count}",
|
||||
f"Distinct: {column.distinct_count}",
|
||||
f"Type: {column.inferred_type}",
|
||||
"",
|
||||
]
|
||||
)
|
||||
|
||||
lines.extend(
|
||||
[
|
||||
"Article numbers",
|
||||
f"Column: {profile.article_numbers.column_name}",
|
||||
f"Distinct: {profile.article_numbers.distinct_count}",
|
||||
f"Duplicates: {profile.article_numbers.duplicate_count}",
|
||||
"",
|
||||
"Duplicate article numbers",
|
||||
]
|
||||
)
|
||||
for duplicate in profile.article_numbers.duplicates:
|
||||
lines.append(f"{duplicate.nr}\t{duplicate.count}")
|
||||
|
||||
if profile.rollcalc_comparison is not None:
|
||||
comparison = profile.rollcalc_comparison
|
||||
lines.extend(
|
||||
[
|
||||
"",
|
||||
"RollCalc comparison",
|
||||
f"RollCalc articles: {comparison.rollcalc_article_count}",
|
||||
f"Found: {comparison.found_count}",
|
||||
f"Missing: {comparison.missing_count}",
|
||||
"Missing article numbers",
|
||||
]
|
||||
)
|
||||
lines.extend(comparison.missing_article_numbers)
|
||||
|
||||
return "\n".join(lines) + "\n"
|
||||
|
||||
|
||||
def _read_csv(path: Path) -> tuple[list[dict[str, str]], list[str]]:
|
||||
with path.open("r", encoding="utf-8-sig", newline="") as csv_file:
|
||||
sample = csv_file.read(4096)
|
||||
csv_file.seek(0)
|
||||
dialect = csv.Sniffer().sniff(sample, delimiters=",;\t")
|
||||
reader = csv.DictReader(csv_file, dialect=dialect)
|
||||
fieldnames = list(reader.fieldnames or [])
|
||||
return list(reader), fieldnames
|
||||
|
||||
|
||||
def _profile_column(name: str, values: list[str]) -> ColumnProfile:
|
||||
filled_values = [value for value in values if not _is_empty(value)]
|
||||
empty_count = len(values) - len(filled_values)
|
||||
return ColumnProfile(
|
||||
name=name,
|
||||
filled_count=len(filled_values),
|
||||
empty_count=empty_count,
|
||||
distinct_count=len(set(filled_values)),
|
||||
inferred_type=infer_value_type(filled_values),
|
||||
)
|
||||
|
||||
|
||||
def infer_value_type(values: list[str]) -> str:
|
||||
"""Infer a simple data type for already-filled CSV values."""
|
||||
|
||||
if not values:
|
||||
return "empty"
|
||||
|
||||
value_types = {_infer_single_value_type(value) for value in values}
|
||||
if value_types == {"integer"}:
|
||||
return "integer"
|
||||
if value_types <= {"integer", "decimal"}:
|
||||
return "decimal"
|
||||
if value_types == {"text"}:
|
||||
return "text"
|
||||
return "mixed"
|
||||
|
||||
|
||||
def _infer_single_value_type(value: str) -> str:
|
||||
normalized = value.strip()
|
||||
if _is_integer(normalized):
|
||||
return "integer"
|
||||
if _is_decimal(normalized):
|
||||
return "decimal"
|
||||
return "text"
|
||||
|
||||
|
||||
def _is_integer(value: str) -> bool:
|
||||
if value.startswith(("+", "-")):
|
||||
value = value[1:]
|
||||
return value.isdecimal()
|
||||
|
||||
|
||||
def _is_decimal(value: str) -> bool:
|
||||
if value.count(",") + value.count(".") != 1:
|
||||
return False
|
||||
separator = "," if "," in value else "."
|
||||
left, right = value.split(separator, 1)
|
||||
if left.startswith(("+", "-")):
|
||||
left = left[1:]
|
||||
return left.isdecimal() and right.isdecimal()
|
||||
|
||||
|
||||
def _profile_article_numbers(
|
||||
rows: list[dict[str, str]],
|
||||
article_number_column: str,
|
||||
) -> ArticleNumberProfile:
|
||||
values = []
|
||||
for row in rows:
|
||||
value = row.get(article_number_column, "")
|
||||
if not _is_empty(value):
|
||||
values.append(value)
|
||||
|
||||
counts = Counter(values)
|
||||
duplicates = tuple(
|
||||
DuplicateArticleNumber(nr=nr, count=count)
|
||||
for nr, count in sorted(counts.items())
|
||||
if count > 1
|
||||
)
|
||||
return ArticleNumberProfile(
|
||||
column_name=article_number_column,
|
||||
distinct_count=len(counts),
|
||||
duplicate_count=len(duplicates),
|
||||
duplicates=duplicates,
|
||||
)
|
||||
|
||||
|
||||
def _compare_rollcalc_articles(
|
||||
rollcalc_path: Path,
|
||||
erp_article_numbers: set[str],
|
||||
) -> RollCalcComparison:
|
||||
rollcalc_articles = load_rollcalc_articles(Path(rollcalc_path))
|
||||
rollcalc_numbers = tuple(article.nr for article in rollcalc_articles)
|
||||
missing_article_numbers = tuple(nr for nr in rollcalc_numbers if nr not in erp_article_numbers)
|
||||
return RollCalcComparison(
|
||||
rollcalc_article_count=len(rollcalc_numbers),
|
||||
found_count=len(rollcalc_numbers) - len(missing_article_numbers),
|
||||
missing_count=len(missing_article_numbers),
|
||||
missing_article_numbers=missing_article_numbers,
|
||||
)
|
||||
|
||||
|
||||
def _is_empty(value: str | None) -> bool:
|
||||
return value is None or value == ""
|
||||
Reference in New Issue
Block a user