278 lines
8.2 KiB
Python
278 lines
8.2 KiB
Python
"""Read-only ERP CSV explorer for data profiling."""
|
|
|
|
from __future__ import annotations
|
|
|
|
import csv
|
|
from collections import Counter
|
|
from dataclasses import dataclass
|
|
from pathlib import Path
|
|
|
|
from article_data_manager.importers.rollcalc_json import load_rollcalc_articles
|
|
|
|
DEFAULT_ARTICLE_NUMBER_COLUMN = "SL_ITEM_NO"
|
|
DEFAULT_REPORT_PATH = Path("data/reports/erp_profile_report.txt")
|
|
|
|
|
|
@dataclass(frozen=True, slots=True)
|
|
class ColumnProfile:
|
|
"""Profile information for one CSV column."""
|
|
|
|
name: str
|
|
filled_count: int
|
|
empty_count: int
|
|
distinct_count: int
|
|
inferred_type: str
|
|
|
|
|
|
@dataclass(frozen=True, slots=True)
|
|
class DuplicateArticleNumber:
|
|
"""Repeated ERP article number and its exact occurrence count."""
|
|
|
|
nr: str
|
|
count: int
|
|
|
|
|
|
@dataclass(frozen=True, slots=True)
|
|
class ArticleNumberProfile:
|
|
"""Profile information for an ERP article number column."""
|
|
|
|
column_name: str
|
|
distinct_count: int
|
|
duplicate_count: int
|
|
duplicates: tuple[DuplicateArticleNumber, ...]
|
|
|
|
|
|
@dataclass(frozen=True, slots=True)
|
|
class RollCalcComparison:
|
|
"""Simple ERP/RollCalc article number coverage comparison."""
|
|
|
|
rollcalc_article_count: int
|
|
found_count: int
|
|
missing_count: int
|
|
missing_article_numbers: tuple[str, ...]
|
|
|
|
|
|
@dataclass(frozen=True, slots=True)
|
|
class ErpProfile:
|
|
"""Complete read-only profile of one ERP CSV export."""
|
|
|
|
file_name: str
|
|
row_count: int
|
|
column_count: int
|
|
columns: tuple[ColumnProfile, ...]
|
|
article_numbers: ArticleNumberProfile
|
|
rollcalc_comparison: RollCalcComparison | None = None
|
|
|
|
|
|
def profile_erp_csv(
|
|
csv_path: Path,
|
|
*,
|
|
article_number_column: str = DEFAULT_ARTICLE_NUMBER_COLUMN,
|
|
rollcalc_path: Path | None = None,
|
|
) -> ErpProfile:
|
|
"""Profile an ERP CSV export without interpreting or changing its source data."""
|
|
|
|
source_path = Path(csv_path)
|
|
rows, fieldnames = _read_csv(source_path)
|
|
column_profiles = tuple(
|
|
_profile_column(name, [row.get(name, "") for row in rows]) for name in fieldnames
|
|
)
|
|
article_number_profile = _profile_article_numbers(rows, article_number_column)
|
|
erp_article_numbers = {row.get(article_number_column, "") for row in rows}
|
|
comparison = (
|
|
_compare_rollcalc_articles(rollcalc_path, erp_article_numbers)
|
|
if rollcalc_path is not None
|
|
else None
|
|
)
|
|
return ErpProfile(
|
|
file_name=source_path.name,
|
|
row_count=len(rows),
|
|
column_count=len(fieldnames),
|
|
columns=column_profiles,
|
|
article_numbers=article_number_profile,
|
|
rollcalc_comparison=comparison,
|
|
)
|
|
|
|
|
|
def write_erp_profile_report(
|
|
csv_path: Path,
|
|
*,
|
|
report_path: Path = DEFAULT_REPORT_PATH,
|
|
article_number_column: str = DEFAULT_ARTICLE_NUMBER_COLUMN,
|
|
rollcalc_path: Path | None = None,
|
|
) -> ErpProfile:
|
|
"""Create a text profile report under the requested report path.
|
|
|
|
The ERP CSV and optional RollCalc JSON are read only. The only write performed
|
|
by this function is the requested text report.
|
|
"""
|
|
|
|
profile = profile_erp_csv(
|
|
csv_path,
|
|
article_number_column=article_number_column,
|
|
rollcalc_path=rollcalc_path,
|
|
)
|
|
destination = Path(report_path)
|
|
destination.parent.mkdir(parents=True, exist_ok=True)
|
|
destination.write_text(render_erp_profile_report(profile), encoding="utf-8")
|
|
return profile
|
|
|
|
|
|
def render_erp_profile_report(profile: ErpProfile) -> str:
|
|
"""Render an ERP profile as deterministic plain text."""
|
|
|
|
lines = [
|
|
"ERP Profile Report",
|
|
f"File: {profile.file_name}",
|
|
f"Rows: {profile.row_count}",
|
|
f"Columns: {profile.column_count}",
|
|
"",
|
|
"Column profiles",
|
|
]
|
|
for column in profile.columns:
|
|
lines.extend(
|
|
[
|
|
f"Column: {column.name}",
|
|
f"Filled: {column.filled_count}",
|
|
f"Empty: {column.empty_count}",
|
|
f"Distinct: {column.distinct_count}",
|
|
f"Type: {column.inferred_type}",
|
|
"",
|
|
]
|
|
)
|
|
|
|
lines.extend(
|
|
[
|
|
"Article numbers",
|
|
f"Column: {profile.article_numbers.column_name}",
|
|
f"Distinct: {profile.article_numbers.distinct_count}",
|
|
f"Duplicates: {profile.article_numbers.duplicate_count}",
|
|
"",
|
|
"Duplicate article numbers",
|
|
]
|
|
)
|
|
for duplicate in profile.article_numbers.duplicates:
|
|
lines.append(f"{duplicate.nr}\t{duplicate.count}")
|
|
|
|
if profile.rollcalc_comparison is not None:
|
|
comparison = profile.rollcalc_comparison
|
|
lines.extend(
|
|
[
|
|
"",
|
|
"RollCalc comparison",
|
|
f"RollCalc articles: {comparison.rollcalc_article_count}",
|
|
f"Found: {comparison.found_count}",
|
|
f"Missing: {comparison.missing_count}",
|
|
"Missing article numbers",
|
|
]
|
|
)
|
|
lines.extend(comparison.missing_article_numbers)
|
|
|
|
return "\n".join(lines) + "\n"
|
|
|
|
|
|
def _read_csv(path: Path) -> tuple[list[dict[str, str]], list[str]]:
|
|
with path.open("r", encoding="utf-8-sig", newline="") as csv_file:
|
|
sample = csv_file.read(4096)
|
|
csv_file.seek(0)
|
|
dialect = csv.Sniffer().sniff(sample, delimiters=",;\t")
|
|
reader = csv.DictReader(csv_file, dialect=dialect)
|
|
fieldnames = list(reader.fieldnames or [])
|
|
return list(reader), fieldnames
|
|
|
|
|
|
def _profile_column(name: str, values: list[str]) -> ColumnProfile:
|
|
filled_values = [value for value in values if not _is_empty(value)]
|
|
empty_count = len(values) - len(filled_values)
|
|
return ColumnProfile(
|
|
name=name,
|
|
filled_count=len(filled_values),
|
|
empty_count=empty_count,
|
|
distinct_count=len(set(filled_values)),
|
|
inferred_type=infer_value_type(filled_values),
|
|
)
|
|
|
|
|
|
def infer_value_type(values: list[str]) -> str:
|
|
"""Infer a simple data type for already-filled CSV values."""
|
|
|
|
if not values:
|
|
return "empty"
|
|
|
|
value_types = {_infer_single_value_type(value) for value in values}
|
|
if value_types == {"integer"}:
|
|
return "integer"
|
|
if value_types <= {"integer", "decimal"}:
|
|
return "decimal"
|
|
if value_types == {"text"}:
|
|
return "text"
|
|
return "mixed"
|
|
|
|
|
|
def _infer_single_value_type(value: str) -> str:
|
|
normalized = value.strip()
|
|
if _is_integer(normalized):
|
|
return "integer"
|
|
if _is_decimal(normalized):
|
|
return "decimal"
|
|
return "text"
|
|
|
|
|
|
def _is_integer(value: str) -> bool:
|
|
if value.startswith(("+", "-")):
|
|
value = value[1:]
|
|
return value.isdecimal()
|
|
|
|
|
|
def _is_decimal(value: str) -> bool:
|
|
if value.count(",") + value.count(".") != 1:
|
|
return False
|
|
separator = "," if "," in value else "."
|
|
left, right = value.split(separator, 1)
|
|
if left.startswith(("+", "-")):
|
|
left = left[1:]
|
|
return left.isdecimal() and right.isdecimal()
|
|
|
|
|
|
def _profile_article_numbers(
|
|
rows: list[dict[str, str]],
|
|
article_number_column: str,
|
|
) -> ArticleNumberProfile:
|
|
values = []
|
|
for row in rows:
|
|
value = row.get(article_number_column, "")
|
|
if not _is_empty(value):
|
|
values.append(value)
|
|
|
|
counts = Counter(values)
|
|
duplicates = tuple(
|
|
DuplicateArticleNumber(nr=nr, count=count)
|
|
for nr, count in sorted(counts.items())
|
|
if count > 1
|
|
)
|
|
return ArticleNumberProfile(
|
|
column_name=article_number_column,
|
|
distinct_count=len(counts),
|
|
duplicate_count=len(duplicates),
|
|
duplicates=duplicates,
|
|
)
|
|
|
|
|
|
def _compare_rollcalc_articles(
|
|
rollcalc_path: Path,
|
|
erp_article_numbers: set[str],
|
|
) -> RollCalcComparison:
|
|
rollcalc_articles = load_rollcalc_articles(Path(rollcalc_path))
|
|
rollcalc_numbers = tuple(article.nr for article in rollcalc_articles)
|
|
missing_article_numbers = tuple(nr for nr in rollcalc_numbers if nr not in erp_article_numbers)
|
|
return RollCalcComparison(
|
|
rollcalc_article_count=len(rollcalc_numbers),
|
|
found_count=len(rollcalc_numbers) - len(missing_article_numbers),
|
|
missing_count=len(missing_article_numbers),
|
|
missing_article_numbers=missing_article_numbers,
|
|
)
|
|
|
|
|
|
def _is_empty(value: str | None) -> bool:
|
|
return value is None or value == ""
|