"""Read-only ERP CSV explorer for data profiling.""" from __future__ import annotations import csv from collections import Counter from dataclasses import dataclass from pathlib import Path from article_data_manager.importers.rollcalc_json import load_rollcalc_articles DEFAULT_ARTICLE_NUMBER_COLUMN = "SL_ITEM_NO" DEFAULT_REPORT_PATH = Path("data/reports/erp_profile_report.txt") @dataclass(frozen=True, slots=True) class ColumnProfile: """Profile information for one CSV column.""" name: str filled_count: int empty_count: int distinct_count: int inferred_type: str @dataclass(frozen=True, slots=True) class DuplicateArticleNumber: """Repeated ERP article number and its exact occurrence count.""" nr: str count: int @dataclass(frozen=True, slots=True) class ArticleNumberProfile: """Profile information for an ERP article number column.""" column_name: str distinct_count: int duplicate_count: int duplicates: tuple[DuplicateArticleNumber, ...] @dataclass(frozen=True, slots=True) class RollCalcComparison: """Simple ERP/RollCalc article number coverage comparison.""" rollcalc_article_count: int found_count: int missing_count: int missing_article_numbers: tuple[str, ...] @dataclass(frozen=True, slots=True) class ErpProfile: """Complete read-only profile of one ERP CSV export.""" file_name: str row_count: int column_count: int columns: tuple[ColumnProfile, ...] article_numbers: ArticleNumberProfile rollcalc_comparison: RollCalcComparison | None = None def profile_erp_csv( csv_path: Path, *, article_number_column: str = DEFAULT_ARTICLE_NUMBER_COLUMN, rollcalc_path: Path | None = None, ) -> ErpProfile: """Profile an ERP CSV export without interpreting or changing its source data.""" source_path = Path(csv_path) rows, fieldnames = _read_csv(source_path) column_profiles = tuple( _profile_column(name, [row.get(name, "") for row in rows]) for name in fieldnames ) article_number_profile = _profile_article_numbers(rows, article_number_column) erp_article_numbers = {row.get(article_number_column, "") for row in rows} comparison = ( _compare_rollcalc_articles(rollcalc_path, erp_article_numbers) if rollcalc_path is not None else None ) return ErpProfile( file_name=source_path.name, row_count=len(rows), column_count=len(fieldnames), columns=column_profiles, article_numbers=article_number_profile, rollcalc_comparison=comparison, ) def write_erp_profile_report( csv_path: Path, *, report_path: Path = DEFAULT_REPORT_PATH, article_number_column: str = DEFAULT_ARTICLE_NUMBER_COLUMN, rollcalc_path: Path | None = None, ) -> ErpProfile: """Create a text profile report under the requested report path. The ERP CSV and optional RollCalc JSON are read only. The only write performed by this function is the requested text report. """ profile = profile_erp_csv( csv_path, article_number_column=article_number_column, rollcalc_path=rollcalc_path, ) destination = Path(report_path) destination.parent.mkdir(parents=True, exist_ok=True) destination.write_text(render_erp_profile_report(profile), encoding="utf-8") return profile def render_erp_profile_report(profile: ErpProfile) -> str: """Render an ERP profile as deterministic plain text.""" lines = [ "ERP Profile Report", f"File: {profile.file_name}", f"Rows: {profile.row_count}", f"Columns: {profile.column_count}", "", "Column profiles", ] for column in profile.columns: lines.extend( [ f"Column: {column.name}", f"Filled: {column.filled_count}", f"Empty: {column.empty_count}", f"Distinct: {column.distinct_count}", f"Type: {column.inferred_type}", "", ] ) lines.extend( [ "Article numbers", f"Column: {profile.article_numbers.column_name}", f"Distinct: {profile.article_numbers.distinct_count}", f"Duplicates: {profile.article_numbers.duplicate_count}", "", "Duplicate article numbers", ] ) for duplicate in profile.article_numbers.duplicates: lines.append(f"{duplicate.nr}\t{duplicate.count}") if profile.rollcalc_comparison is not None: comparison = profile.rollcalc_comparison lines.extend( [ "", "RollCalc comparison", f"RollCalc articles: {comparison.rollcalc_article_count}", f"Found: {comparison.found_count}", f"Missing: {comparison.missing_count}", "Missing article numbers", ] ) lines.extend(comparison.missing_article_numbers) return "\n".join(lines) + "\n" def _read_csv(path: Path) -> tuple[list[dict[str, str]], list[str]]: with path.open("r", encoding="utf-8-sig", newline="") as csv_file: sample = csv_file.read(4096) csv_file.seek(0) dialect = csv.Sniffer().sniff(sample, delimiters=",;\t") reader = csv.DictReader(csv_file, dialect=dialect) fieldnames = list(reader.fieldnames or []) return list(reader), fieldnames def _profile_column(name: str, values: list[str]) -> ColumnProfile: filled_values = [value for value in values if not _is_empty(value)] empty_count = len(values) - len(filled_values) return ColumnProfile( name=name, filled_count=len(filled_values), empty_count=empty_count, distinct_count=len(set(filled_values)), inferred_type=infer_value_type(filled_values), ) def infer_value_type(values: list[str]) -> str: """Infer a simple data type for already-filled CSV values.""" if not values: return "empty" value_types = {_infer_single_value_type(value) for value in values} if value_types == {"integer"}: return "integer" if value_types <= {"integer", "decimal"}: return "decimal" if value_types == {"text"}: return "text" return "mixed" def _infer_single_value_type(value: str) -> str: normalized = value.strip() if _is_integer(normalized): return "integer" if _is_decimal(normalized): return "decimal" return "text" def _is_integer(value: str) -> bool: if value.startswith(("+", "-")): value = value[1:] return value.isdecimal() def _is_decimal(value: str) -> bool: if value.count(",") + value.count(".") != 1: return False separator = "," if "," in value else "." left, right = value.split(separator, 1) if left.startswith(("+", "-")): left = left[1:] return left.isdecimal() and right.isdecimal() def _profile_article_numbers( rows: list[dict[str, str]], article_number_column: str, ) -> ArticleNumberProfile: values = [] for row in rows: value = row.get(article_number_column, "") if not _is_empty(value): values.append(value) counts = Counter(values) duplicates = tuple( DuplicateArticleNumber(nr=nr, count=count) for nr, count in sorted(counts.items()) if count > 1 ) return ArticleNumberProfile( column_name=article_number_column, distinct_count=len(counts), duplicate_count=len(duplicates), duplicates=duplicates, ) def _compare_rollcalc_articles( rollcalc_path: Path, erp_article_numbers: set[str], ) -> RollCalcComparison: rollcalc_articles = load_rollcalc_articles(Path(rollcalc_path)) rollcalc_numbers = tuple(article.nr for article in rollcalc_articles) missing_article_numbers = tuple(nr for nr in rollcalc_numbers if nr not in erp_article_numbers) return RollCalcComparison( rollcalc_article_count=len(rollcalc_numbers), found_count=len(rollcalc_numbers) - len(missing_article_numbers), missing_count=len(missing_article_numbers), missing_article_numbers=missing_article_numbers, ) def _is_empty(value: str | None) -> bool: return value is None or value == ""