diff --git a/CHANGELOG.md b/CHANGELOG.md index 43f56d3..5677ac3 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -2,6 +2,7 @@ ## Unreleased +- ERP Explorer für CSV-Profiling, Dublettenanalyse und optionalen RollCalc-Abgleich ergänzt. - RollCalc-JSON-Importer mit strikter Struktur- und Typvalidierung ergänzt. - Initiale Projektstruktur angelegt. - Dokumentation fuer Architektur, Datenmodell, Datenwoerterbuch und Merge-Regeln erstellt. diff --git a/README.md b/README.md index 692ec64..e9fdf05 100644 --- a/README.md +++ b/README.md @@ -100,6 +100,23 @@ articles = load_rollcalc_articles(Path("data/source/rollcalc/article-data.json") Er validiert die bestehende RollCalc-JSON-Struktur strikt, erhält Artikelnummern als Strings und verändert die Quelldatei nicht. Details stehen in `docs/importers.md`. +## Interne Analysewerkzeuge + +Der ERP Explorer profiliert einen ERP-CSV-Export, ohne daraus Import- oder Merge-Regeln abzuleiten: + +```python +from pathlib import Path + +from article_data_manager.tools.erp_explorer import write_erp_profile_report + +write_erp_profile_report( + Path("data/source/erp/production-key-data.csv"), + rollcalc_path=Path("data/source/rollcalc/article-data.json"), +) +``` + +Der Textreport wird standardmäßig unter `data/reports/erp_profile_report.txt` erzeugt. Die Quelldaten werden nicht verändert. + ## Tests ```bash diff --git a/docs/importers.md b/docs/importers.md index 29dfe6e..48eda5f 100644 --- a/docs/importers.md +++ b/docs/importers.md @@ -58,3 +58,20 @@ Fehlermeldungen enthalten Datei, Datensatzindex und Feldname, soweit anwendbar. ## Quelldatei Der Importer liest die Quelldatei nur. Er schreibt, verändert oder repariert die Datei nicht und erzeugt keine Ausgabe auf stdout oder stderr. + +## ERP Explorer + +Der ERP Explorer unter `article_data_manager.tools.erp_explorer` ist ein internes Analysewerkzeug und kein ERP-Importer. Er liest einen ERP-CSV-Export, profiliert alle Spalten, zählt doppelte Artikelnummern in `SL_ITEM_NO` und kann optional RollCalc-Artikelnummern gegen den ERP-Export abgleichen. + +Öffentliche API: + +```python +from pathlib import Path + +from article_data_manager.tools.erp_explorer import profile_erp_csv, write_erp_profile_report + +profile = profile_erp_csv(Path("data/source/erp/production-key-data.csv")) +write_erp_profile_report(Path("data/source/erp/production-key-data.csv")) +``` + +Der Report wird standardmäßig als `data/reports/erp_profile_report.txt` geschrieben. Der Explorer verändert weder ERP-CSV noch RollCalc-JSON und erzeugt keine `article-data.json`. diff --git a/src/article_data_manager/tools/__init__.py b/src/article_data_manager/tools/__init__.py new file mode 100644 index 0000000..172dd9d --- /dev/null +++ b/src/article_data_manager/tools/__init__.py @@ -0,0 +1,2 @@ +"""Internal development tools.""" + diff --git a/src/article_data_manager/tools/erp_explorer.py b/src/article_data_manager/tools/erp_explorer.py new file mode 100644 index 0000000..0cf3955 --- /dev/null +++ b/src/article_data_manager/tools/erp_explorer.py @@ -0,0 +1,277 @@ +"""Read-only ERP CSV explorer for data profiling.""" + +from __future__ import annotations + +import csv +from collections import Counter +from dataclasses import dataclass +from pathlib import Path + +from article_data_manager.importers.rollcalc_json import load_rollcalc_articles + +DEFAULT_ARTICLE_NUMBER_COLUMN = "SL_ITEM_NO" +DEFAULT_REPORT_PATH = Path("data/reports/erp_profile_report.txt") + + +@dataclass(frozen=True, slots=True) +class ColumnProfile: + """Profile information for one CSV column.""" + + name: str + filled_count: int + empty_count: int + distinct_count: int + inferred_type: str + + +@dataclass(frozen=True, slots=True) +class DuplicateArticleNumber: + """Repeated ERP article number and its exact occurrence count.""" + + nr: str + count: int + + +@dataclass(frozen=True, slots=True) +class ArticleNumberProfile: + """Profile information for an ERP article number column.""" + + column_name: str + distinct_count: int + duplicate_count: int + duplicates: tuple[DuplicateArticleNumber, ...] + + +@dataclass(frozen=True, slots=True) +class RollCalcComparison: + """Simple ERP/RollCalc article number coverage comparison.""" + + rollcalc_article_count: int + found_count: int + missing_count: int + missing_article_numbers: tuple[str, ...] + + +@dataclass(frozen=True, slots=True) +class ErpProfile: + """Complete read-only profile of one ERP CSV export.""" + + file_name: str + row_count: int + column_count: int + columns: tuple[ColumnProfile, ...] + article_numbers: ArticleNumberProfile + rollcalc_comparison: RollCalcComparison | None = None + + +def profile_erp_csv( + csv_path: Path, + *, + article_number_column: str = DEFAULT_ARTICLE_NUMBER_COLUMN, + rollcalc_path: Path | None = None, +) -> ErpProfile: + """Profile an ERP CSV export without interpreting or changing its source data.""" + + source_path = Path(csv_path) + rows, fieldnames = _read_csv(source_path) + column_profiles = tuple( + _profile_column(name, [row.get(name, "") for row in rows]) for name in fieldnames + ) + article_number_profile = _profile_article_numbers(rows, article_number_column) + erp_article_numbers = {row.get(article_number_column, "") for row in rows} + comparison = ( + _compare_rollcalc_articles(rollcalc_path, erp_article_numbers) + if rollcalc_path is not None + else None + ) + return ErpProfile( + file_name=source_path.name, + row_count=len(rows), + column_count=len(fieldnames), + columns=column_profiles, + article_numbers=article_number_profile, + rollcalc_comparison=comparison, + ) + + +def write_erp_profile_report( + csv_path: Path, + *, + report_path: Path = DEFAULT_REPORT_PATH, + article_number_column: str = DEFAULT_ARTICLE_NUMBER_COLUMN, + rollcalc_path: Path | None = None, +) -> ErpProfile: + """Create a text profile report under the requested report path. + + The ERP CSV and optional RollCalc JSON are read only. The only write performed + by this function is the requested text report. + """ + + profile = profile_erp_csv( + csv_path, + article_number_column=article_number_column, + rollcalc_path=rollcalc_path, + ) + destination = Path(report_path) + destination.parent.mkdir(parents=True, exist_ok=True) + destination.write_text(render_erp_profile_report(profile), encoding="utf-8") + return profile + + +def render_erp_profile_report(profile: ErpProfile) -> str: + """Render an ERP profile as deterministic plain text.""" + + lines = [ + "ERP Profile Report", + f"File: {profile.file_name}", + f"Rows: {profile.row_count}", + f"Columns: {profile.column_count}", + "", + "Column profiles", + ] + for column in profile.columns: + lines.extend( + [ + f"Column: {column.name}", + f"Filled: {column.filled_count}", + f"Empty: {column.empty_count}", + f"Distinct: {column.distinct_count}", + f"Type: {column.inferred_type}", + "", + ] + ) + + lines.extend( + [ + "Article numbers", + f"Column: {profile.article_numbers.column_name}", + f"Distinct: {profile.article_numbers.distinct_count}", + f"Duplicates: {profile.article_numbers.duplicate_count}", + "", + "Duplicate article numbers", + ] + ) + for duplicate in profile.article_numbers.duplicates: + lines.append(f"{duplicate.nr}\t{duplicate.count}") + + if profile.rollcalc_comparison is not None: + comparison = profile.rollcalc_comparison + lines.extend( + [ + "", + "RollCalc comparison", + f"RollCalc articles: {comparison.rollcalc_article_count}", + f"Found: {comparison.found_count}", + f"Missing: {comparison.missing_count}", + "Missing article numbers", + ] + ) + lines.extend(comparison.missing_article_numbers) + + return "\n".join(lines) + "\n" + + +def _read_csv(path: Path) -> tuple[list[dict[str, str]], list[str]]: + with path.open("r", encoding="utf-8-sig", newline="") as csv_file: + sample = csv_file.read(4096) + csv_file.seek(0) + dialect = csv.Sniffer().sniff(sample, delimiters=",;\t") + reader = csv.DictReader(csv_file, dialect=dialect) + fieldnames = list(reader.fieldnames or []) + return list(reader), fieldnames + + +def _profile_column(name: str, values: list[str]) -> ColumnProfile: + filled_values = [value for value in values if not _is_empty(value)] + empty_count = len(values) - len(filled_values) + return ColumnProfile( + name=name, + filled_count=len(filled_values), + empty_count=empty_count, + distinct_count=len(set(filled_values)), + inferred_type=infer_value_type(filled_values), + ) + + +def infer_value_type(values: list[str]) -> str: + """Infer a simple data type for already-filled CSV values.""" + + if not values: + return "empty" + + value_types = {_infer_single_value_type(value) for value in values} + if value_types == {"integer"}: + return "integer" + if value_types <= {"integer", "decimal"}: + return "decimal" + if value_types == {"text"}: + return "text" + return "mixed" + + +def _infer_single_value_type(value: str) -> str: + normalized = value.strip() + if _is_integer(normalized): + return "integer" + if _is_decimal(normalized): + return "decimal" + return "text" + + +def _is_integer(value: str) -> bool: + if value.startswith(("+", "-")): + value = value[1:] + return value.isdecimal() + + +def _is_decimal(value: str) -> bool: + if value.count(",") + value.count(".") != 1: + return False + separator = "," if "," in value else "." + left, right = value.split(separator, 1) + if left.startswith(("+", "-")): + left = left[1:] + return left.isdecimal() and right.isdecimal() + + +def _profile_article_numbers( + rows: list[dict[str, str]], + article_number_column: str, +) -> ArticleNumberProfile: + values = [] + for row in rows: + value = row.get(article_number_column, "") + if not _is_empty(value): + values.append(value) + + counts = Counter(values) + duplicates = tuple( + DuplicateArticleNumber(nr=nr, count=count) + for nr, count in sorted(counts.items()) + if count > 1 + ) + return ArticleNumberProfile( + column_name=article_number_column, + distinct_count=len(counts), + duplicate_count=len(duplicates), + duplicates=duplicates, + ) + + +def _compare_rollcalc_articles( + rollcalc_path: Path, + erp_article_numbers: set[str], +) -> RollCalcComparison: + rollcalc_articles = load_rollcalc_articles(Path(rollcalc_path)) + rollcalc_numbers = tuple(article.nr for article in rollcalc_articles) + missing_article_numbers = tuple(nr for nr in rollcalc_numbers if nr not in erp_article_numbers) + return RollCalcComparison( + rollcalc_article_count=len(rollcalc_numbers), + found_count=len(rollcalc_numbers) - len(missing_article_numbers), + missing_count=len(missing_article_numbers), + missing_article_numbers=missing_article_numbers, + ) + + +def _is_empty(value: str | None) -> bool: + return value is None or value == "" diff --git a/tests/unit/tools/test_erp_explorer.py b/tests/unit/tools/test_erp_explorer.py new file mode 100644 index 0000000..8375ae3 --- /dev/null +++ b/tests/unit/tools/test_erp_explorer.py @@ -0,0 +1,152 @@ +import json +from pathlib import Path + +from article_data_manager.tools.erp_explorer import ( + infer_value_type, + profile_erp_csv, + render_erp_profile_report, + write_erp_profile_report, +) + + +def write_text(path: Path, content: str) -> Path: + path.write_text(content, encoding="utf-8") + return path + + +def write_rollcalc_json(path: Path, article_numbers: list[str]) -> Path: + payload = [ + { + "nr": nr, + "name": f"Article {nr}", + "thickness": 1.0, + "area_weight": 0.0, + "core_type": 0.0, + } + for nr in article_numbers + ] + path.write_text(json.dumps(payload), encoding="utf-8") + return path + + +def test_profiles_columns(tmp_path: Path) -> None: + csv_path = write_text( + tmp_path / "erp.csv", + "\n".join( + [ + "SL_ITEM_NO,ROP_PRODUCT_WIDTH,COMMENT,EMPTY_COLUMN", + "214700,\"6,00\",Alpha,", + "000123,1.20,Beta,", + "777777,,Alpha,", + ] + ), + ) + + profile = profile_erp_csv(csv_path) + + assert profile.file_name == "erp.csv" + assert profile.row_count == 3 + assert profile.column_count == 4 + columns = {column.name: column for column in profile.columns} + assert columns["ROP_PRODUCT_WIDTH"].filled_count == 2 + assert columns["ROP_PRODUCT_WIDTH"].empty_count == 1 + assert columns["ROP_PRODUCT_WIDTH"].distinct_count == 2 + assert columns["ROP_PRODUCT_WIDTH"].inferred_type == "decimal" + assert columns["COMMENT"].distinct_count == 2 + assert columns["COMMENT"].inferred_type == "text" + assert columns["EMPTY_COLUMN"].inferred_type == "empty" + + +def test_infers_value_types() -> None: + assert infer_value_type([]) == "empty" + assert infer_value_type(["1", "002", "-3"]) == "integer" + assert infer_value_type(["1", "2.5", "3,75"]) == "decimal" + assert infer_value_type(["Alpha", "Beta"]) == "text" + assert infer_value_type(["1", "Alpha"]) == "mixed" + + +def test_detects_duplicate_article_numbers(tmp_path: Path) -> None: + csv_path = write_text( + tmp_path / "erp.csv", + "\n".join( + [ + "SL_ITEM_NO,NAME", + "00001,A", + "1,B", + "00001,C", + "214700,D", + "214700,E", + ] + ), + ) + + article_numbers = profile_erp_csv(csv_path).article_numbers + + assert article_numbers.distinct_count == 3 + assert article_numbers.duplicate_count == 2 + assert [(duplicate.nr, duplicate.count) for duplicate in article_numbers.duplicates] == [ + ("00001", 2), + ("214700", 2), + ] + + +def test_compares_rollcalc_articles_when_path_is_provided(tmp_path: Path) -> None: + csv_path = write_text( + tmp_path / "erp.csv", + "\n".join( + [ + "SL_ITEM_NO,NAME", + "214700,A", + "000123,B", + ] + ), + ) + rollcalc_path = write_rollcalc_json(tmp_path / "article-data.json", ["214700", "999999"]) + + comparison = profile_erp_csv(csv_path, rollcalc_path=rollcalc_path).rollcalc_comparison + + assert comparison is not None + assert comparison.rollcalc_article_count == 2 + assert comparison.found_count == 1 + assert comparison.missing_count == 1 + assert comparison.missing_article_numbers == ("999999",) + + +def test_rollcalc_comparison_is_optional(tmp_path: Path) -> None: + csv_path = write_text( + tmp_path / "erp.csv", + "\n".join( + [ + "SL_ITEM_NO,NAME", + "214700,A", + ] + ), + ) + + assert profile_erp_csv(csv_path).rollcalc_comparison is None + + +def test_renders_and_writes_report(tmp_path: Path) -> None: + csv_path = write_text( + tmp_path / "erp.csv", + "\n".join( + [ + "SL_ITEM_NO,ROP_PRODUCT_WIDTH", + "214700,\"6,00\"", + "214700,\"6,00\"", + ] + ), + ) + report_path = tmp_path / "reports" / "erp_profile_report.txt" + + profile = write_erp_profile_report(csv_path, report_path=report_path) + report = report_path.read_text(encoding="utf-8") + + assert report == render_erp_profile_report(profile) + assert "ERP Profile Report" in report + assert "File: erp.csv" in report + assert "Rows: 2" in report + assert "Columns: 2" in report + assert "Column: ROP_PRODUCT_WIDTH" in report + assert "Duplicate article numbers" in report + assert "214700\t2" in report