Add ERP explorer

This commit is contained in:
2026-07-29 14:03:06 +02:00
parent 720d789914
commit d92dc61277
6 changed files with 466 additions and 0 deletions
+1
View File
@@ -2,6 +2,7 @@
## Unreleased
- ERP Explorer für CSV-Profiling, Dublettenanalyse und optionalen RollCalc-Abgleich ergänzt.
- RollCalc-JSON-Importer mit strikter Struktur- und Typvalidierung ergänzt.
- Initiale Projektstruktur angelegt.
- Dokumentation fuer Architektur, Datenmodell, Datenwoerterbuch und Merge-Regeln erstellt.
+17
View File
@@ -100,6 +100,23 @@ articles = load_rollcalc_articles(Path("data/source/rollcalc/article-data.json")
Er validiert die bestehende RollCalc-JSON-Struktur strikt, erhält Artikelnummern als Strings und verändert die Quelldatei nicht. Details stehen in `docs/importers.md`.
## Interne Analysewerkzeuge
Der ERP Explorer profiliert einen ERP-CSV-Export, ohne daraus Import- oder Merge-Regeln abzuleiten:
```python
from pathlib import Path
from article_data_manager.tools.erp_explorer import write_erp_profile_report
write_erp_profile_report(
Path("data/source/erp/production-key-data.csv"),
rollcalc_path=Path("data/source/rollcalc/article-data.json"),
)
```
Der Textreport wird standardmäßig unter `data/reports/erp_profile_report.txt` erzeugt. Die Quelldaten werden nicht verändert.
## Tests
```bash
+17
View File
@@ -58,3 +58,20 @@ Fehlermeldungen enthalten Datei, Datensatzindex und Feldname, soweit anwendbar.
## Quelldatei
Der Importer liest die Quelldatei nur. Er schreibt, verändert oder repariert die Datei nicht und erzeugt keine Ausgabe auf stdout oder stderr.
## ERP Explorer
Der ERP Explorer unter `article_data_manager.tools.erp_explorer` ist ein internes Analysewerkzeug und kein ERP-Importer. Er liest einen ERP-CSV-Export, profiliert alle Spalten, zählt doppelte Artikelnummern in `SL_ITEM_NO` und kann optional RollCalc-Artikelnummern gegen den ERP-Export abgleichen.
Öffentliche API:
```python
from pathlib import Path
from article_data_manager.tools.erp_explorer import profile_erp_csv, write_erp_profile_report
profile = profile_erp_csv(Path("data/source/erp/production-key-data.csv"))
write_erp_profile_report(Path("data/source/erp/production-key-data.csv"))
```
Der Report wird standardmäßig als `data/reports/erp_profile_report.txt` geschrieben. Der Explorer verändert weder ERP-CSV noch RollCalc-JSON und erzeugt keine `article-data.json`.
@@ -0,0 +1,2 @@
"""Internal development tools."""
@@ -0,0 +1,277 @@
"""Read-only ERP CSV explorer for data profiling."""
from __future__ import annotations
import csv
from collections import Counter
from dataclasses import dataclass
from pathlib import Path
from article_data_manager.importers.rollcalc_json import load_rollcalc_articles
DEFAULT_ARTICLE_NUMBER_COLUMN = "SL_ITEM_NO"
DEFAULT_REPORT_PATH = Path("data/reports/erp_profile_report.txt")
@dataclass(frozen=True, slots=True)
class ColumnProfile:
"""Profile information for one CSV column."""
name: str
filled_count: int
empty_count: int
distinct_count: int
inferred_type: str
@dataclass(frozen=True, slots=True)
class DuplicateArticleNumber:
"""Repeated ERP article number and its exact occurrence count."""
nr: str
count: int
@dataclass(frozen=True, slots=True)
class ArticleNumberProfile:
"""Profile information for an ERP article number column."""
column_name: str
distinct_count: int
duplicate_count: int
duplicates: tuple[DuplicateArticleNumber, ...]
@dataclass(frozen=True, slots=True)
class RollCalcComparison:
"""Simple ERP/RollCalc article number coverage comparison."""
rollcalc_article_count: int
found_count: int
missing_count: int
missing_article_numbers: tuple[str, ...]
@dataclass(frozen=True, slots=True)
class ErpProfile:
"""Complete read-only profile of one ERP CSV export."""
file_name: str
row_count: int
column_count: int
columns: tuple[ColumnProfile, ...]
article_numbers: ArticleNumberProfile
rollcalc_comparison: RollCalcComparison | None = None
def profile_erp_csv(
csv_path: Path,
*,
article_number_column: str = DEFAULT_ARTICLE_NUMBER_COLUMN,
rollcalc_path: Path | None = None,
) -> ErpProfile:
"""Profile an ERP CSV export without interpreting or changing its source data."""
source_path = Path(csv_path)
rows, fieldnames = _read_csv(source_path)
column_profiles = tuple(
_profile_column(name, [row.get(name, "") for row in rows]) for name in fieldnames
)
article_number_profile = _profile_article_numbers(rows, article_number_column)
erp_article_numbers = {row.get(article_number_column, "") for row in rows}
comparison = (
_compare_rollcalc_articles(rollcalc_path, erp_article_numbers)
if rollcalc_path is not None
else None
)
return ErpProfile(
file_name=source_path.name,
row_count=len(rows),
column_count=len(fieldnames),
columns=column_profiles,
article_numbers=article_number_profile,
rollcalc_comparison=comparison,
)
def write_erp_profile_report(
csv_path: Path,
*,
report_path: Path = DEFAULT_REPORT_PATH,
article_number_column: str = DEFAULT_ARTICLE_NUMBER_COLUMN,
rollcalc_path: Path | None = None,
) -> ErpProfile:
"""Create a text profile report under the requested report path.
The ERP CSV and optional RollCalc JSON are read only. The only write performed
by this function is the requested text report.
"""
profile = profile_erp_csv(
csv_path,
article_number_column=article_number_column,
rollcalc_path=rollcalc_path,
)
destination = Path(report_path)
destination.parent.mkdir(parents=True, exist_ok=True)
destination.write_text(render_erp_profile_report(profile), encoding="utf-8")
return profile
def render_erp_profile_report(profile: ErpProfile) -> str:
"""Render an ERP profile as deterministic plain text."""
lines = [
"ERP Profile Report",
f"File: {profile.file_name}",
f"Rows: {profile.row_count}",
f"Columns: {profile.column_count}",
"",
"Column profiles",
]
for column in profile.columns:
lines.extend(
[
f"Column: {column.name}",
f"Filled: {column.filled_count}",
f"Empty: {column.empty_count}",
f"Distinct: {column.distinct_count}",
f"Type: {column.inferred_type}",
"",
]
)
lines.extend(
[
"Article numbers",
f"Column: {profile.article_numbers.column_name}",
f"Distinct: {profile.article_numbers.distinct_count}",
f"Duplicates: {profile.article_numbers.duplicate_count}",
"",
"Duplicate article numbers",
]
)
for duplicate in profile.article_numbers.duplicates:
lines.append(f"{duplicate.nr}\t{duplicate.count}")
if profile.rollcalc_comparison is not None:
comparison = profile.rollcalc_comparison
lines.extend(
[
"",
"RollCalc comparison",
f"RollCalc articles: {comparison.rollcalc_article_count}",
f"Found: {comparison.found_count}",
f"Missing: {comparison.missing_count}",
"Missing article numbers",
]
)
lines.extend(comparison.missing_article_numbers)
return "\n".join(lines) + "\n"
def _read_csv(path: Path) -> tuple[list[dict[str, str]], list[str]]:
with path.open("r", encoding="utf-8-sig", newline="") as csv_file:
sample = csv_file.read(4096)
csv_file.seek(0)
dialect = csv.Sniffer().sniff(sample, delimiters=",;\t")
reader = csv.DictReader(csv_file, dialect=dialect)
fieldnames = list(reader.fieldnames or [])
return list(reader), fieldnames
def _profile_column(name: str, values: list[str]) -> ColumnProfile:
filled_values = [value for value in values if not _is_empty(value)]
empty_count = len(values) - len(filled_values)
return ColumnProfile(
name=name,
filled_count=len(filled_values),
empty_count=empty_count,
distinct_count=len(set(filled_values)),
inferred_type=infer_value_type(filled_values),
)
def infer_value_type(values: list[str]) -> str:
"""Infer a simple data type for already-filled CSV values."""
if not values:
return "empty"
value_types = {_infer_single_value_type(value) for value in values}
if value_types == {"integer"}:
return "integer"
if value_types <= {"integer", "decimal"}:
return "decimal"
if value_types == {"text"}:
return "text"
return "mixed"
def _infer_single_value_type(value: str) -> str:
normalized = value.strip()
if _is_integer(normalized):
return "integer"
if _is_decimal(normalized):
return "decimal"
return "text"
def _is_integer(value: str) -> bool:
if value.startswith(("+", "-")):
value = value[1:]
return value.isdecimal()
def _is_decimal(value: str) -> bool:
if value.count(",") + value.count(".") != 1:
return False
separator = "," if "," in value else "."
left, right = value.split(separator, 1)
if left.startswith(("+", "-")):
left = left[1:]
return left.isdecimal() and right.isdecimal()
def _profile_article_numbers(
rows: list[dict[str, str]],
article_number_column: str,
) -> ArticleNumberProfile:
values = []
for row in rows:
value = row.get(article_number_column, "")
if not _is_empty(value):
values.append(value)
counts = Counter(values)
duplicates = tuple(
DuplicateArticleNumber(nr=nr, count=count)
for nr, count in sorted(counts.items())
if count > 1
)
return ArticleNumberProfile(
column_name=article_number_column,
distinct_count=len(counts),
duplicate_count=len(duplicates),
duplicates=duplicates,
)
def _compare_rollcalc_articles(
rollcalc_path: Path,
erp_article_numbers: set[str],
) -> RollCalcComparison:
rollcalc_articles = load_rollcalc_articles(Path(rollcalc_path))
rollcalc_numbers = tuple(article.nr for article in rollcalc_articles)
missing_article_numbers = tuple(nr for nr in rollcalc_numbers if nr not in erp_article_numbers)
return RollCalcComparison(
rollcalc_article_count=len(rollcalc_numbers),
found_count=len(rollcalc_numbers) - len(missing_article_numbers),
missing_count=len(missing_article_numbers),
missing_article_numbers=missing_article_numbers,
)
def _is_empty(value: str | None) -> bool:
return value is None or value == ""
+152
View File
@@ -0,0 +1,152 @@
import json
from pathlib import Path
from article_data_manager.tools.erp_explorer import (
infer_value_type,
profile_erp_csv,
render_erp_profile_report,
write_erp_profile_report,
)
def write_text(path: Path, content: str) -> Path:
path.write_text(content, encoding="utf-8")
return path
def write_rollcalc_json(path: Path, article_numbers: list[str]) -> Path:
payload = [
{
"nr": nr,
"name": f"Article {nr}",
"thickness": 1.0,
"area_weight": 0.0,
"core_type": 0.0,
}
for nr in article_numbers
]
path.write_text(json.dumps(payload), encoding="utf-8")
return path
def test_profiles_columns(tmp_path: Path) -> None:
csv_path = write_text(
tmp_path / "erp.csv",
"\n".join(
[
"SL_ITEM_NO,ROP_PRODUCT_WIDTH,COMMENT,EMPTY_COLUMN",
"214700,\"6,00\",Alpha,",
"000123,1.20,Beta,",
"777777,,Alpha,",
]
),
)
profile = profile_erp_csv(csv_path)
assert profile.file_name == "erp.csv"
assert profile.row_count == 3
assert profile.column_count == 4
columns = {column.name: column for column in profile.columns}
assert columns["ROP_PRODUCT_WIDTH"].filled_count == 2
assert columns["ROP_PRODUCT_WIDTH"].empty_count == 1
assert columns["ROP_PRODUCT_WIDTH"].distinct_count == 2
assert columns["ROP_PRODUCT_WIDTH"].inferred_type == "decimal"
assert columns["COMMENT"].distinct_count == 2
assert columns["COMMENT"].inferred_type == "text"
assert columns["EMPTY_COLUMN"].inferred_type == "empty"
def test_infers_value_types() -> None:
assert infer_value_type([]) == "empty"
assert infer_value_type(["1", "002", "-3"]) == "integer"
assert infer_value_type(["1", "2.5", "3,75"]) == "decimal"
assert infer_value_type(["Alpha", "Beta"]) == "text"
assert infer_value_type(["1", "Alpha"]) == "mixed"
def test_detects_duplicate_article_numbers(tmp_path: Path) -> None:
csv_path = write_text(
tmp_path / "erp.csv",
"\n".join(
[
"SL_ITEM_NO,NAME",
"00001,A",
"1,B",
"00001,C",
"214700,D",
"214700,E",
]
),
)
article_numbers = profile_erp_csv(csv_path).article_numbers
assert article_numbers.distinct_count == 3
assert article_numbers.duplicate_count == 2
assert [(duplicate.nr, duplicate.count) for duplicate in article_numbers.duplicates] == [
("00001", 2),
("214700", 2),
]
def test_compares_rollcalc_articles_when_path_is_provided(tmp_path: Path) -> None:
csv_path = write_text(
tmp_path / "erp.csv",
"\n".join(
[
"SL_ITEM_NO,NAME",
"214700,A",
"000123,B",
]
),
)
rollcalc_path = write_rollcalc_json(tmp_path / "article-data.json", ["214700", "999999"])
comparison = profile_erp_csv(csv_path, rollcalc_path=rollcalc_path).rollcalc_comparison
assert comparison is not None
assert comparison.rollcalc_article_count == 2
assert comparison.found_count == 1
assert comparison.missing_count == 1
assert comparison.missing_article_numbers == ("999999",)
def test_rollcalc_comparison_is_optional(tmp_path: Path) -> None:
csv_path = write_text(
tmp_path / "erp.csv",
"\n".join(
[
"SL_ITEM_NO,NAME",
"214700,A",
]
),
)
assert profile_erp_csv(csv_path).rollcalc_comparison is None
def test_renders_and_writes_report(tmp_path: Path) -> None:
csv_path = write_text(
tmp_path / "erp.csv",
"\n".join(
[
"SL_ITEM_NO,ROP_PRODUCT_WIDTH",
"214700,\"6,00\"",
"214700,\"6,00\"",
]
),
)
report_path = tmp_path / "reports" / "erp_profile_report.txt"
profile = write_erp_profile_report(csv_path, report_path=report_path)
report = report_path.read_text(encoding="utf-8")
assert report == render_erp_profile_report(profile)
assert "ERP Profile Report" in report
assert "File: erp.csv" in report
assert "Rows: 2" in report
assert "Columns: 2" in report
assert "Column: ROP_PRODUCT_WIDTH" in report
assert "Duplicate article numbers" in report
assert "214700\t2" in report