Add ERP explorer
This commit is contained in:
@@ -2,6 +2,7 @@
|
||||
|
||||
## Unreleased
|
||||
|
||||
- ERP Explorer für CSV-Profiling, Dublettenanalyse und optionalen RollCalc-Abgleich ergänzt.
|
||||
- RollCalc-JSON-Importer mit strikter Struktur- und Typvalidierung ergänzt.
|
||||
- Initiale Projektstruktur angelegt.
|
||||
- Dokumentation fuer Architektur, Datenmodell, Datenwoerterbuch und Merge-Regeln erstellt.
|
||||
|
||||
@@ -100,6 +100,23 @@ articles = load_rollcalc_articles(Path("data/source/rollcalc/article-data.json")
|
||||
|
||||
Er validiert die bestehende RollCalc-JSON-Struktur strikt, erhält Artikelnummern als Strings und verändert die Quelldatei nicht. Details stehen in `docs/importers.md`.
|
||||
|
||||
## Interne Analysewerkzeuge
|
||||
|
||||
Der ERP Explorer profiliert einen ERP-CSV-Export, ohne daraus Import- oder Merge-Regeln abzuleiten:
|
||||
|
||||
```python
|
||||
from pathlib import Path
|
||||
|
||||
from article_data_manager.tools.erp_explorer import write_erp_profile_report
|
||||
|
||||
write_erp_profile_report(
|
||||
Path("data/source/erp/production-key-data.csv"),
|
||||
rollcalc_path=Path("data/source/rollcalc/article-data.json"),
|
||||
)
|
||||
```
|
||||
|
||||
Der Textreport wird standardmäßig unter `data/reports/erp_profile_report.txt` erzeugt. Die Quelldaten werden nicht verändert.
|
||||
|
||||
## Tests
|
||||
|
||||
```bash
|
||||
|
||||
@@ -58,3 +58,20 @@ Fehlermeldungen enthalten Datei, Datensatzindex und Feldname, soweit anwendbar.
|
||||
## Quelldatei
|
||||
|
||||
Der Importer liest die Quelldatei nur. Er schreibt, verändert oder repariert die Datei nicht und erzeugt keine Ausgabe auf stdout oder stderr.
|
||||
|
||||
## ERP Explorer
|
||||
|
||||
Der ERP Explorer unter `article_data_manager.tools.erp_explorer` ist ein internes Analysewerkzeug und kein ERP-Importer. Er liest einen ERP-CSV-Export, profiliert alle Spalten, zählt doppelte Artikelnummern in `SL_ITEM_NO` und kann optional RollCalc-Artikelnummern gegen den ERP-Export abgleichen.
|
||||
|
||||
Öffentliche API:
|
||||
|
||||
```python
|
||||
from pathlib import Path
|
||||
|
||||
from article_data_manager.tools.erp_explorer import profile_erp_csv, write_erp_profile_report
|
||||
|
||||
profile = profile_erp_csv(Path("data/source/erp/production-key-data.csv"))
|
||||
write_erp_profile_report(Path("data/source/erp/production-key-data.csv"))
|
||||
```
|
||||
|
||||
Der Report wird standardmäßig als `data/reports/erp_profile_report.txt` geschrieben. Der Explorer verändert weder ERP-CSV noch RollCalc-JSON und erzeugt keine `article-data.json`.
|
||||
|
||||
@@ -0,0 +1,2 @@
|
||||
"""Internal development tools."""
|
||||
|
||||
@@ -0,0 +1,277 @@
|
||||
"""Read-only ERP CSV explorer for data profiling."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import csv
|
||||
from collections import Counter
|
||||
from dataclasses import dataclass
|
||||
from pathlib import Path
|
||||
|
||||
from article_data_manager.importers.rollcalc_json import load_rollcalc_articles
|
||||
|
||||
DEFAULT_ARTICLE_NUMBER_COLUMN = "SL_ITEM_NO"
|
||||
DEFAULT_REPORT_PATH = Path("data/reports/erp_profile_report.txt")
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class ColumnProfile:
|
||||
"""Profile information for one CSV column."""
|
||||
|
||||
name: str
|
||||
filled_count: int
|
||||
empty_count: int
|
||||
distinct_count: int
|
||||
inferred_type: str
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class DuplicateArticleNumber:
|
||||
"""Repeated ERP article number and its exact occurrence count."""
|
||||
|
||||
nr: str
|
||||
count: int
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class ArticleNumberProfile:
|
||||
"""Profile information for an ERP article number column."""
|
||||
|
||||
column_name: str
|
||||
distinct_count: int
|
||||
duplicate_count: int
|
||||
duplicates: tuple[DuplicateArticleNumber, ...]
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class RollCalcComparison:
|
||||
"""Simple ERP/RollCalc article number coverage comparison."""
|
||||
|
||||
rollcalc_article_count: int
|
||||
found_count: int
|
||||
missing_count: int
|
||||
missing_article_numbers: tuple[str, ...]
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class ErpProfile:
|
||||
"""Complete read-only profile of one ERP CSV export."""
|
||||
|
||||
file_name: str
|
||||
row_count: int
|
||||
column_count: int
|
||||
columns: tuple[ColumnProfile, ...]
|
||||
article_numbers: ArticleNumberProfile
|
||||
rollcalc_comparison: RollCalcComparison | None = None
|
||||
|
||||
|
||||
def profile_erp_csv(
|
||||
csv_path: Path,
|
||||
*,
|
||||
article_number_column: str = DEFAULT_ARTICLE_NUMBER_COLUMN,
|
||||
rollcalc_path: Path | None = None,
|
||||
) -> ErpProfile:
|
||||
"""Profile an ERP CSV export without interpreting or changing its source data."""
|
||||
|
||||
source_path = Path(csv_path)
|
||||
rows, fieldnames = _read_csv(source_path)
|
||||
column_profiles = tuple(
|
||||
_profile_column(name, [row.get(name, "") for row in rows]) for name in fieldnames
|
||||
)
|
||||
article_number_profile = _profile_article_numbers(rows, article_number_column)
|
||||
erp_article_numbers = {row.get(article_number_column, "") for row in rows}
|
||||
comparison = (
|
||||
_compare_rollcalc_articles(rollcalc_path, erp_article_numbers)
|
||||
if rollcalc_path is not None
|
||||
else None
|
||||
)
|
||||
return ErpProfile(
|
||||
file_name=source_path.name,
|
||||
row_count=len(rows),
|
||||
column_count=len(fieldnames),
|
||||
columns=column_profiles,
|
||||
article_numbers=article_number_profile,
|
||||
rollcalc_comparison=comparison,
|
||||
)
|
||||
|
||||
|
||||
def write_erp_profile_report(
|
||||
csv_path: Path,
|
||||
*,
|
||||
report_path: Path = DEFAULT_REPORT_PATH,
|
||||
article_number_column: str = DEFAULT_ARTICLE_NUMBER_COLUMN,
|
||||
rollcalc_path: Path | None = None,
|
||||
) -> ErpProfile:
|
||||
"""Create a text profile report under the requested report path.
|
||||
|
||||
The ERP CSV and optional RollCalc JSON are read only. The only write performed
|
||||
by this function is the requested text report.
|
||||
"""
|
||||
|
||||
profile = profile_erp_csv(
|
||||
csv_path,
|
||||
article_number_column=article_number_column,
|
||||
rollcalc_path=rollcalc_path,
|
||||
)
|
||||
destination = Path(report_path)
|
||||
destination.parent.mkdir(parents=True, exist_ok=True)
|
||||
destination.write_text(render_erp_profile_report(profile), encoding="utf-8")
|
||||
return profile
|
||||
|
||||
|
||||
def render_erp_profile_report(profile: ErpProfile) -> str:
|
||||
"""Render an ERP profile as deterministic plain text."""
|
||||
|
||||
lines = [
|
||||
"ERP Profile Report",
|
||||
f"File: {profile.file_name}",
|
||||
f"Rows: {profile.row_count}",
|
||||
f"Columns: {profile.column_count}",
|
||||
"",
|
||||
"Column profiles",
|
||||
]
|
||||
for column in profile.columns:
|
||||
lines.extend(
|
||||
[
|
||||
f"Column: {column.name}",
|
||||
f"Filled: {column.filled_count}",
|
||||
f"Empty: {column.empty_count}",
|
||||
f"Distinct: {column.distinct_count}",
|
||||
f"Type: {column.inferred_type}",
|
||||
"",
|
||||
]
|
||||
)
|
||||
|
||||
lines.extend(
|
||||
[
|
||||
"Article numbers",
|
||||
f"Column: {profile.article_numbers.column_name}",
|
||||
f"Distinct: {profile.article_numbers.distinct_count}",
|
||||
f"Duplicates: {profile.article_numbers.duplicate_count}",
|
||||
"",
|
||||
"Duplicate article numbers",
|
||||
]
|
||||
)
|
||||
for duplicate in profile.article_numbers.duplicates:
|
||||
lines.append(f"{duplicate.nr}\t{duplicate.count}")
|
||||
|
||||
if profile.rollcalc_comparison is not None:
|
||||
comparison = profile.rollcalc_comparison
|
||||
lines.extend(
|
||||
[
|
||||
"",
|
||||
"RollCalc comparison",
|
||||
f"RollCalc articles: {comparison.rollcalc_article_count}",
|
||||
f"Found: {comparison.found_count}",
|
||||
f"Missing: {comparison.missing_count}",
|
||||
"Missing article numbers",
|
||||
]
|
||||
)
|
||||
lines.extend(comparison.missing_article_numbers)
|
||||
|
||||
return "\n".join(lines) + "\n"
|
||||
|
||||
|
||||
def _read_csv(path: Path) -> tuple[list[dict[str, str]], list[str]]:
|
||||
with path.open("r", encoding="utf-8-sig", newline="") as csv_file:
|
||||
sample = csv_file.read(4096)
|
||||
csv_file.seek(0)
|
||||
dialect = csv.Sniffer().sniff(sample, delimiters=",;\t")
|
||||
reader = csv.DictReader(csv_file, dialect=dialect)
|
||||
fieldnames = list(reader.fieldnames or [])
|
||||
return list(reader), fieldnames
|
||||
|
||||
|
||||
def _profile_column(name: str, values: list[str]) -> ColumnProfile:
|
||||
filled_values = [value for value in values if not _is_empty(value)]
|
||||
empty_count = len(values) - len(filled_values)
|
||||
return ColumnProfile(
|
||||
name=name,
|
||||
filled_count=len(filled_values),
|
||||
empty_count=empty_count,
|
||||
distinct_count=len(set(filled_values)),
|
||||
inferred_type=infer_value_type(filled_values),
|
||||
)
|
||||
|
||||
|
||||
def infer_value_type(values: list[str]) -> str:
|
||||
"""Infer a simple data type for already-filled CSV values."""
|
||||
|
||||
if not values:
|
||||
return "empty"
|
||||
|
||||
value_types = {_infer_single_value_type(value) for value in values}
|
||||
if value_types == {"integer"}:
|
||||
return "integer"
|
||||
if value_types <= {"integer", "decimal"}:
|
||||
return "decimal"
|
||||
if value_types == {"text"}:
|
||||
return "text"
|
||||
return "mixed"
|
||||
|
||||
|
||||
def _infer_single_value_type(value: str) -> str:
|
||||
normalized = value.strip()
|
||||
if _is_integer(normalized):
|
||||
return "integer"
|
||||
if _is_decimal(normalized):
|
||||
return "decimal"
|
||||
return "text"
|
||||
|
||||
|
||||
def _is_integer(value: str) -> bool:
|
||||
if value.startswith(("+", "-")):
|
||||
value = value[1:]
|
||||
return value.isdecimal()
|
||||
|
||||
|
||||
def _is_decimal(value: str) -> bool:
|
||||
if value.count(",") + value.count(".") != 1:
|
||||
return False
|
||||
separator = "," if "," in value else "."
|
||||
left, right = value.split(separator, 1)
|
||||
if left.startswith(("+", "-")):
|
||||
left = left[1:]
|
||||
return left.isdecimal() and right.isdecimal()
|
||||
|
||||
|
||||
def _profile_article_numbers(
|
||||
rows: list[dict[str, str]],
|
||||
article_number_column: str,
|
||||
) -> ArticleNumberProfile:
|
||||
values = []
|
||||
for row in rows:
|
||||
value = row.get(article_number_column, "")
|
||||
if not _is_empty(value):
|
||||
values.append(value)
|
||||
|
||||
counts = Counter(values)
|
||||
duplicates = tuple(
|
||||
DuplicateArticleNumber(nr=nr, count=count)
|
||||
for nr, count in sorted(counts.items())
|
||||
if count > 1
|
||||
)
|
||||
return ArticleNumberProfile(
|
||||
column_name=article_number_column,
|
||||
distinct_count=len(counts),
|
||||
duplicate_count=len(duplicates),
|
||||
duplicates=duplicates,
|
||||
)
|
||||
|
||||
|
||||
def _compare_rollcalc_articles(
|
||||
rollcalc_path: Path,
|
||||
erp_article_numbers: set[str],
|
||||
) -> RollCalcComparison:
|
||||
rollcalc_articles = load_rollcalc_articles(Path(rollcalc_path))
|
||||
rollcalc_numbers = tuple(article.nr for article in rollcalc_articles)
|
||||
missing_article_numbers = tuple(nr for nr in rollcalc_numbers if nr not in erp_article_numbers)
|
||||
return RollCalcComparison(
|
||||
rollcalc_article_count=len(rollcalc_numbers),
|
||||
found_count=len(rollcalc_numbers) - len(missing_article_numbers),
|
||||
missing_count=len(missing_article_numbers),
|
||||
missing_article_numbers=missing_article_numbers,
|
||||
)
|
||||
|
||||
|
||||
def _is_empty(value: str | None) -> bool:
|
||||
return value is None or value == ""
|
||||
@@ -0,0 +1,152 @@
|
||||
import json
|
||||
from pathlib import Path
|
||||
|
||||
from article_data_manager.tools.erp_explorer import (
|
||||
infer_value_type,
|
||||
profile_erp_csv,
|
||||
render_erp_profile_report,
|
||||
write_erp_profile_report,
|
||||
)
|
||||
|
||||
|
||||
def write_text(path: Path, content: str) -> Path:
|
||||
path.write_text(content, encoding="utf-8")
|
||||
return path
|
||||
|
||||
|
||||
def write_rollcalc_json(path: Path, article_numbers: list[str]) -> Path:
|
||||
payload = [
|
||||
{
|
||||
"nr": nr,
|
||||
"name": f"Article {nr}",
|
||||
"thickness": 1.0,
|
||||
"area_weight": 0.0,
|
||||
"core_type": 0.0,
|
||||
}
|
||||
for nr in article_numbers
|
||||
]
|
||||
path.write_text(json.dumps(payload), encoding="utf-8")
|
||||
return path
|
||||
|
||||
|
||||
def test_profiles_columns(tmp_path: Path) -> None:
|
||||
csv_path = write_text(
|
||||
tmp_path / "erp.csv",
|
||||
"\n".join(
|
||||
[
|
||||
"SL_ITEM_NO,ROP_PRODUCT_WIDTH,COMMENT,EMPTY_COLUMN",
|
||||
"214700,\"6,00\",Alpha,",
|
||||
"000123,1.20,Beta,",
|
||||
"777777,,Alpha,",
|
||||
]
|
||||
),
|
||||
)
|
||||
|
||||
profile = profile_erp_csv(csv_path)
|
||||
|
||||
assert profile.file_name == "erp.csv"
|
||||
assert profile.row_count == 3
|
||||
assert profile.column_count == 4
|
||||
columns = {column.name: column for column in profile.columns}
|
||||
assert columns["ROP_PRODUCT_WIDTH"].filled_count == 2
|
||||
assert columns["ROP_PRODUCT_WIDTH"].empty_count == 1
|
||||
assert columns["ROP_PRODUCT_WIDTH"].distinct_count == 2
|
||||
assert columns["ROP_PRODUCT_WIDTH"].inferred_type == "decimal"
|
||||
assert columns["COMMENT"].distinct_count == 2
|
||||
assert columns["COMMENT"].inferred_type == "text"
|
||||
assert columns["EMPTY_COLUMN"].inferred_type == "empty"
|
||||
|
||||
|
||||
def test_infers_value_types() -> None:
|
||||
assert infer_value_type([]) == "empty"
|
||||
assert infer_value_type(["1", "002", "-3"]) == "integer"
|
||||
assert infer_value_type(["1", "2.5", "3,75"]) == "decimal"
|
||||
assert infer_value_type(["Alpha", "Beta"]) == "text"
|
||||
assert infer_value_type(["1", "Alpha"]) == "mixed"
|
||||
|
||||
|
||||
def test_detects_duplicate_article_numbers(tmp_path: Path) -> None:
|
||||
csv_path = write_text(
|
||||
tmp_path / "erp.csv",
|
||||
"\n".join(
|
||||
[
|
||||
"SL_ITEM_NO,NAME",
|
||||
"00001,A",
|
||||
"1,B",
|
||||
"00001,C",
|
||||
"214700,D",
|
||||
"214700,E",
|
||||
]
|
||||
),
|
||||
)
|
||||
|
||||
article_numbers = profile_erp_csv(csv_path).article_numbers
|
||||
|
||||
assert article_numbers.distinct_count == 3
|
||||
assert article_numbers.duplicate_count == 2
|
||||
assert [(duplicate.nr, duplicate.count) for duplicate in article_numbers.duplicates] == [
|
||||
("00001", 2),
|
||||
("214700", 2),
|
||||
]
|
||||
|
||||
|
||||
def test_compares_rollcalc_articles_when_path_is_provided(tmp_path: Path) -> None:
|
||||
csv_path = write_text(
|
||||
tmp_path / "erp.csv",
|
||||
"\n".join(
|
||||
[
|
||||
"SL_ITEM_NO,NAME",
|
||||
"214700,A",
|
||||
"000123,B",
|
||||
]
|
||||
),
|
||||
)
|
||||
rollcalc_path = write_rollcalc_json(tmp_path / "article-data.json", ["214700", "999999"])
|
||||
|
||||
comparison = profile_erp_csv(csv_path, rollcalc_path=rollcalc_path).rollcalc_comparison
|
||||
|
||||
assert comparison is not None
|
||||
assert comparison.rollcalc_article_count == 2
|
||||
assert comparison.found_count == 1
|
||||
assert comparison.missing_count == 1
|
||||
assert comparison.missing_article_numbers == ("999999",)
|
||||
|
||||
|
||||
def test_rollcalc_comparison_is_optional(tmp_path: Path) -> None:
|
||||
csv_path = write_text(
|
||||
tmp_path / "erp.csv",
|
||||
"\n".join(
|
||||
[
|
||||
"SL_ITEM_NO,NAME",
|
||||
"214700,A",
|
||||
]
|
||||
),
|
||||
)
|
||||
|
||||
assert profile_erp_csv(csv_path).rollcalc_comparison is None
|
||||
|
||||
|
||||
def test_renders_and_writes_report(tmp_path: Path) -> None:
|
||||
csv_path = write_text(
|
||||
tmp_path / "erp.csv",
|
||||
"\n".join(
|
||||
[
|
||||
"SL_ITEM_NO,ROP_PRODUCT_WIDTH",
|
||||
"214700,\"6,00\"",
|
||||
"214700,\"6,00\"",
|
||||
]
|
||||
),
|
||||
)
|
||||
report_path = tmp_path / "reports" / "erp_profile_report.txt"
|
||||
|
||||
profile = write_erp_profile_report(csv_path, report_path=report_path)
|
||||
report = report_path.read_text(encoding="utf-8")
|
||||
|
||||
assert report == render_erp_profile_report(profile)
|
||||
assert "ERP Profile Report" in report
|
||||
assert "File: erp.csv" in report
|
||||
assert "Rows: 2" in report
|
||||
assert "Columns: 2" in report
|
||||
assert "Column: ROP_PRODUCT_WIDTH" in report
|
||||
assert "Duplicate article numbers" in report
|
||||
assert "214700\t2" in report
|
||||
Reference in New Issue
Block a user