Add ERP explorer
This commit is contained in:
@@ -2,6 +2,7 @@
|
|||||||
|
|
||||||
## Unreleased
|
## Unreleased
|
||||||
|
|
||||||
|
- ERP Explorer für CSV-Profiling, Dublettenanalyse und optionalen RollCalc-Abgleich ergänzt.
|
||||||
- RollCalc-JSON-Importer mit strikter Struktur- und Typvalidierung ergänzt.
|
- RollCalc-JSON-Importer mit strikter Struktur- und Typvalidierung ergänzt.
|
||||||
- Initiale Projektstruktur angelegt.
|
- Initiale Projektstruktur angelegt.
|
||||||
- Dokumentation fuer Architektur, Datenmodell, Datenwoerterbuch und Merge-Regeln erstellt.
|
- Dokumentation fuer Architektur, Datenmodell, Datenwoerterbuch und Merge-Regeln erstellt.
|
||||||
|
|||||||
@@ -100,6 +100,23 @@ articles = load_rollcalc_articles(Path("data/source/rollcalc/article-data.json")
|
|||||||
|
|
||||||
Er validiert die bestehende RollCalc-JSON-Struktur strikt, erhält Artikelnummern als Strings und verändert die Quelldatei nicht. Details stehen in `docs/importers.md`.
|
Er validiert die bestehende RollCalc-JSON-Struktur strikt, erhält Artikelnummern als Strings und verändert die Quelldatei nicht. Details stehen in `docs/importers.md`.
|
||||||
|
|
||||||
|
## Interne Analysewerkzeuge
|
||||||
|
|
||||||
|
Der ERP Explorer profiliert einen ERP-CSV-Export, ohne daraus Import- oder Merge-Regeln abzuleiten:
|
||||||
|
|
||||||
|
```python
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
from article_data_manager.tools.erp_explorer import write_erp_profile_report
|
||||||
|
|
||||||
|
write_erp_profile_report(
|
||||||
|
Path("data/source/erp/production-key-data.csv"),
|
||||||
|
rollcalc_path=Path("data/source/rollcalc/article-data.json"),
|
||||||
|
)
|
||||||
|
```
|
||||||
|
|
||||||
|
Der Textreport wird standardmäßig unter `data/reports/erp_profile_report.txt` erzeugt. Die Quelldaten werden nicht verändert.
|
||||||
|
|
||||||
## Tests
|
## Tests
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
|
|||||||
@@ -58,3 +58,20 @@ Fehlermeldungen enthalten Datei, Datensatzindex und Feldname, soweit anwendbar.
|
|||||||
## Quelldatei
|
## Quelldatei
|
||||||
|
|
||||||
Der Importer liest die Quelldatei nur. Er schreibt, verändert oder repariert die Datei nicht und erzeugt keine Ausgabe auf stdout oder stderr.
|
Der Importer liest die Quelldatei nur. Er schreibt, verändert oder repariert die Datei nicht und erzeugt keine Ausgabe auf stdout oder stderr.
|
||||||
|
|
||||||
|
## ERP Explorer
|
||||||
|
|
||||||
|
Der ERP Explorer unter `article_data_manager.tools.erp_explorer` ist ein internes Analysewerkzeug und kein ERP-Importer. Er liest einen ERP-CSV-Export, profiliert alle Spalten, zählt doppelte Artikelnummern in `SL_ITEM_NO` und kann optional RollCalc-Artikelnummern gegen den ERP-Export abgleichen.
|
||||||
|
|
||||||
|
Öffentliche API:
|
||||||
|
|
||||||
|
```python
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
from article_data_manager.tools.erp_explorer import profile_erp_csv, write_erp_profile_report
|
||||||
|
|
||||||
|
profile = profile_erp_csv(Path("data/source/erp/production-key-data.csv"))
|
||||||
|
write_erp_profile_report(Path("data/source/erp/production-key-data.csv"))
|
||||||
|
```
|
||||||
|
|
||||||
|
Der Report wird standardmäßig als `data/reports/erp_profile_report.txt` geschrieben. Der Explorer verändert weder ERP-CSV noch RollCalc-JSON und erzeugt keine `article-data.json`.
|
||||||
|
|||||||
@@ -0,0 +1,2 @@
|
|||||||
|
"""Internal development tools."""
|
||||||
|
|
||||||
@@ -0,0 +1,277 @@
|
|||||||
|
"""Read-only ERP CSV explorer for data profiling."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import csv
|
||||||
|
from collections import Counter
|
||||||
|
from dataclasses import dataclass
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
from article_data_manager.importers.rollcalc_json import load_rollcalc_articles
|
||||||
|
|
||||||
|
DEFAULT_ARTICLE_NUMBER_COLUMN = "SL_ITEM_NO"
|
||||||
|
DEFAULT_REPORT_PATH = Path("data/reports/erp_profile_report.txt")
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass(frozen=True, slots=True)
|
||||||
|
class ColumnProfile:
|
||||||
|
"""Profile information for one CSV column."""
|
||||||
|
|
||||||
|
name: str
|
||||||
|
filled_count: int
|
||||||
|
empty_count: int
|
||||||
|
distinct_count: int
|
||||||
|
inferred_type: str
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass(frozen=True, slots=True)
|
||||||
|
class DuplicateArticleNumber:
|
||||||
|
"""Repeated ERP article number and its exact occurrence count."""
|
||||||
|
|
||||||
|
nr: str
|
||||||
|
count: int
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass(frozen=True, slots=True)
|
||||||
|
class ArticleNumberProfile:
|
||||||
|
"""Profile information for an ERP article number column."""
|
||||||
|
|
||||||
|
column_name: str
|
||||||
|
distinct_count: int
|
||||||
|
duplicate_count: int
|
||||||
|
duplicates: tuple[DuplicateArticleNumber, ...]
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass(frozen=True, slots=True)
|
||||||
|
class RollCalcComparison:
|
||||||
|
"""Simple ERP/RollCalc article number coverage comparison."""
|
||||||
|
|
||||||
|
rollcalc_article_count: int
|
||||||
|
found_count: int
|
||||||
|
missing_count: int
|
||||||
|
missing_article_numbers: tuple[str, ...]
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass(frozen=True, slots=True)
|
||||||
|
class ErpProfile:
|
||||||
|
"""Complete read-only profile of one ERP CSV export."""
|
||||||
|
|
||||||
|
file_name: str
|
||||||
|
row_count: int
|
||||||
|
column_count: int
|
||||||
|
columns: tuple[ColumnProfile, ...]
|
||||||
|
article_numbers: ArticleNumberProfile
|
||||||
|
rollcalc_comparison: RollCalcComparison | None = None
|
||||||
|
|
||||||
|
|
||||||
|
def profile_erp_csv(
|
||||||
|
csv_path: Path,
|
||||||
|
*,
|
||||||
|
article_number_column: str = DEFAULT_ARTICLE_NUMBER_COLUMN,
|
||||||
|
rollcalc_path: Path | None = None,
|
||||||
|
) -> ErpProfile:
|
||||||
|
"""Profile an ERP CSV export without interpreting or changing its source data."""
|
||||||
|
|
||||||
|
source_path = Path(csv_path)
|
||||||
|
rows, fieldnames = _read_csv(source_path)
|
||||||
|
column_profiles = tuple(
|
||||||
|
_profile_column(name, [row.get(name, "") for row in rows]) for name in fieldnames
|
||||||
|
)
|
||||||
|
article_number_profile = _profile_article_numbers(rows, article_number_column)
|
||||||
|
erp_article_numbers = {row.get(article_number_column, "") for row in rows}
|
||||||
|
comparison = (
|
||||||
|
_compare_rollcalc_articles(rollcalc_path, erp_article_numbers)
|
||||||
|
if rollcalc_path is not None
|
||||||
|
else None
|
||||||
|
)
|
||||||
|
return ErpProfile(
|
||||||
|
file_name=source_path.name,
|
||||||
|
row_count=len(rows),
|
||||||
|
column_count=len(fieldnames),
|
||||||
|
columns=column_profiles,
|
||||||
|
article_numbers=article_number_profile,
|
||||||
|
rollcalc_comparison=comparison,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def write_erp_profile_report(
|
||||||
|
csv_path: Path,
|
||||||
|
*,
|
||||||
|
report_path: Path = DEFAULT_REPORT_PATH,
|
||||||
|
article_number_column: str = DEFAULT_ARTICLE_NUMBER_COLUMN,
|
||||||
|
rollcalc_path: Path | None = None,
|
||||||
|
) -> ErpProfile:
|
||||||
|
"""Create a text profile report under the requested report path.
|
||||||
|
|
||||||
|
The ERP CSV and optional RollCalc JSON are read only. The only write performed
|
||||||
|
by this function is the requested text report.
|
||||||
|
"""
|
||||||
|
|
||||||
|
profile = profile_erp_csv(
|
||||||
|
csv_path,
|
||||||
|
article_number_column=article_number_column,
|
||||||
|
rollcalc_path=rollcalc_path,
|
||||||
|
)
|
||||||
|
destination = Path(report_path)
|
||||||
|
destination.parent.mkdir(parents=True, exist_ok=True)
|
||||||
|
destination.write_text(render_erp_profile_report(profile), encoding="utf-8")
|
||||||
|
return profile
|
||||||
|
|
||||||
|
|
||||||
|
def render_erp_profile_report(profile: ErpProfile) -> str:
|
||||||
|
"""Render an ERP profile as deterministic plain text."""
|
||||||
|
|
||||||
|
lines = [
|
||||||
|
"ERP Profile Report",
|
||||||
|
f"File: {profile.file_name}",
|
||||||
|
f"Rows: {profile.row_count}",
|
||||||
|
f"Columns: {profile.column_count}",
|
||||||
|
"",
|
||||||
|
"Column profiles",
|
||||||
|
]
|
||||||
|
for column in profile.columns:
|
||||||
|
lines.extend(
|
||||||
|
[
|
||||||
|
f"Column: {column.name}",
|
||||||
|
f"Filled: {column.filled_count}",
|
||||||
|
f"Empty: {column.empty_count}",
|
||||||
|
f"Distinct: {column.distinct_count}",
|
||||||
|
f"Type: {column.inferred_type}",
|
||||||
|
"",
|
||||||
|
]
|
||||||
|
)
|
||||||
|
|
||||||
|
lines.extend(
|
||||||
|
[
|
||||||
|
"Article numbers",
|
||||||
|
f"Column: {profile.article_numbers.column_name}",
|
||||||
|
f"Distinct: {profile.article_numbers.distinct_count}",
|
||||||
|
f"Duplicates: {profile.article_numbers.duplicate_count}",
|
||||||
|
"",
|
||||||
|
"Duplicate article numbers",
|
||||||
|
]
|
||||||
|
)
|
||||||
|
for duplicate in profile.article_numbers.duplicates:
|
||||||
|
lines.append(f"{duplicate.nr}\t{duplicate.count}")
|
||||||
|
|
||||||
|
if profile.rollcalc_comparison is not None:
|
||||||
|
comparison = profile.rollcalc_comparison
|
||||||
|
lines.extend(
|
||||||
|
[
|
||||||
|
"",
|
||||||
|
"RollCalc comparison",
|
||||||
|
f"RollCalc articles: {comparison.rollcalc_article_count}",
|
||||||
|
f"Found: {comparison.found_count}",
|
||||||
|
f"Missing: {comparison.missing_count}",
|
||||||
|
"Missing article numbers",
|
||||||
|
]
|
||||||
|
)
|
||||||
|
lines.extend(comparison.missing_article_numbers)
|
||||||
|
|
||||||
|
return "\n".join(lines) + "\n"
|
||||||
|
|
||||||
|
|
||||||
|
def _read_csv(path: Path) -> tuple[list[dict[str, str]], list[str]]:
|
||||||
|
with path.open("r", encoding="utf-8-sig", newline="") as csv_file:
|
||||||
|
sample = csv_file.read(4096)
|
||||||
|
csv_file.seek(0)
|
||||||
|
dialect = csv.Sniffer().sniff(sample, delimiters=",;\t")
|
||||||
|
reader = csv.DictReader(csv_file, dialect=dialect)
|
||||||
|
fieldnames = list(reader.fieldnames or [])
|
||||||
|
return list(reader), fieldnames
|
||||||
|
|
||||||
|
|
||||||
|
def _profile_column(name: str, values: list[str]) -> ColumnProfile:
|
||||||
|
filled_values = [value for value in values if not _is_empty(value)]
|
||||||
|
empty_count = len(values) - len(filled_values)
|
||||||
|
return ColumnProfile(
|
||||||
|
name=name,
|
||||||
|
filled_count=len(filled_values),
|
||||||
|
empty_count=empty_count,
|
||||||
|
distinct_count=len(set(filled_values)),
|
||||||
|
inferred_type=infer_value_type(filled_values),
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def infer_value_type(values: list[str]) -> str:
|
||||||
|
"""Infer a simple data type for already-filled CSV values."""
|
||||||
|
|
||||||
|
if not values:
|
||||||
|
return "empty"
|
||||||
|
|
||||||
|
value_types = {_infer_single_value_type(value) for value in values}
|
||||||
|
if value_types == {"integer"}:
|
||||||
|
return "integer"
|
||||||
|
if value_types <= {"integer", "decimal"}:
|
||||||
|
return "decimal"
|
||||||
|
if value_types == {"text"}:
|
||||||
|
return "text"
|
||||||
|
return "mixed"
|
||||||
|
|
||||||
|
|
||||||
|
def _infer_single_value_type(value: str) -> str:
|
||||||
|
normalized = value.strip()
|
||||||
|
if _is_integer(normalized):
|
||||||
|
return "integer"
|
||||||
|
if _is_decimal(normalized):
|
||||||
|
return "decimal"
|
||||||
|
return "text"
|
||||||
|
|
||||||
|
|
||||||
|
def _is_integer(value: str) -> bool:
|
||||||
|
if value.startswith(("+", "-")):
|
||||||
|
value = value[1:]
|
||||||
|
return value.isdecimal()
|
||||||
|
|
||||||
|
|
||||||
|
def _is_decimal(value: str) -> bool:
|
||||||
|
if value.count(",") + value.count(".") != 1:
|
||||||
|
return False
|
||||||
|
separator = "," if "," in value else "."
|
||||||
|
left, right = value.split(separator, 1)
|
||||||
|
if left.startswith(("+", "-")):
|
||||||
|
left = left[1:]
|
||||||
|
return left.isdecimal() and right.isdecimal()
|
||||||
|
|
||||||
|
|
||||||
|
def _profile_article_numbers(
|
||||||
|
rows: list[dict[str, str]],
|
||||||
|
article_number_column: str,
|
||||||
|
) -> ArticleNumberProfile:
|
||||||
|
values = []
|
||||||
|
for row in rows:
|
||||||
|
value = row.get(article_number_column, "")
|
||||||
|
if not _is_empty(value):
|
||||||
|
values.append(value)
|
||||||
|
|
||||||
|
counts = Counter(values)
|
||||||
|
duplicates = tuple(
|
||||||
|
DuplicateArticleNumber(nr=nr, count=count)
|
||||||
|
for nr, count in sorted(counts.items())
|
||||||
|
if count > 1
|
||||||
|
)
|
||||||
|
return ArticleNumberProfile(
|
||||||
|
column_name=article_number_column,
|
||||||
|
distinct_count=len(counts),
|
||||||
|
duplicate_count=len(duplicates),
|
||||||
|
duplicates=duplicates,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def _compare_rollcalc_articles(
|
||||||
|
rollcalc_path: Path,
|
||||||
|
erp_article_numbers: set[str],
|
||||||
|
) -> RollCalcComparison:
|
||||||
|
rollcalc_articles = load_rollcalc_articles(Path(rollcalc_path))
|
||||||
|
rollcalc_numbers = tuple(article.nr for article in rollcalc_articles)
|
||||||
|
missing_article_numbers = tuple(nr for nr in rollcalc_numbers if nr not in erp_article_numbers)
|
||||||
|
return RollCalcComparison(
|
||||||
|
rollcalc_article_count=len(rollcalc_numbers),
|
||||||
|
found_count=len(rollcalc_numbers) - len(missing_article_numbers),
|
||||||
|
missing_count=len(missing_article_numbers),
|
||||||
|
missing_article_numbers=missing_article_numbers,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def _is_empty(value: str | None) -> bool:
|
||||||
|
return value is None or value == ""
|
||||||
@@ -0,0 +1,152 @@
|
|||||||
|
import json
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
from article_data_manager.tools.erp_explorer import (
|
||||||
|
infer_value_type,
|
||||||
|
profile_erp_csv,
|
||||||
|
render_erp_profile_report,
|
||||||
|
write_erp_profile_report,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def write_text(path: Path, content: str) -> Path:
|
||||||
|
path.write_text(content, encoding="utf-8")
|
||||||
|
return path
|
||||||
|
|
||||||
|
|
||||||
|
def write_rollcalc_json(path: Path, article_numbers: list[str]) -> Path:
|
||||||
|
payload = [
|
||||||
|
{
|
||||||
|
"nr": nr,
|
||||||
|
"name": f"Article {nr}",
|
||||||
|
"thickness": 1.0,
|
||||||
|
"area_weight": 0.0,
|
||||||
|
"core_type": 0.0,
|
||||||
|
}
|
||||||
|
for nr in article_numbers
|
||||||
|
]
|
||||||
|
path.write_text(json.dumps(payload), encoding="utf-8")
|
||||||
|
return path
|
||||||
|
|
||||||
|
|
||||||
|
def test_profiles_columns(tmp_path: Path) -> None:
|
||||||
|
csv_path = write_text(
|
||||||
|
tmp_path / "erp.csv",
|
||||||
|
"\n".join(
|
||||||
|
[
|
||||||
|
"SL_ITEM_NO,ROP_PRODUCT_WIDTH,COMMENT,EMPTY_COLUMN",
|
||||||
|
"214700,\"6,00\",Alpha,",
|
||||||
|
"000123,1.20,Beta,",
|
||||||
|
"777777,,Alpha,",
|
||||||
|
]
|
||||||
|
),
|
||||||
|
)
|
||||||
|
|
||||||
|
profile = profile_erp_csv(csv_path)
|
||||||
|
|
||||||
|
assert profile.file_name == "erp.csv"
|
||||||
|
assert profile.row_count == 3
|
||||||
|
assert profile.column_count == 4
|
||||||
|
columns = {column.name: column for column in profile.columns}
|
||||||
|
assert columns["ROP_PRODUCT_WIDTH"].filled_count == 2
|
||||||
|
assert columns["ROP_PRODUCT_WIDTH"].empty_count == 1
|
||||||
|
assert columns["ROP_PRODUCT_WIDTH"].distinct_count == 2
|
||||||
|
assert columns["ROP_PRODUCT_WIDTH"].inferred_type == "decimal"
|
||||||
|
assert columns["COMMENT"].distinct_count == 2
|
||||||
|
assert columns["COMMENT"].inferred_type == "text"
|
||||||
|
assert columns["EMPTY_COLUMN"].inferred_type == "empty"
|
||||||
|
|
||||||
|
|
||||||
|
def test_infers_value_types() -> None:
|
||||||
|
assert infer_value_type([]) == "empty"
|
||||||
|
assert infer_value_type(["1", "002", "-3"]) == "integer"
|
||||||
|
assert infer_value_type(["1", "2.5", "3,75"]) == "decimal"
|
||||||
|
assert infer_value_type(["Alpha", "Beta"]) == "text"
|
||||||
|
assert infer_value_type(["1", "Alpha"]) == "mixed"
|
||||||
|
|
||||||
|
|
||||||
|
def test_detects_duplicate_article_numbers(tmp_path: Path) -> None:
|
||||||
|
csv_path = write_text(
|
||||||
|
tmp_path / "erp.csv",
|
||||||
|
"\n".join(
|
||||||
|
[
|
||||||
|
"SL_ITEM_NO,NAME",
|
||||||
|
"00001,A",
|
||||||
|
"1,B",
|
||||||
|
"00001,C",
|
||||||
|
"214700,D",
|
||||||
|
"214700,E",
|
||||||
|
]
|
||||||
|
),
|
||||||
|
)
|
||||||
|
|
||||||
|
article_numbers = profile_erp_csv(csv_path).article_numbers
|
||||||
|
|
||||||
|
assert article_numbers.distinct_count == 3
|
||||||
|
assert article_numbers.duplicate_count == 2
|
||||||
|
assert [(duplicate.nr, duplicate.count) for duplicate in article_numbers.duplicates] == [
|
||||||
|
("00001", 2),
|
||||||
|
("214700", 2),
|
||||||
|
]
|
||||||
|
|
||||||
|
|
||||||
|
def test_compares_rollcalc_articles_when_path_is_provided(tmp_path: Path) -> None:
|
||||||
|
csv_path = write_text(
|
||||||
|
tmp_path / "erp.csv",
|
||||||
|
"\n".join(
|
||||||
|
[
|
||||||
|
"SL_ITEM_NO,NAME",
|
||||||
|
"214700,A",
|
||||||
|
"000123,B",
|
||||||
|
]
|
||||||
|
),
|
||||||
|
)
|
||||||
|
rollcalc_path = write_rollcalc_json(tmp_path / "article-data.json", ["214700", "999999"])
|
||||||
|
|
||||||
|
comparison = profile_erp_csv(csv_path, rollcalc_path=rollcalc_path).rollcalc_comparison
|
||||||
|
|
||||||
|
assert comparison is not None
|
||||||
|
assert comparison.rollcalc_article_count == 2
|
||||||
|
assert comparison.found_count == 1
|
||||||
|
assert comparison.missing_count == 1
|
||||||
|
assert comparison.missing_article_numbers == ("999999",)
|
||||||
|
|
||||||
|
|
||||||
|
def test_rollcalc_comparison_is_optional(tmp_path: Path) -> None:
|
||||||
|
csv_path = write_text(
|
||||||
|
tmp_path / "erp.csv",
|
||||||
|
"\n".join(
|
||||||
|
[
|
||||||
|
"SL_ITEM_NO,NAME",
|
||||||
|
"214700,A",
|
||||||
|
]
|
||||||
|
),
|
||||||
|
)
|
||||||
|
|
||||||
|
assert profile_erp_csv(csv_path).rollcalc_comparison is None
|
||||||
|
|
||||||
|
|
||||||
|
def test_renders_and_writes_report(tmp_path: Path) -> None:
|
||||||
|
csv_path = write_text(
|
||||||
|
tmp_path / "erp.csv",
|
||||||
|
"\n".join(
|
||||||
|
[
|
||||||
|
"SL_ITEM_NO,ROP_PRODUCT_WIDTH",
|
||||||
|
"214700,\"6,00\"",
|
||||||
|
"214700,\"6,00\"",
|
||||||
|
]
|
||||||
|
),
|
||||||
|
)
|
||||||
|
report_path = tmp_path / "reports" / "erp_profile_report.txt"
|
||||||
|
|
||||||
|
profile = write_erp_profile_report(csv_path, report_path=report_path)
|
||||||
|
report = report_path.read_text(encoding="utf-8")
|
||||||
|
|
||||||
|
assert report == render_erp_profile_report(profile)
|
||||||
|
assert "ERP Profile Report" in report
|
||||||
|
assert "File: erp.csv" in report
|
||||||
|
assert "Rows: 2" in report
|
||||||
|
assert "Columns: 2" in report
|
||||||
|
assert "Column: ROP_PRODUCT_WIDTH" in report
|
||||||
|
assert "Duplicate article numbers" in report
|
||||||
|
assert "214700\t2" in report
|
||||||
Reference in New Issue
Block a user