commit f017d187ebcb81b36b1e0f9732fc4ebe61e9ebba Author: Martin Tazl Date: Fri Sep 4 05:34:52 2026 +0200 Initial production analytics foundation diff --git a/.env.example b/.env.example new file mode 100644 index 0000000..c8681b0 --- /dev/null +++ b/.env.example @@ -0,0 +1,13 @@ +# Local-only secrets/configuration. Copy to .env; do not commit that file. +# The actual credential belongs in ignored secrets/enlyze.env, not this file. +# Public ENLYZE documentation verifies Bearer-token authentication using this key. +ENLYZE_BASE_URL= +ENLYZE_API_KEY= +ENLYZE_HTTP_TIMEOUT_SECONDS=20 + +# Future persistence configuration. +POSTGRES_HOST=localhost +POSTGRES_PORT=5432 +POSTGRES_DB=production_analytics +POSTGRES_USER=production_analytics +POSTGRES_PASSWORD= diff --git a/.gitignore b/.gitignore new file mode 100644 index 0000000..57347a8 --- /dev/null +++ b/.gitignore @@ -0,0 +1,29 @@ +# Python +__pycache__/ +*.py[cod] +*.egg-info/ +.pytest_cache/ +.ruff_cache/ +.coverage +htmlcov/ +.venv/ +venv/ + +# Local configuration and secrets +.env +.env.* +!.env.example + +# Unmodified API captures are local only. Commit reviewed sanitized fixtures +# from fixtures/enlyze/ instead. +data/raw/enlyze/ + +# Dedicated local credentials. The marker is safe to retain; all actual secret +# files, including secrets/enlyze.env, are ignored. +secrets/* +!secrets/.gitkeep + +# Editors and operating systems +.idea/ +.vscode/ +.DS_Store diff --git a/PROJECT_KNOWLEDGE.md b/PROJECT_KNOWLEDGE.md new file mode 100644 index 0000000..1652a46 --- /dev/null +++ b/PROJECT_KNOWLEDGE.md @@ -0,0 +1,49 @@ +# Project knowledge + +## Purpose + +This service is a calculation and persistence layer between ENLYZE and +Grafana. It must not become a second historian: raw process data stays in +ENLYZE and is retrieved again when historical calculations need reproduction. + +## Core architectural decisions + +- Python with a `src/` package layout. +- FastAPI is the preferred future service layer; no HTTP API is needed now. +- PostgreSQL with TimescaleDB is the target derived-data store. +- Grafana reads derived data from that database as an additional datasource. +- Secrets are environment variables only; no credentials or site-specific + configuration are committed. +- Calculations are modular, testable Python implementations. Configuration + declares instances; it is not a generic low-code language. + +## Central domain context + +Production orders connect machine, article/material, source interval, and +derived results. Metrics and events must retain calculation type/version and +source-time-range provenance. + +## Known versus unknown + +Known from the ENLYZE UI: machine identity, operational/downtime state, +current production order, article/material number, and process signals exist. +It is **not yet verified** how, or whether, each is exposed by the ENLYZE API. +Do not infer endpoint paths, authentication mechanisms, identifiers, paging, +timestamp semantics, or signal payloads. The active open-question list is in +[docs/enlyze-api.md](docs/enlyze-api.md). + +## Exploration workflow + +A deliberately generic, read-only CLI is available as +`production-analytics enlyze raw PATH`. It only makes GET requests, never +guesses endpoint schemas, and emits sanitized response metadata/body. The +operator must first obtain an authorized base URL, a verified safe path, and +the authentication method. Credentials reside in ignored `secrets/enlyze.env`; +the CLI parses simple assignments without sourcing/executing the file. Public +ENLYZE documentation verifies an `Authorization: Bearer ` +authentication header; its value is never printed. + +Unmodified captures are local in ignored `data/raw/enlyze/`. Fixtures written +to `fixtures/enlyze/` are sanitized, but must still be reviewed before commit. +The deliverable is verified API observations and sanitized fixtures, not +production calculations or database ingestion. diff --git a/README.md b/README.md new file mode 100644 index 0000000..42552cd --- /dev/null +++ b/README.md @@ -0,0 +1,69 @@ +# production-analytics + +`production-analytics` is a small Python service that derives production +analytics from ENLYZE data for Grafana. ENLYZE remains the authority for raw +process data; this project persists only derived metrics, events, relevant +production-order context, and calculation state. + +## Status + +This repository includes a read-only ENLYZE API exploration CLI. It contains +no verified ENLYZE operation wrappers, database migrations, or HTTP endpoints. + +## Intended flow + +```text +ENLYZE (raw data) -> retrieval boundary -> calculations -> TimescaleDB -> Grafana + \-> calculation provenance/state +``` + +Production orders are the primary attribution context for results. Every +persisted result will ultimately be traceable to its machine, order/article +context, source time range, and calculation implementation version. + +## Development + +Requires Python 3.11 or newer. Install the project and development tools: + +```bash +python3 -m pip install -e '.[dev]' +pytest +ruff check . +``` + +Copy `.env.example` to `.env` for public/local configuration documentation. +Put real ENLYZE credentials only in `secrets/enlyze.env`, which is ignored. +The CLI reads that file by default without executing it; it accepts only simple +`KEY=VALUE` lines (quoted values are supported). Environment variables may be +used instead when appropriate. Never pass keys as command-line arguments. + +The generic request command only performs `GET` requests and requires the +operator to provide an API path that is known to be safe and authorized: + +```bash +production-analytics enlyze raw /verified/path --pretty +production-analytics enlyze raw /verified/path --save-fixture response.json +``` + +Public ENLYZE documentation verifies Bearer-token authentication. Put +`ENLYZE_API_KEY='...'` in the local secret file; the exploration client sends +it as an `Authorization: Bearer …` header and never prints that header/value. + +The second command saves a sanitized, reviewable fixture under +`fixtures/enlyze/` by default. Raw captures belong in the ignored +`data/raw/enlyze/` directory and must not be committed. + +The OpenAPI server URL is `https://app.enlyze.com/api/`; its operation paths +begin with `/v2/`. Set `ENLYZE_BASE_URL` to that server URL, not to an +operation path. The documented read-only time-series operation is exposed for +exploration as `production-analytics enlyze timeseries`. + +The current Compose file has no application service, so it deliberately does +not pass ENLYZE credentials to TimescaleDB. A future application service should +use `env_file: ./secrets/enlyze.env` rather than copying secrets into Compose. + +`docker compose up -d timescaledb` is an optional local database design for a +future persistence milestone. It is not required for the bootstrap tests. + +See [PROJECT_KNOWLEDGE.md](PROJECT_KNOWLEDGE.md) for durable project context +and [docs/roadmap.md](docs/roadmap.md) for the implementation sequence. diff --git a/compose.yaml b/compose.yaml new file mode 100644 index 0000000..5832d02 --- /dev/null +++ b/compose.yaml @@ -0,0 +1,14 @@ +services: + timescaledb: + image: timescale/timescaledb:latest-pg16 + environment: + POSTGRES_DB: production_analytics + POSTGRES_USER: production_analytics + POSTGRES_PASSWORD: ${POSTGRES_PASSWORD:?Set POSTGRES_PASSWORD in .env} + ports: + - "${POSTGRES_PORT:-5432}:5432" + volumes: + - timescaledb_data:/var/lib/postgresql/data + +volumes: + timescaledb_data: diff --git a/config/calculations.example.yaml b/config/calculations.example.yaml new file mode 100644 index 0000000..04c85e5 --- /dev/null +++ b/config/calculations.example.yaml @@ -0,0 +1,21 @@ +# Illustrative only. Signal and machine identifiers are placeholders until the +# ENLYZE API exploration milestone verifies their identifiers and semantics. +calculations: + - id: example-electrical-energy + type: integration + version: "1" + machine_ref: machine-to-be-verified + signal_ref: electrical-power-to-be-verified + input_unit: kW + output_metric: electrical_energy + output_unit: kWh + group_by: production_order + + - id: example-cycle-peak + type: peak_detector + version: "1" + machine_ref: machine-to-be-verified + signal_ref: process-signal-to-be-verified + below_peak_fraction: 0.75 + below_peak_duration_seconds: 60 + output_event: cycle_peak diff --git a/docs/architecture.md b/docs/architecture.md new file mode 100644 index 0000000..116a226 --- /dev/null +++ b/docs/architecture.md @@ -0,0 +1,41 @@ +# Architecture + +## Scope boundary + +ENLYZE is authoritative for raw process data. This service retrieves raw data +only to calculate derived results and persists the results, their context, and +the minimum state needed for incremental calculation. It does not mirror raw +time series into PostgreSQL/TimescaleDB. + +## Components + +| Component | Responsibility | Must not do | +| --- | --- | --- | +| `enlyze` | Isolate ENLYZE retrieval and API-specific mapping | Leak assumed API schemas into calculations | +| `domain` | Small, stable models for orders, provenance, metrics, and events | Encode database or HTTP details | +| `calculations` | Versioned, testable operators such as integration and peak detection | Fetch data or write directly to a database | +| `persistence` | Store results and calculation state through repository ports | Become a raw-data historian | +| `cli` | Operator commands; first use is API exploration | Contain business calculations | +| `service` | Future FastAPI composition/API layer | Require endpoints during bootstrap | + +## Result provenance + +Persisted metrics/events need machine and production-order context where +available, article context, observed/result time or interval, calculation +type/version, and the raw ENLYZE source interval used. A calculation run +records operational metadata that ties a batch of results to its implementation +version and source range. + +## Calculation instances + +Configuration declares a named calculation instance with a calculation type, +version, and signal/context references. The engine selects a tested Python +operator implementation for that type. This keeps configuration simple and +avoids a generic expression language. + +## Incremental execution + +Future calculation state is stored per calculation instance and relevant +context partition. It enables safe continuation (for example an open peak +cycle), while the source interval recorded on results keeps a historical run +reproducible by fetching ENLYZE data again. diff --git a/docs/data-model.md b/docs/data-model.md new file mode 100644 index 0000000..7906ec4 --- /dev/null +++ b/docs/data-model.md @@ -0,0 +1,27 @@ +# Data model + +The entities below describe intended persistence and domain boundaries; this +bootstrap does not define a database schema or migrations. + +| Entity | Intended contents | Purpose | +| --- | --- | --- | +| `ProductionOrder` | source ID, machine ID, article/material reference, known interval | Central result-attribution context | +| `MachineState` | machine ID, state, observed interval | Runtime/downtime and contextual calculations | +| `CalculatedMetric` | name, value/unit, result interval, attribution, provenance | Integrated consumption and future aggregates | +| `ProcessEvent` | event type, time, payload/value, attribution, provenance | Completed peaks and threshold/cycle events | +| `CalculationRun` | type/version, execution time, source interval, status/metadata | Auditing and reproducibility | +| calculation state | calculation-instance/context key and serializable state | Incremental execution only | + +## Provenance minimum + +Each metric/event must store `calculation_type`, `calculation_version`, and +the ENLYZE source time range. Values should retain their unit. Production order +and article can be absent when ENLYZE cannot provide them for a particular +record; absence is explicit rather than fabricated. + +## Database direction + +TimescaleDB will store result/event time series and supporting relational +context. Schema design comes after the ENLYZE exploration milestone confirms +source identifiers, timestamp precision/timezone, ordering, correction, and +context-history behavior. diff --git a/docs/enlyze-api.md b/docs/enlyze-api.md new file mode 100644 index 0000000..d528864 --- /dev/null +++ b/docs/enlyze-api.md @@ -0,0 +1,132 @@ +# ENLYZE API investigation + +This document separates public facts, schema facts, and customer-specific observations. Raw captures are local under ignored `data/raw/enlyze/`; reviewed sanitized fixtures may be stored in `fixtures/enlyze/`. + +## VERIFIED FROM PUBLIC DOCUMENTATION + +- ENLYZE uses Bearer-token authentication with `ENLYZE_API_KEY`. +- Public resources include machines, sites, products, production runs, and downtimes. + +## VERIFIED FROM OPENAPI SCHEMA + +Source: locally supplied `data/raw/enlyze/openapi.json`, verified by the operator as OpenAPI 3.1.0, `ENLYZE API` v2. The schema server is the relative URL `/api`; resolving it against `https://app.enlyze.com` and combining it with paths such as `/v2/machines` yields `https://app.enlyze.com/api/v2/machines`. + +Every operation below declares `Bearer` security: an `apiKey` in the `Authorization` header. The exploration CLI sends `Authorization: Bearer ` without logging it. + +### Common pagination + +Paginated list/time-series responses contain `data` and `metadata`. The required `metadata.next_cursor` is a string continuation token or null. Reuse a non-null value as `cursor` with otherwise identical parameters. No general `limit`/page-size parameter is declared. UUID-array filters permit at most 50 items; one time-series request permits at most 100 variables. + +### Machines and sites + +| Operation | Parameters | Response | +| --- | --- | --- | +| `GET /v2/machines` | Optional `site` UUID array (max 50), `cursor` | `PaginatedMachines` | +| `GET /v2/machines/{uuid}` | Required machine UUID | `Machine` | +| `GET /v2/sites` | Optional `cursor` | `PaginatedSites` | +| `GET /v2/sites/{uuid}` | Required site UUID | `Site` | + +`Machine` requires UUID `uuid`, `name`, site UUID `site`, and date-only `genesis_date`. `Site` requires UUID `uuid` and `name`. No current machine-running state is exposed by these schemas. + +### Products and production runs + +| Operation | Parameters | Response | +| --- | --- | --- | +| `GET /v2/products` | Optional `cursor` | `PaginatedProducts` | +| `GET /v2/products/{uuid}` | Required product UUID | `Product` | +| `GET /v2/production-runs` | `cursor`; optional `machine` UUID array (max 50), `product-external-id`, `product` UUID, `production-order-external-id`, `start`, `end`, `uuid` UUID array (max 50), `q` | `PaginatedProductionRuns` | +| `GET /v2/production-runs/{uuid}` | Required production-run UUID | `ProductionRun` | + +`Product` requires UUID `uuid`, ERP/MES identifier `external_id`, and nullable description/name `name`. + +`ProductionRun` requires UUID `uuid`, machine UUID `machine`, string `production_order` (the schema calls this the identifier of the production order), product UUID `product`, timezone-aware ISO 8601 `start`, and nullable timezone-aware ISO 8601 `end`. A null end denotes an unfinished run. It also contains quantities (`{value, unit}` when present), OEE components (score 0–1 and seconds of time loss), optional maximum run speed with unit and observation period, data-quality metrics, and arbitrary attributes. + +For list filtering, `start` means runs executed starting at/after the supplied time; `end` means runs executed up to the supplied time. This does not prove interval-overlap behavior needed for integration. + +### Variables and data sources + +| Operation | Parameters | Response | +| --- | --- | --- | +| `GET /v2/variables` | Optional `machine`, `type`, `state`, `data_source`, `data_type`, `uuid`, `depends_on` UUID arrays (each max 50 where applicable), `cursor` | `PaginatedVariables` | +| `GET /v2/data-sources` | Optional `cursor`, `uuid`, `machine`, `spark` UUID arrays (max 50) | `PaginatedDataSources` | +| `GET /v2/data-sources/{uuid}` | Required data-source UUID | `DataSource` | + +`VariableResponse` requires variable UUID, associated machine UUID, data type, and discriminated details. It can include `display_name`, nullable string `unit`, nullable `scaling_factor`, comment, and types `FLOAT`, `INTEGER`, `STRING`, `BOOLEAN`, or array variants. Spark details include data-source UUID, structured origin identifier, origin name/comment, and current capture state (`inactive`, `active`, `evaluating`). Derived details list dependencies and description. The deprecated top-level `type` mirrors `details.type` (`spark` or `derived`). + +`DataSource` can provide `readout_limit`, `readout_interval`, current `readout_status`, and machine UUID. This is a data-source/readout state, not a documented machine operational/downtime state. + +### Historical time series + +| Operation | Parameters | Response | +| --- | --- | --- | +| `POST /v2/timeseries` | Required JSON `machine`, timezone-aware ISO 8601 `start`, `end`, and 1–100 `variables`; optional `resampling_interval`, `cursor` | `TimeseriesResponse` | + +Each requested variable contains required UUID and optional `resampling_method`: `first`, `last`, `max`, `min`, `count`, `sum`, `avg`, `median`, `std`, `q5`, `q25`, `q75`, `q95`. `resampling_interval` is an optional integer from 10 to 604800; the schema does not define its unit or whether omission means native/raw resolution. + +`TimeseriesResponse.data.columns` is ordered with `time` always first, followed by requested variable UUIDs. `data.records` is tabular in that column order. The schema example uses timestamps such as `2023-01-03T14:00:00.0+00:00` and numeric values. It defines no per-sample unit, quality/status field, null/missing-value behavior, boundary inclusion, ordering, or correction/backfill semantics. Units belong to `VariableResponse.unit`, not the time-series envelope. + +### Downtimes + +| Operation | Parameters | Response | +| --- | --- | --- | +| `GET /v2/downtimes` | Optional UUID/machine UUID arrays (max 50), timezone-aware ISO 8601 `start`, `end`, reason filters, reason state, `cursor` | `PaginatedDowntimes` | + +`Downtime` requires UUID, machine UUID, type (`THRESHOLD` or `NO_DATA`), timezone-aware ISO 8601 start, nullable end, and nullable comment/reason/update information. A null end denotes an ongoing downtime. It supports historical downtime retrieval but does not establish a complete machine-state timeline. + +## VERIFIED WITH LIVE CUSTOMER API + +None. This execution environment has no ENLYZE DNS/network access, so no customer operation has been sent in this milestone. + +## STILL UNKNOWN + +- Customer response values, permissions, authentication success/failure behavior, default page size, and rate/retention limits. +- Whether production-run filters have the desired interval-overlap behavior; completed-run `start`/`end` are candidate integration boundaries but require live confirmation. +- Which customer variable is a consumption signal and whether its unit/scaling factor makes it usable for integration. +- Whether omitted `resampling_interval` yields native/raw resolution; its unit and exact temporal meaning of resampling methods. +- Sample ordering, boundary inclusion, native sampling behavior, gaps, nulls, quality/bad samples, late arrivals, corrections, and backfills. +- A complete historical machine operational-state resource beyond downtimes. + +## Exact bounded manual exploration sequence + +Set the schema server URL locally; the key remains only in `secrets/enlyze.env`: + +```bash +export ENLYZE_BASE_URL='https://app.enlyze.com/api/' +``` + +1. First page of machines (the schema has no limit parameter): + +```bash +production-analytics enlyze raw /v2/machines --pretty \ + --save-raw machines-page-1.raw.json --save-fixture machines-page-1.json +``` + +2. Select a harmless machine UUID from the ignored raw capture and retrieve a narrow completed-run page. Replace uppercase placeholders with local values: + +```bash +production-analytics enlyze raw /v2/production-runs \ + --query 'machine=MACHINE_UUID' \ + --query 'start=RUN_SEARCH_START_ISO8601' \ + --query 'end=RUN_SEARCH_END_ISO8601' --pretty \ + --save-raw production-runs.raw.json --save-fixture production-runs.json +``` + +3. Resolve the selected run’s product and discover variables for its machine: + +```bash +production-analytics enlyze raw /v2/products/PRODUCT_UUID --pretty \ + --save-raw product.raw.json --save-fixture product.json +production-analytics enlyze raw /v2/variables --query 'machine=MACHINE_UUID' --pretty \ + --save-raw variables.raw.json --save-fixture variables.json +``` + +4. Select a numeric variable with a meaningful unit and query a short completed-run subinterval. Omit resampling initially to test the optional/default behavior: + +```bash +production-analytics enlyze timeseries \ + --machine MACHINE_UUID --variable VARIABLE_UUID \ + --start SAMPLE_START_ISO8601 --end SAMPLE_END_ISO8601 --pretty \ + --save-raw timeseries.raw.json --save-fixture timeseries.json +``` + +`--resampling-interval 10` and `--resampling-method avg` are available only for a subsequent bounded comparison. The CLI reads `secrets/enlyze.env` by default and does not print the token/header. Raw files remain ignored; review every sanitized fixture before committing it. diff --git a/docs/roadmap.md b/docs/roadmap.md new file mode 100644 index 0000000..df6e7a9 --- /dev/null +++ b/docs/roadmap.md @@ -0,0 +1,39 @@ +# Roadmap + +## 0. Bootstrap (this milestone) + +Establish package boundaries, architecture documentation, example calculation +configuration, basic domain/protocol models, tests, and an optional local +TimescaleDB Compose design. + +## 1. ENLYZE API exploration client/CLI (recommended exact next milestone) + +A read-only, GET-only raw inspection CLI and fixture sanitizer are implemented. +The remaining work is live, authorized exploration of machines, signal catalog, +bounded samples, state history, production-order history, and article/material +context. Record sanitized fixtures and update `docs/enlyze-api.md` with +evidence. Do not implement ingestion or a database schema before this is +complete. + +## 2. Persistence foundation + +Design and migrate a TimescaleDB schema for derived metrics/events, calculation +runs, context, and incremental state. Add Grafana-oriented query examples. + +## 3. First vertical slice + +Implement one configured integration calculation over a verified signal and +production-order context. Include backfill, rerun/provenance behavior, and +tests against sanitized fixtures. + +## 4. Peak detection + +Implement the configurable sawtooth peak detector: retain the current maximum; +close a cycle after a configurable below-peak fraction remains true for a +configurable duration; emit one completed peak event; preserve open-cycle +state. + +## Later + +State/cycle durations, aggregates by order, rolling statistics, threshold +events, derivatives, and specific-consumption calculations. diff --git a/fixtures/enlyze/README.md b/fixtures/enlyze/README.md new file mode 100644 index 0000000..2426b4b --- /dev/null +++ b/fixtures/enlyze/README.md @@ -0,0 +1,9 @@ +# Sanitized ENLYZE fixtures + +This directory may contain small, reviewed, sanitized JSON responses used for +development and tests. It currently contains no live API capture. + +Never copy unmodified ENLYZE responses here. Keep transient raw captures in +the ignored `data/raw/enlyze/` directory. The CLI sanitizer preserves shape, +timestamps, units, numeric values, status fields, and pagination structure, +but every fixture still requires human review before commit. diff --git a/pyproject.toml b/pyproject.toml new file mode 100644 index 0000000..57fabe9 --- /dev/null +++ b/pyproject.toml @@ -0,0 +1,36 @@ +[build-system] +requires = ["hatchling>=1.25"] +build-backend = "hatchling.build" + +[project] +name = "production-analytics" +version = "0.1.0" +description = "Derived production analytics between ENLYZE and Grafana." +readme = "README.md" +requires-python = ">=3.11" +license = { text = "Proprietary" } +authors = [{ name = "Production Analytics Team" }] +dependencies = [] + +[project.optional-dependencies] +dev = [ + "pytest>=8.0", + "ruff>=0.6", +] + +[project.scripts] +production-analytics = "production_analytics.cli.__main__:main" + +[tool.hatch.build.targets.wheel] +packages = ["src/production_analytics"] + +[tool.pytest.ini_options] +testpaths = ["tests"] +addopts = "-ra" + +[tool.ruff] +target-version = "py311" +line-length = 100 + +[tool.ruff.lint] +select = ["E", "F", "I", "UP"] diff --git a/secrets/.gitkeep b/secrets/.gitkeep new file mode 100644 index 0000000..e69de29 diff --git a/src/production_analytics/__init__.py b/src/production_analytics/__init__.py new file mode 100644 index 0000000..1be50df --- /dev/null +++ b/src/production_analytics/__init__.py @@ -0,0 +1 @@ +"""Derived production analytics between ENLYZE and Grafana.""" diff --git a/src/production_analytics/calculations/__init__.py b/src/production_analytics/calculations/__init__.py new file mode 100644 index 0000000..26c27b2 --- /dev/null +++ b/src/production_analytics/calculations/__init__.py @@ -0,0 +1,5 @@ +"""Versioned, pure calculation operators and their contracts.""" + +from .base import Calculation + +__all__ = ["Calculation"] diff --git a/src/production_analytics/calculations/base.py b/src/production_analytics/calculations/base.py new file mode 100644 index 0000000..8a7e3bd --- /dev/null +++ b/src/production_analytics/calculations/base.py @@ -0,0 +1,15 @@ +"""Calculation contracts; concrete operators follow verified source semantics.""" + +from collections.abc import Iterable +from typing import Protocol + +from production_analytics.domain import CalculatedMetric, ProcessEvent + + +class Calculation(Protocol): + """A testable operator that emits derived results from supplied input.""" + + calculation_type: str + version: str + + def calculate(self, samples: Iterable[object]) -> Iterable[CalculatedMetric | ProcessEvent]: ... diff --git a/src/production_analytics/cli/__init__.py b/src/production_analytics/cli/__init__.py new file mode 100644 index 0000000..56885f4 --- /dev/null +++ b/src/production_analytics/cli/__init__.py @@ -0,0 +1 @@ +"""Command-line entry points; API exploration is the first planned command.""" diff --git a/src/production_analytics/cli/__main__.py b/src/production_analytics/cli/__main__.py new file mode 100644 index 0000000..8c01de4 --- /dev/null +++ b/src/production_analytics/cli/__main__.py @@ -0,0 +1,147 @@ +"""Command-line entry point for safe, read-only API exploration.""" + +from __future__ import annotations + +import argparse +import json +import os +import sys +from collections.abc import Sequence +from pathlib import Path + +from production_analytics.enlyze.exploration import ( + ExplorationClient, + ExplorationError, + ExplorationSettings, + load_secret_file, +) +from production_analytics.enlyze.sanitize import sanitize + + +def _query_item(value: str) -> tuple[str, str]: + if "=" not in value: + raise argparse.ArgumentTypeError("query parameters must use KEY=VALUE format") + key, item_value = value.split("=", 1) + if not key: + raise argparse.ArgumentTypeError("query parameter key must not be empty") + return key, item_value + + +def _parser() -> argparse.ArgumentParser: + parser = argparse.ArgumentParser(prog="production-analytics") + namespaces = parser.add_subparsers(dest="namespace", required=True) + enlyze = namespaces.add_parser("enlyze", help="Read-only ENLYZE API exploration") + commands = enlyze.add_subparsers(dest="command", required=True) + raw = commands.add_parser("raw", help="GET an operator-verified relative API path") + raw.add_argument("path", help="Relative API path beginning with '/'") + raw.add_argument("--query", action="append", type=_query_item, default=[], metavar="KEY=VALUE") + raw.add_argument("--pretty", action="store_true", help="Pretty-print sanitized JSON") + raw.add_argument("--verbose", action="store_true", help="Show safe request progress on stderr") + raw.add_argument("--save-fixture", metavar="NAME", help="Save sanitized JSON under fixtures/enlyze/") + raw.add_argument("--save-raw", metavar="NAME", help="Save unmodified response under ignored data/raw/enlyze/") + raw.add_argument("--fixture-dir", type=Path, default=Path("fixtures/enlyze"), help=argparse.SUPPRESS) + raw.add_argument("--raw-dir", type=Path, default=Path("data/raw/enlyze"), help=argparse.SUPPRESS) + raw.add_argument( + "--secrets-file", + type=Path, + default=Path("secrets/enlyze.env"), + help="Local dotenv-style secret file (default: secrets/enlyze.env)", + ) + timeseries = commands.add_parser("timeseries", help="Read-only POST /v2/timeseries exploration") + timeseries.add_argument("--machine", required=True, help="Machine UUID") + timeseries.add_argument("--start", required=True, help="ISO 8601 datetime with timezone") + timeseries.add_argument("--end", required=True, help="ISO 8601 datetime with timezone") + timeseries.add_argument("--variable", required=True, help="Variable UUID") + timeseries.add_argument("--resampling-interval", type=int, help="Seconds; schema range is 10..604800") + timeseries.add_argument( + "--resampling-method", + choices=["first", "last", "max", "min", "count", "sum", "avg", "median", "std", "q5", "q25", "q75", "q95"], + help="Optional schema-defined method for this variable", + ) + timeseries.add_argument("--pretty", action="store_true", help="Pretty-print sanitized JSON") + timeseries.add_argument("--verbose", action="store_true", help="Show safe request progress on stderr") + timeseries.add_argument("--save-fixture", metavar="NAME", help="Save sanitized JSON under fixtures/enlyze/") + timeseries.add_argument("--save-raw", metavar="NAME", help="Save unmodified response under ignored data/raw/enlyze/") + timeseries.add_argument("--fixture-dir", type=Path, default=Path("fixtures/enlyze"), help=argparse.SUPPRESS) + timeseries.add_argument("--raw-dir", type=Path, default=Path("data/raw/enlyze"), help=argparse.SUPPRESS) + timeseries.add_argument( + "--secrets-file", + type=Path, + default=Path("secrets/enlyze.env"), + help="Local dotenv-style secret file (default: secrets/enlyze.env)", + ) + return parser + + +def _write_fixture(directory: Path, name: str, payload: object) -> Path: + candidate = Path(name) + if candidate.name != name or candidate.suffix.lower() != ".json": + raise ExplorationError("Fixture name must be a plain filename ending in .json.") + directory.mkdir(parents=True, exist_ok=True) + destination = directory / candidate + destination.write_text(json.dumps(payload, indent=2, sort_keys=True) + "\n", encoding="utf-8") + return destination + + +def main(argv: Sequence[str] | None = None) -> int: + args = _parser().parse_args(argv) + if args.namespace != "enlyze" or args.command not in {"raw", "timeseries"}: + return 2 + + try: + environment = dict(os.environ) + environment.update(load_secret_file(args.secrets_file)) + settings = ExplorationSettings.from_environment(environment) + if args.verbose: + authentication = "configured" if settings.api_key else "not configured" + operation = "GET" if args.command == "raw" else "POST" + target = args.path if args.command == "raw" else "/v2/timeseries" + print( + f"Requesting {operation} {target} (timeout={settings.timeout_seconds:g}s; authentication {authentication}).", + file=sys.stderr, + ) + client = ExplorationClient(settings) + if args.command == "raw": + response = client.get(args.path, dict(args.query)) + request = {"method": "GET", "path": response.path} + else: + if args.resampling_interval is not None and not 10 <= args.resampling_interval <= 604800: + raise ExplorationError("--resampling-interval must be between 10 and 604800 seconds.") + variable: dict[str, str] = {"uuid": args.variable} + if args.resampling_method: + variable["resampling_method"] = args.resampling_method + request_body: dict[str, object] = { + "machine": args.machine, + "start": args.start, + "end": args.end, + "variables": [variable], + } + if args.resampling_interval is not None: + request_body["resampling_interval"] = args.resampling_interval + response = client.post_json("/v2/timeseries", request_body) + request = {"method": "POST", "path": response.path, "body": request_body} + if args.verbose: + print(f"Received HTTP {response.status_code} for {response.path}.", file=sys.stderr) + raw_payload = { + "request": request, + "response": { + "status_code": response.status_code, + "headers": response.headers, + "body": response.body, + }, + } + payload = sanitize(raw_payload) + indent = 2 if args.pretty or args.save_fixture else None + print(json.dumps(payload, indent=indent, sort_keys=True)) + if args.save_fixture: + print(f"Sanitized fixture written to {_write_fixture(args.fixture_dir, args.save_fixture, payload)}") + if args.save_raw: + print(f"Raw response written to {_write_fixture(args.raw_dir, args.save_raw, raw_payload)}") + except ExplorationError as error: + print(f"ENLYZE exploration error: {error}") + return 1 + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/src/production_analytics/domain/__init__.py b/src/production_analytics/domain/__init__.py new file mode 100644 index 0000000..5a4cea6 --- /dev/null +++ b/src/production_analytics/domain/__init__.py @@ -0,0 +1,19 @@ +"""Stable domain models independent of API and database implementations.""" + +from .models import ( + CalculatedMetric, + CalculationRun, + MachineState, + ProcessEvent, + ProductionOrder, + SourceRange, +) + +__all__ = [ + "CalculatedMetric", + "CalculationRun", + "MachineState", + "ProcessEvent", + "ProductionOrder", + "SourceRange", +] diff --git a/src/production_analytics/domain/models.py b/src/production_analytics/domain/models.py new file mode 100644 index 0000000..b607ce0 --- /dev/null +++ b/src/production_analytics/domain/models.py @@ -0,0 +1,72 @@ +"""Small shared models for derived-result attribution and provenance.""" + +from dataclasses import dataclass +from datetime import datetime +from enum import StrEnum +from typing import Mapping + + +@dataclass(frozen=True, slots=True) +class SourceRange: + """Inclusive/exclusive raw-data interval used by a calculation.""" + + start: datetime + end: datetime + + def __post_init__(self) -> None: + if self.end < self.start: + raise ValueError("source range end must not precede start") + + +@dataclass(frozen=True, slots=True) +class ProductionOrder: + source_id: str + machine_id: str + article_number: str | None = None + start: datetime | None = None + end: datetime | None = None + + +class MachineState(StrEnum): + OPERATING = "operating" + DOWNTIME = "downtime" + UNKNOWN = "unknown" + + +@dataclass(frozen=True, slots=True) +class CalculatedMetric: + name: str + value: float + unit: str + machine_id: str + observed_at: datetime + source_range: SourceRange + calculation_type: str + calculation_version: str + production_order_id: str | None = None + article_number: str | None = None + + +@dataclass(frozen=True, slots=True) +class ProcessEvent: + event_type: str + occurred_at: datetime + machine_id: str + source_range: SourceRange + calculation_type: str + calculation_version: str + value: float | None = None + unit: str | None = None + production_order_id: str | None = None + article_number: str | None = None + attributes: Mapping[str, str | float | int | bool | None] | None = None + + +@dataclass(frozen=True, slots=True) +class CalculationRun: + calculation_id: str + calculation_type: str + calculation_version: str + source_range: SourceRange + started_at: datetime + completed_at: datetime | None = None diff --git a/src/production_analytics/enlyze/__init__.py b/src/production_analytics/enlyze/__init__.py new file mode 100644 index 0000000..3c2e2a0 --- /dev/null +++ b/src/production_analytics/enlyze/__init__.py @@ -0,0 +1,6 @@ +"""ENLYZE adapter boundary; concrete API behavior awaits verification.""" + +from .exploration import ExplorationClient, ExplorationSettings, load_secret_file +from .ports import EnlyzeGateway + +__all__ = ["EnlyzeGateway", "ExplorationClient", "ExplorationSettings", "load_secret_file"] diff --git a/src/production_analytics/enlyze/exploration.py b/src/production_analytics/enlyze/exploration.py new file mode 100644 index 0000000..a7808cc --- /dev/null +++ b/src/production_analytics/enlyze/exploration.py @@ -0,0 +1,165 @@ +"""Small, read-only HTTP boundary for observing an unverified ENLYZE API.""" + +from __future__ import annotations + +import json +import os +import shlex +from dataclasses import dataclass +from typing import Any, Mapping +from urllib.error import HTTPError, URLError +from urllib.parse import urlencode, urljoin, urlsplit +from urllib.request import Request, urlopen + + +class ExplorationError(RuntimeError): + """Base error whose message is safe to show to an operator.""" + + +class ConfigurationError(ExplorationError): + """Raised when local exploration configuration is incomplete or unsafe.""" + + +class AuthenticationError(ExplorationError): + """Raised for HTTP 401/403 without exposing credentials or response bodies.""" + + +class HttpResponseError(ExplorationError): + """Raised for a non-successful HTTP response.""" + + +class NonJsonResponseError(ExplorationError): + """Raised when a successful response cannot be decoded as JSON.""" + + +@dataclass(frozen=True, slots=True) +class ExplorationSettings: + """Local HTTP settings for documented ENLYZE Bearer-token access.""" + + base_url: str + timeout_seconds: float = 20.0 + api_key: str | None = None + + @classmethod + def from_environment(cls, environment: Mapping[str, str] | None = None) -> ExplorationSettings: + environment = environment or os.environ + base_url = environment.get("ENLYZE_BASE_URL", "").strip() + api_key = environment.get("ENLYZE_API_KEY", "").strip() or None + timeout_value = environment.get("ENLYZE_HTTP_TIMEOUT_SECONDS", "20") + + if not base_url: + raise ConfigurationError("ENLYZE_BASE_URL must be set for API exploration.") + try: + timeout_seconds = float(timeout_value) + except ValueError as error: + raise ConfigurationError("ENLYZE_HTTP_TIMEOUT_SECONDS must be a number.") from error + if timeout_seconds <= 0: + raise ConfigurationError("ENLYZE_HTTP_TIMEOUT_SECONDS must be greater than zero.") + return cls(base_url.rstrip("/") + "/", timeout_seconds, api_key) + + +@dataclass(frozen=True, slots=True) +class ExplorationResponse: + """JSON response plus safe metadata useful for API investigation.""" + + status_code: int + path: str + headers: Mapping[str, str] + body: Any + + +class ExplorationClient: + """Client for documented, read-only exploration operations.""" + + def __init__(self, settings: ExplorationSettings) -> None: + self._settings = settings + + def get(self, path: str, query: Mapping[str, str] | None = None) -> ExplorationResponse: + """Request one relative path and decode its JSON response.""" + return self._request_json("GET", path, query=query) + + def post_json(self, path: str, payload: Mapping[str, Any]) -> ExplorationResponse: + """Issue a documented read-only POST operation with a JSON body.""" + return self._request_json("POST", path, payload=payload) + + def _request_json( + self, + method: str, + path: str, + *, + query: Mapping[str, str] | None = None, + payload: Mapping[str, Any] | None = None, + ) -> ExplorationResponse: + parsed_path = urlsplit(path) + if ( + not path.startswith("/") + or path.startswith("//") + or parsed_path.scheme + or parsed_path.netloc + or ".." in parsed_path.path.split("/") + ): + raise ConfigurationError("PATH must be a relative path beginning with one slash.") + + encoded_query = urlencode(query or {}) + request_path = f"{path}?{encoded_query}" if encoded_query else path + request_url = urljoin(self._settings.base_url, request_path.lstrip("/")) + request_headers = {"Accept": "application/json"} + request_body = None + if payload is not None: + request_headers["Content-Type"] = "application/json" + request_body = json.dumps(payload).encode("utf-8") + if self._settings.api_key: + request_headers["Authorization"] = f"Bearer {self._settings.api_key}" + + request = Request(request_url, data=request_body, headers=request_headers, method=method) + try: + with urlopen(request, timeout=self._settings.timeout_seconds) as response: # noqa: S310 + raw_body = response.read() + status_code = response.status + response_headers = dict(response.headers.items()) + except HTTPError as error: + if error.code in {401, 403}: + raise AuthenticationError(f"Authentication or authorization failed (HTTP {error.code}).") from error + raise HttpResponseError(f"HTTP request failed with status {error.code}.") from error + except URLError as error: + raise HttpResponseError("HTTP request could not be completed.") from error + + try: + body = json.loads(raw_body) + except (UnicodeDecodeError, json.JSONDecodeError) as error: + content_type = response_headers.get("Content-Type", "unknown") + raise NonJsonResponseError( + f"Expected a JSON response; received Content-Type {content_type!r}." + ) from error + + return ExplorationResponse(status_code, request_path, response_headers, body) + + +def load_secret_file(path: str | os.PathLike[str]) -> dict[str, str]: + """Read simple dotenv-style assignments without executing local secret content.""" + secrets: dict[str, str] = {} + try: + lines = open(path, encoding="utf-8") + except FileNotFoundError: + return secrets + with lines: + for line_number, line in enumerate(lines, start=1): + stripped = line.strip() + if not stripped or stripped.startswith("#"): + continue + if stripped.startswith("export "): + stripped = stripped.removeprefix("export ").lstrip() + if "=" not in stripped: + raise ConfigurationError(f"Invalid secret-file assignment on line {line_number}.") + key, raw_value = stripped.split("=", 1) + key = key.strip() + if not key.isidentifier(): + raise ConfigurationError(f"Invalid secret-file variable name on line {line_number}.") + try: + parsed = shlex.split(raw_value, comments=True, posix=True) + except ValueError as error: + raise ConfigurationError(f"Invalid secret-file value on line {line_number}.") from error + if len(parsed) > 1: + raise ConfigurationError(f"Invalid secret-file value on line {line_number}.") + secrets[key] = parsed[0] if parsed else "" + return secrets diff --git a/src/production_analytics/enlyze/ports.py b/src/production_analytics/enlyze/ports.py new file mode 100644 index 0000000..d87f167 --- /dev/null +++ b/src/production_analytics/enlyze/ports.py @@ -0,0 +1,7 @@ +"""Ports intentionally avoid unverified ENLYZE endpoint or payload assumptions.""" + +from typing import Protocol + + +class EnlyzeGateway(Protocol): + """Marker for a verified ENLYZE adapter, defined after API exploration.""" diff --git a/src/production_analytics/enlyze/sanitize.py b/src/production_analytics/enlyze/sanitize.py new file mode 100644 index 0000000..f203b78 --- /dev/null +++ b/src/production_analytics/enlyze/sanitize.py @@ -0,0 +1,62 @@ +"""Conservative sanitization for reviewable API fixtures and CLI output.""" + +from __future__ import annotations + +import re +from collections.abc import Mapping +from dataclasses import dataclass, field +from typing import Any +from urllib.parse import urlsplit, urlunsplit + +_SECRET_KEY = re.compile( + r"(?:authorization|token|api[_-]?key|password|secret|cookie|credential|session)", re.IGNORECASE +) +_IDENTIFIER_KEY = re.compile(r"(?:^|[_-])(id|uuid|guid)(?:$|[_-])", re.IGNORECASE) +_SENSITIVE_TEXT_KEY = re.compile( + r"(?:email|phone|first[_-]?name|last[_-]?name|full[_-]?name|user[_-]?name|customer|site|tenant|name)", + re.IGNORECASE, +) +_HOST_KEY = re.compile(r"(?:host|hostname|base[_-]?url|url|uri|endpoint)", re.IGNORECASE) + + +@dataclass +class _SanitizationContext: + identifiers: dict[tuple[str, str], str | int] = field(default_factory=dict) + + def replacement_identifier(self, key: str, value: object) -> str | int: + original = str(value) + identity = (key.lower(), original) + if identity not in self.identifiers: + number = len(self.identifiers) + 1 + self.identifiers[identity] = 1000 + number if isinstance(value, int) else f"redacted-{key}-{number}" + return self.identifiers[identity] + + +def sanitize(value: Any) -> Any: + """Return a structurally equivalent value with common sensitive fields replaced.""" + return _sanitize(value, _SanitizationContext()) + + +def _sanitize(value: Any, context: _SanitizationContext(), key: str | None = None) -> Any: + if isinstance(value, Mapping): + return {str(item_key): _sanitize(item_value, context, str(item_key)) for item_key, item_value in value.items()} + if isinstance(value, list): + return [_sanitize(item, context, key) for item in value] + if isinstance(value, tuple): + return [_sanitize(item, context, key) for item in value] + if key and _SECRET_KEY.search(key): + return "" + if key and _IDENTIFIER_KEY.search(key) and isinstance(value, (str, int)) and not isinstance(value, bool): + return context.replacement_identifier(key, value) + if key and _HOST_KEY.search(key) and isinstance(value, str): + return _redact_host(value) + if key and _SENSITIVE_TEXT_KEY.search(key) and isinstance(value, str): + return f"" + return value + + +def _redact_host(value: str) -> str: + parsed = urlsplit(value) + if parsed.scheme and parsed.netloc: + return urlunsplit((parsed.scheme, "redacted-host.invalid", parsed.path, parsed.query, parsed.fragment)) + return "" diff --git a/src/production_analytics/persistence/__init__.py b/src/production_analytics/persistence/__init__.py new file mode 100644 index 0000000..d1e329f --- /dev/null +++ b/src/production_analytics/persistence/__init__.py @@ -0,0 +1,5 @@ +"""Persistence ports for derived data and incremental calculation state.""" + +from .ports import DerivedResultRepository + +__all__ = ["DerivedResultRepository"] diff --git a/src/production_analytics/persistence/ports.py b/src/production_analytics/persistence/ports.py new file mode 100644 index 0000000..b0b18a4 --- /dev/null +++ b/src/production_analytics/persistence/ports.py @@ -0,0 +1,13 @@ +"""Database-independent interfaces; no raw ENLYZE sample persistence.""" + +from typing import Protocol + +from production_analytics.domain import CalculatedMetric, ProcessEvent + + +class DerivedResultRepository(Protocol): + """Stores only derived results; implementation follows schema design.""" + + def save_metric(self, metric: CalculatedMetric) -> None: ... + + def save_event(self, event: ProcessEvent) -> None: ... diff --git a/src/production_analytics/service/__init__.py b/src/production_analytics/service/__init__.py new file mode 100644 index 0000000..b8b9745 --- /dev/null +++ b/src/production_analytics/service/__init__.py @@ -0,0 +1 @@ +"""Future FastAPI composition layer; intentionally endpoint-free for now.""" diff --git a/tests/test_cli.py b/tests/test_cli.py new file mode 100644 index 0000000..2eb133e --- /dev/null +++ b/tests/test_cli.py @@ -0,0 +1,81 @@ +import io +import json +import os +import tempfile +import unittest +from contextlib import redirect_stdout +from pathlib import Path +from unittest.mock import patch + +from production_analytics.cli.__main__ import main +from production_analytics.enlyze.exploration import ExplorationResponse + + +class CliTests(unittest.TestCase): + @patch.dict(os.environ, {"ENLYZE_BASE_URL": "https://enlyze.example"}, clear=True) + @patch("production_analytics.cli.__main__.ExplorationClient.get") + def test_raw_prints_and_saves_sanitized_response(self, get_mock: object) -> None: + get_mock.return_value = ExplorationResponse( # type: ignore[attr-defined] + status_code=200, + path="/verified/path", + headers={"Content-Type": "application/json"}, + body={"api_token": "secret", "authorization": "Bearer secret", "value": 4.2}, + ) + with tempfile.TemporaryDirectory() as temporary_directory: + output = io.StringIO() + with redirect_stdout(output): + result = main( + [ + "enlyze", + "raw", + "/verified/path", + "--save-fixture", + "example.json", + "--fixture-dir", + temporary_directory, + "--save-raw", + "raw-example.json", + "--raw-dir", + temporary_directory, + ] + ) + fixture = json.loads((Path(temporary_directory) / "example.json").read_text()) + raw_capture = json.loads((Path(temporary_directory) / "raw-example.json").read_text()) + + self.assertEqual(result, 0) + self.assertEqual(fixture["response"]["body"]["api_token"], "") + self.assertEqual(fixture["response"]["body"]["authorization"], "") + self.assertNotIn("Bearer secret", output.getvalue()) + self.assertEqual(raw_capture["response"]["body"]["api_token"], "secret") + + def test_raw_rejects_unsafe_fixture_name(self) -> None: + with patch.dict(os.environ, {"ENLYZE_BASE_URL": "https://enlyze.example"}, clear=True): + with patch("production_analytics.cli.__main__.ExplorationClient.get") as get_mock: + get_mock.return_value = ExplorationResponse(200, "/path", {}, {}) + output = io.StringIO() + with redirect_stdout(output): + result = main(["enlyze", "raw", "/path", "--save-fixture", "../unsafe.json"]) + + self.assertEqual(result, 1) + self.assertIn("plain filename", output.getvalue()) + + @patch.dict(os.environ, {"ENLYZE_BASE_URL": "https://enlyze.example"}, clear=True) + @patch("production_analytics.cli.__main__.ExplorationClient.post_json") + def test_timeseries_builds_schema_supported_read_only_post(self, post_mock: object) -> None: + post_mock.return_value = ExplorationResponse( # type: ignore[attr-defined] + 200, "/v2/timeseries", {}, {"metadata": {"next_cursor": None}, "data": {"columns": [], "records": []}} + ) + with tempfile.TemporaryDirectory() as temporary_directory: + result = main( + [ + "enlyze", "timeseries", "--machine", "machine-id", "--start", "2026-01-01T00:00:00+00:00", + "--end", "2026-01-01T00:10:00+00:00", "--variable", "variable-id", "--resampling-interval", "10", + "--resampling-method", "avg", "--save-fixture", "timeseries.json", "--fixture-dir", temporary_directory, + ] + ) + fixture = json.loads((Path(temporary_directory) / "timeseries.json").read_text()) + + self.assertEqual(result, 0) + self.assertEqual(post_mock.call_args.args[0], "/v2/timeseries") # type: ignore[attr-defined] + self.assertEqual(post_mock.call_args.args[1]["resampling_interval"], 10) # type: ignore[attr-defined] + self.assertEqual(fixture["request"]["method"], "POST") diff --git a/tests/test_domain_models.py b/tests/test_domain_models.py new file mode 100644 index 0000000..10b9f8c --- /dev/null +++ b/tests/test_domain_models.py @@ -0,0 +1,19 @@ +from datetime import UTC, datetime, timedelta +import unittest + +from production_analytics.domain import SourceRange + + +class SourceRangeTests(unittest.TestCase): + def test_accepts_ordered_timestamps(self) -> None: + start = datetime(2026, 1, 1, tzinfo=UTC) + source_range = SourceRange(start=start, end=start + timedelta(minutes=1)) + + self.assertEqual(source_range.start, start) + + + def test_rejects_reverse_interval(self) -> None: + end = datetime(2026, 1, 1, tzinfo=UTC) + + with self.assertRaisesRegex(ValueError, "must not precede"): + SourceRange(start=end + timedelta(seconds=1), end=end) diff --git a/tests/test_enlyze_exploration.py b/tests/test_enlyze_exploration.py new file mode 100644 index 0000000..3fdd7b1 --- /dev/null +++ b/tests/test_enlyze_exploration.py @@ -0,0 +1,121 @@ +import io +import json +import tempfile +import unittest +from unittest.mock import patch +from urllib.error import HTTPError + +from production_analytics.enlyze.exploration import ( + AuthenticationError, + ExplorationClient, + ExplorationSettings, + NonJsonResponseError, + load_secret_file, +) +from production_analytics.enlyze.sanitize import sanitize + + +class _Response: + status = 200 + headers = {"Content-Type": "application/json", "X-Request-Id": "request-123"} + + def __init__(self, body: bytes) -> None: + self._body = body + + def read(self) -> bytes: + return self._body + + def __enter__(self): + return self + + def __exit__(self, *_: object) -> None: + return None + + +class ExplorationClientTests(unittest.TestCase): + def setUp(self) -> None: + self.settings = ExplorationSettings.from_environment( + { + "ENLYZE_BASE_URL": "https://enlyze.example/", + "ENLYZE_API_KEY": "secret", + } + ) + + @patch("production_analytics.enlyze.exploration.urlopen") + def test_get_uses_explicit_header_and_decodes_json(self, urlopen_mock: object) -> None: + urlopen_mock.return_value = _Response(b'{"items": [1]}') # type: ignore[attr-defined] + + response = ExplorationClient(self.settings).get("/observed/path", {"limit": "1"}) + + request = urlopen_mock.call_args.args[0] # type: ignore[attr-defined] + self.assertEqual(request.get_method(), "GET") + self.assertEqual(request.get_header("Authorization"), "Bearer secret") + self.assertEqual(response.body, {"items": [1]}) + + @patch("production_analytics.enlyze.exploration.urlopen") + def test_non_json_response_is_clear_error(self, urlopen_mock: object) -> None: + urlopen_mock.return_value = _Response(b"not json") # type: ignore[attr-defined] + + with self.assertRaises(NonJsonResponseError): + ExplorationClient(self.settings).get("/observed/path") + + @patch("production_analytics.enlyze.exploration.urlopen") + def test_post_json_sends_json_body_and_bearer_header(self, urlopen_mock: object) -> None: + urlopen_mock.return_value = _Response(b'{"data": {}}') # type: ignore[attr-defined] + + ExplorationClient(self.settings).post_json("/v2/timeseries", {"machine": "machine-id"}) + + request = urlopen_mock.call_args.args[0] # type: ignore[attr-defined] + self.assertEqual(request.get_method(), "POST") + self.assertEqual(request.get_header("Content-type"), "application/json") + self.assertEqual(request.data, b'{"machine": "machine-id"}') + self.assertEqual(request.get_header("Authorization"), "Bearer secret") + + @patch("production_analytics.enlyze.exploration.urlopen") + def test_authentication_error_does_not_expose_response_body(self, urlopen_mock: object) -> None: + urlopen_mock.side_effect = HTTPError("https://example.invalid", 401, "Unauthorized", {}, io.BytesIO(b"secret")) # type: ignore[attr-defined] + + with self.assertRaisesRegex(AuthenticationError, "HTTP 401"): + ExplorationClient(self.settings).get("/observed/path") + + +class SanitizationTests(unittest.TestCase): + def test_sanitization_preserves_shape_and_redacts_sensitive_values(self) -> None: + response = sanitize( + { + "machine_id": "machine-47", + "token": "very-secret", + "Authorization": "Bearer very-secret", + "site_name": "Sensitive Site", + "api_url": "https://internal.example/v1/items", + "timestamp": "2026-01-01T00:00:00Z", + "unit": "kW", + "value": 12.5, + "quality": "good", + "items": [{"machine_id": "machine-47"}], + } + ) + + self.assertEqual(response["token"], "") + self.assertEqual(response["Authorization"], "") + self.assertEqual(response["site_name"], "") + self.assertEqual(response["api_url"], "https://redacted-host.invalid/v1/items") + self.assertEqual(response["timestamp"], "2026-01-01T00:00:00Z") + self.assertEqual(response["unit"], "kW") + self.assertEqual(response["value"], 12.5) + self.assertEqual(response["machine_id"], response["items"][0]["machine_id"]) + + +class SecretFileTests(unittest.TestCase): + def test_secret_file_is_parsed_without_execution(self) -> None: + with tempfile.NamedTemporaryFile(mode="w", encoding="utf-8") as secret_file: + secret_file.write("ENLYZE_API_KEY='key value'\nUNRELATED=plain # comment\n") + secret_file.flush() + + values = load_secret_file(secret_file.name) + + self.assertEqual(values, {"ENLYZE_API_KEY": "key value", "UNRELATED": "plain"}) + + +if __name__ == "__main__": + unittest.main()