from __future__ import annotations import logging from collections.abc import Sequence from dataclasses import dataclass from app.ml.training import TrainingPipeline logger = logging.getLogger(__name__) @dataclass class Metric: name: str value: float threshold: float | None = None @dataclass class EvalReport: artifact_id: str sample_size: int metrics: list[Metric] class Evaluator: def __init__(self, pipeline: TrainingPipeline) -> None: self._pipeline = pipeline def evaluate(self, artifact_id: str, predictions: Sequence[str]) -> EvalReport: try: supported_sensors = set(self._pipeline.export(artifact_id).supported_sensors) except KeyError as exc: raise ValueError("Kein trainiertes Modell für Evaluation vorhanden.") from exc parsed_sensors = [_prediction_sensor(prediction) for prediction in predictions] supported_hits = sum(sensor in supported_sensors for sensor in parsed_sensors) unknown_hits = sum(sensor not in supported_sensors for sensor in parsed_sensors) sample_size = len(predictions) coverage = supported_hits / sample_size if sample_size else 0.0 unknown_rate = unknown_hits / sample_size if sample_size else 0.0 coverage_metric = Metric(name="coverage", value=coverage, threshold=0.8) unknown_metric = Metric(name="unknown_rate", value=unknown_rate, threshold=0.1) report = EvalReport( artifact_id=artifact_id, sample_size=sample_size, metrics=[coverage_metric, unknown_metric], ) logger.info( "Evaluation %s -> coverage=%.2f, unknown_rate=%.2f", artifact_id, coverage, unknown_rate, ) return report def _prediction_sensor(prediction: str) -> str | None: parts = prediction.split(":", 2) if len(parts) != 3 or not parts[0] or not parts[1]: return None return parts[1]