Files
sillyhome-next-dev/app/ml/evaluation.py

65 lines
1.9 KiB
Python

from __future__ import annotations
import logging
from collections.abc import Sequence
from dataclasses import dataclass
from app.ml.training import TrainingPipeline
logger = logging.getLogger(__name__)
@dataclass
class Metric:
name: str
value: float
threshold: float | None = None
@dataclass
class EvalReport:
artifact_id: str
sample_size: int
metrics: list[Metric]
class Evaluator:
def __init__(self, pipeline: TrainingPipeline) -> None:
self._pipeline = pipeline
def evaluate(self, artifact_id: str, predictions: Sequence[str]) -> EvalReport:
try:
supported_sensors = set(self._pipeline.export(artifact_id).supported_sensors)
except KeyError as exc:
raise ValueError("Kein trainiertes Modell für Evaluation vorhanden.") from exc
parsed_sensors = [_prediction_sensor(prediction) for prediction in predictions]
supported_hits = sum(sensor in supported_sensors for sensor in parsed_sensors)
unknown_hits = sum(sensor not in supported_sensors for sensor in parsed_sensors)
sample_size = len(predictions)
coverage = supported_hits / sample_size if sample_size else 0.0
unknown_rate = unknown_hits / sample_size if sample_size else 0.0
coverage_metric = Metric(name="coverage", value=coverage, threshold=0.8)
unknown_metric = Metric(name="unknown_rate", value=unknown_rate, threshold=0.1)
report = EvalReport(
artifact_id=artifact_id,
sample_size=sample_size,
metrics=[coverage_metric, unknown_metric],
)
logger.info(
"Evaluation %s -> coverage=%.2f, unknown_rate=%.2f",
artifact_id,
coverage,
unknown_rate,
)
return report
def _prediction_sensor(prediction: str) -> str | None:
parts = prediction.split(":", 2)
if len(parts) != 3 or not parts[0] or not parts[1]:
return None
return parts[1]