- Introduced experiment_name parameter in TrainModelResult to enhance tracking of training experiments. - Updated the Training class to utilize run_name and experiment_name for improved MLflow run management.
54 lines
2.4 KiB
Python
54 lines
2.4 KiB
Python
from dataclasses import dataclass
|
|
|
|
import pandas as pd
|
|
|
|
from model_manager.utils.models.train_model_params import TrainModelParams
|
|
|
|
|
|
@dataclass
|
|
class TrainModelResult:
|
|
"""
|
|
A data container for storing the results of a machine learning training process.
|
|
|
|
This dataclass encapsulates all outputs from the training pipeline, including
|
|
the prepared datasets, evaluation metrics, and paths to generated artifacts.
|
|
It is used to pass results between activities in the training workflow.
|
|
|
|
Attributes:
|
|
params (TrainModelParams): The parameters used to train the model.
|
|
x_train (pd.DataFrame): The training dataset features.
|
|
x_test (pd.DataFrame): The testing dataset features.
|
|
y_train (pd.DataFrame): The training dataset target values.
|
|
y_test (pd.DataFrame): The testing dataset target values.
|
|
y_pred (pd.Series | None): The predicted target values for the testing dataset. Default is None.
|
|
y_train_pred (pd.Series | None): The predicted target values for the training dataset. Default is None.
|
|
mse_val (float | None): The Mean Squared Error (MSE) of the predictions. Default is None.
|
|
mae_val (float | None): The Mean Absolute Error (MAE) of the predictions. Default is None.
|
|
r2_val (float | None): The R-squared (R²) value of the predictions. Default is None.
|
|
equation (dict | None): The equation of the model. Default is None.
|
|
equation_path (str | None): The path to the equation file. Default is None.
|
|
run_name (str | None): The name of the MLFlow run. Default is None.
|
|
report_path (str | None): The path to the generated HTML report file. Default is None.
|
|
train_data_path (str | None): The path to the training dataset CSV file. Default is None.
|
|
test_data_path (str | None): The path to the testing dataset CSV file. Default is None.
|
|
"""
|
|
|
|
params: TrainModelParams
|
|
train_data: pd.DataFrame
|
|
val_data: pd.DataFrame
|
|
y_pred: pd.DataFrame | None = None
|
|
y_train_pred: pd.DataFrame | None = None
|
|
mse_val: float | None = None
|
|
mae_val: float | None = None
|
|
r2_val: float | None = None
|
|
equation: dict | None = None
|
|
equation_path: str | None = None
|
|
run_name: str | None = None
|
|
experiment_name: str | None = None
|
|
run_id: str | None = None
|
|
report_path: str | None = None
|
|
train_data_path: str | None = None
|
|
test_data_path: str | None = None
|
|
|
|
run_dir: str | None = None
|