diff --git a/docs/test-scenarios/01-linear-regression-basic.json b/docs/test-scenarios/01-linear-regression-basic.json index faec507..28ff007 100644 --- a/docs/test-scenarios/01-linear-regression-basic.json +++ b/docs/test-scenarios/01-linear-regression-basic.json @@ -1,30 +1,40 @@ { "_description": "Cenário básico de regressão linear sem scaler", - "experimentName": "test-linear-regression-basic", - "username": "bruno.domingues@aignosi.com.br", - "modelName": "Linear Regression", - "targetVariable": "03CV020/CORRENTE_N_M1_PV(Value)", - "variableColumns": ["303-WIT-200(Value)"], - "lagTrain": {"303-WIT-200(Value)": 0}, - "lagVal": {"303-WIT-200(Value)": 0}, - "remStaticWin": false, - "lowLim": {}, - "uppLim": {}, - "window": 0, - "useScaler": false, - "includeAr": false, - "trainSize": 80, + "experiment_run_id": 1001, + "variable_columns": [ + "303-WIT-200(Value)" + ], + "target_variable": "03CV020/CORRENTE_N_M1_PV(Value)", + "bucket_name": "model-training", + "file_name": "training_data.csv", + "line_separator": ",", + "decimal_separator": ".", + "date_column": "timestamp", + "date_format": "yyyy-MM-dd HH:mm:ss", + "train_size": 80, "shuffle": true, - "lineSeparator": ",", - "decimalSeparator": ".", - "dateColumn": "timestamp", - "dateFormat": "yyyy-MM-dd HH:mm:ss", - "removedIntervals": [], - "degree": 1, - "interactionOnly": false, - "nanTreatment": "drop", - "startDate": null, - "endDate": null, - "scalerName": "None", - "supportFilters": {} -} + "random_state": 42, + "model_name": "Linear Regression", + "model_type": "linear_regression", + "data_model_kwargs": { + "lag_train": { + "303-WIT-200(Value)": 0 + }, + "lag_val": { + "303-WIT-200(Value)": 0 + }, + "nan_treatment": "drop", + "rem_static_win": false, + "static_threshold": null, + "start_date": null, + "end_date": null, + "support_filters": {}, + "removed_intervals": [] + }, + "model_kwargs": { + "degree": 1, + "interaction_only": false, + "scaler_name": "None" + }, + "opt_params": {} +} \ No newline at end of file diff --git a/docs/test-scenarios/02-linear-regression-with-scaler.json b/docs/test-scenarios/02-linear-regression-with-scaler.json index 794f645..1c02fb5 100644 --- a/docs/test-scenarios/02-linear-regression-with-scaler.json +++ b/docs/test-scenarios/02-linear-regression-with-scaler.json @@ -1,30 +1,40 @@ { "_description": "Regressão linear com Standard Scaler habilitado", - "experimentName": "test-linear-regression-scaler", - "username": "bruno.domingues@aignosi.com.br", - "modelName": "Linear Regression", - "targetVariable": "03CV020/CORRENTE_N_M1_PV(Value)", - "variableColumns": ["303-WIT-200(Value)"], - "lagTrain": {"303-WIT-200(Value)": 0}, - "lagVal": {"303-WIT-200(Value)": 0}, - "remStaticWin": false, - "lowLim": {}, - "uppLim": {}, - "window": 0, - "useScaler": true, - "includeAr": false, - "trainSize": 80, + "experiment_run_id": 1002, + "variable_columns": [ + "303-WIT-200(Value)" + ], + "target_variable": "03CV020/CORRENTE_N_M1_PV(Value)", + "bucket_name": "model-training", + "file_name": "training_data.csv", + "line_separator": ",", + "decimal_separator": ".", + "date_column": "timestamp", + "date_format": "yyyy-MM-dd HH:mm:ss", + "train_size": 80, "shuffle": true, - "lineSeparator": ",", - "decimalSeparator": ".", - "dateColumn": "timestamp", - "dateFormat": "yyyy-MM-dd HH:mm:ss", - "removedIntervals": [], - "degree": 1, - "interactionOnly": false, - "nanTreatment": "drop", - "startDate": null, - "endDate": null, - "scalerName": "Standard Scaler", - "supportFilters": {} -} + "random_state": 42, + "model_name": "Linear Regression", + "model_type": "linear_regression", + "data_model_kwargs": { + "lag_train": { + "303-WIT-200(Value)": 0 + }, + "lag_val": { + "303-WIT-200(Value)": 0 + }, + "nan_treatment": "drop", + "rem_static_win": false, + "static_threshold": null, + "start_date": null, + "end_date": null, + "support_filters": {}, + "removed_intervals": [] + }, + "model_kwargs": { + "degree": 1, + "interaction_only": false, + "scaler_name": "Standard Scaler" + }, + "opt_params": {} +} \ No newline at end of file diff --git a/docs/test-scenarios/03-polynomial-regression-degree2.json b/docs/test-scenarios/03-polynomial-regression-degree2.json index e39a8c7..050c44f 100644 --- a/docs/test-scenarios/03-polynomial-regression-degree2.json +++ b/docs/test-scenarios/03-polynomial-regression-degree2.json @@ -1,30 +1,40 @@ { "_description": "Regressão polinomial de grau 2 com scaler (obrigatório para evitar overflow)", - "experimentName": "test-polynomial-degree2", - "username": "bruno.domingues@aignosi.com.br", - "modelName": "Polynomial Regression", - "targetVariable": "03CV020/CORRENTE_N_M1_PV(Value)", - "variableColumns": ["303-WIT-200(Value)"], - "lagTrain": {"303-WIT-200(Value)": 0}, - "lagVal": {"303-WIT-200(Value)": 0}, - "remStaticWin": false, - "lowLim": {}, - "uppLim": {}, - "window": 0, - "useScaler": true, - "includeAr": false, - "trainSize": 80, + "experiment_run_id": 1003, + "variable_columns": [ + "303-WIT-200(Value)" + ], + "target_variable": "03CV020/CORRENTE_N_M1_PV(Value)", + "bucket_name": "model-training", + "file_name": "training_data.csv", + "line_separator": ",", + "decimal_separator": ".", + "date_column": "timestamp", + "date_format": "yyyy-MM-dd HH:mm:ss", + "train_size": 80, "shuffle": true, - "lineSeparator": ",", - "decimalSeparator": ".", - "dateColumn": "timestamp", - "dateFormat": "yyyy-MM-dd HH:mm:ss", - "removedIntervals": [], - "degree": 2, - "interactionOnly": false, - "nanTreatment": "drop", - "startDate": null, - "endDate": null, - "scalerName": "Standard Scaler", - "supportFilters": {} -} + "random_state": 42, + "model_name": "Polynomial Regression", + "model_type": "polynomial_regression", + "data_model_kwargs": { + "lag_train": { + "303-WIT-200(Value)": 0 + }, + "lag_val": { + "303-WIT-200(Value)": 0 + }, + "nan_treatment": "drop", + "rem_static_win": false, + "static_threshold": null, + "start_date": null, + "end_date": null, + "support_filters": {}, + "removed_intervals": [] + }, + "model_kwargs": { + "degree": 2, + "interaction_only": false, + "scaler_name": "Standard Scaler" + }, + "opt_params": {} +} \ No newline at end of file diff --git a/docs/test-scenarios/04-polynomial-regression-degree3.json b/docs/test-scenarios/04-polynomial-regression-degree3.json index 49d1eb5..5e64c10 100644 --- a/docs/test-scenarios/04-polynomial-regression-degree3.json +++ b/docs/test-scenarios/04-polynomial-regression-degree3.json @@ -1,30 +1,40 @@ { "_description": "Regressão polinomial de grau 3 com scaler", - "experimentName": "test-polynomial-degree3", - "username": "bruno.domingues@aignosi.com.br", - "modelName": "Polynomial Regression", - "targetVariable": "03CV020/CORRENTE_N_M1_PV(Value)", - "variableColumns": ["303-WIT-200(Value)"], - "lagTrain": {"303-WIT-200(Value)": 0}, - "lagVal": {"303-WIT-200(Value)": 0}, - "remStaticWin": false, - "lowLim": {}, - "uppLim": {}, - "window": 0, - "useScaler": true, - "includeAr": false, - "trainSize": 80, + "experiment_run_id": 1004, + "variable_columns": [ + "303-WIT-200(Value)" + ], + "target_variable": "03CV020/CORRENTE_N_M1_PV(Value)", + "bucket_name": "model-training", + "file_name": "training_data.csv", + "line_separator": ",", + "decimal_separator": ".", + "date_column": "timestamp", + "date_format": "yyyy-MM-dd HH:mm:ss", + "train_size": 80, "shuffle": true, - "lineSeparator": ",", - "decimalSeparator": ".", - "dateColumn": "timestamp", - "dateFormat": "yyyy-MM-dd HH:mm:ss", - "removedIntervals": [], - "degree": 3, - "interactionOnly": false, - "nanTreatment": "drop", - "startDate": null, - "endDate": null, - "scalerName": "Standard Scaler", - "supportFilters": {} -} + "random_state": 42, + "model_name": "Polynomial Regression", + "model_type": "polynomial_regression", + "data_model_kwargs": { + "lag_train": { + "303-WIT-200(Value)": 0 + }, + "lag_val": { + "303-WIT-200(Value)": 0 + }, + "nan_treatment": "drop", + "rem_static_win": false, + "static_threshold": null, + "start_date": null, + "end_date": null, + "support_filters": {}, + "removed_intervals": [] + }, + "model_kwargs": { + "degree": 3, + "interaction_only": false, + "scaler_name": "Standard Scaler" + }, + "opt_params": {} +} \ No newline at end of file diff --git a/docs/test-scenarios/05-linear-regression-with-lags.json b/docs/test-scenarios/05-linear-regression-with-lags.json index fad3c8f..f72cd81 100644 --- a/docs/test-scenarios/05-linear-regression-with-lags.json +++ b/docs/test-scenarios/05-linear-regression-with-lags.json @@ -1,30 +1,40 @@ { "_description": "Regressão linear com lags de treino e validação", - "experimentName": "test-linear-with-lags", - "username": "bruno.domingues@aignosi.com.br", - "modelName": "Linear Regression", - "targetVariable": "03CV020/CORRENTE_N_M1_PV(Value)", - "variableColumns": ["303-WIT-200(Value)"], - "lagTrain": {"303-WIT-200(Value)": 5}, - "lagVal": {"303-WIT-200(Value)": 3}, - "remStaticWin": false, - "lowLim": {}, - "uppLim": {}, - "window": 0, - "useScaler": false, - "includeAr": false, - "trainSize": 80, + "experiment_run_id": 1005, + "variable_columns": [ + "303-WIT-200(Value)" + ], + "target_variable": "03CV020/CORRENTE_N_M1_PV(Value)", + "bucket_name": "model-training", + "file_name": "training_data.csv", + "line_separator": ",", + "decimal_separator": ".", + "date_column": "timestamp", + "date_format": "yyyy-MM-dd HH:mm:ss", + "train_size": 80, "shuffle": true, - "lineSeparator": ",", - "decimalSeparator": ".", - "dateColumn": "timestamp", - "dateFormat": "yyyy-MM-dd HH:mm:ss", - "removedIntervals": [], - "degree": 1, - "interactionOnly": false, - "nanTreatment": "drop", - "startDate": null, - "endDate": null, - "scalerName": "None", - "supportFilters": {} -} + "random_state": 42, + "model_name": "Linear Regression", + "model_type": "linear_regression", + "data_model_kwargs": { + "lag_train": { + "303-WIT-200(Value)": 5 + }, + "lag_val": { + "303-WIT-200(Value)": 3 + }, + "nan_treatment": "drop", + "rem_static_win": false, + "static_threshold": null, + "start_date": null, + "end_date": null, + "support_filters": {}, + "removed_intervals": [] + }, + "model_kwargs": { + "degree": 1, + "interaction_only": false, + "scaler_name": "None" + }, + "opt_params": {} +} \ No newline at end of file diff --git a/docs/test-scenarios/06-linear-regression-nan-interpolation.json b/docs/test-scenarios/06-linear-regression-nan-interpolation.json index 2bdf1a4..8d906b0 100644 --- a/docs/test-scenarios/06-linear-regression-nan-interpolation.json +++ b/docs/test-scenarios/06-linear-regression-nan-interpolation.json @@ -1,30 +1,40 @@ { "_description": "Regressão linear com tratamento de NaN por interpolação linear", - "experimentName": "test-linear-nan-interpolation", - "username": "bruno.domingues@aignosi.com.br", - "modelName": "Linear Regression", - "targetVariable": "03CV020/CORRENTE_N_M1_PV(Value)", - "variableColumns": ["303-WIT-200(Value)"], - "lagTrain": {"303-WIT-200(Value)": 0}, - "lagVal": {"303-WIT-200(Value)": 0}, - "remStaticWin": false, - "lowLim": {}, - "uppLim": {}, - "window": 0, - "useScaler": false, - "includeAr": false, - "trainSize": 80, + "experiment_run_id": 1006, + "variable_columns": [ + "303-WIT-200(Value)" + ], + "target_variable": "03CV020/CORRENTE_N_M1_PV(Value)", + "bucket_name": "model-training", + "file_name": "training_data.csv", + "line_separator": ",", + "decimal_separator": ".", + "date_column": "timestamp", + "date_format": "yyyy-MM-dd HH:mm:ss", + "train_size": 80, "shuffle": true, - "lineSeparator": ",", - "decimalSeparator": ".", - "dateColumn": "timestamp", - "dateFormat": "yyyy-MM-dd HH:mm:ss", - "removedIntervals": [], - "degree": 1, - "interactionOnly": false, - "nanTreatment": "linear interpolation", - "startDate": null, - "endDate": null, - "scalerName": "None", - "supportFilters": {} -} + "random_state": 42, + "model_name": "Linear Regression", + "model_type": "linear_regression", + "data_model_kwargs": { + "lag_train": { + "303-WIT-200(Value)": 0 + }, + "lag_val": { + "303-WIT-200(Value)": 0 + }, + "nan_treatment": "linear interpolation", + "rem_static_win": false, + "static_threshold": null, + "start_date": null, + "end_date": null, + "support_filters": {}, + "removed_intervals": [] + }, + "model_kwargs": { + "degree": 1, + "interaction_only": false, + "scaler_name": "None" + }, + "opt_params": {} +} \ No newline at end of file diff --git a/docs/test-scenarios/07-linear-regression-static-window-removal.json b/docs/test-scenarios/07-linear-regression-static-window-removal.json index 3058387..26721be 100644 --- a/docs/test-scenarios/07-linear-regression-static-window-removal.json +++ b/docs/test-scenarios/07-linear-regression-static-window-removal.json @@ -1,30 +1,40 @@ { "_description": "Regressão linear com remoção de janelas estáticas", - "experimentName": "test-linear-static-removal", - "username": "bruno.domingues@aignosi.com.br", - "modelName": "Linear Regression", - "targetVariable": "03CV020/CORRENTE_N_M1_PV(Value)", - "variableColumns": ["303-WIT-200(Value)"], - "lagTrain": {"303-WIT-200(Value)": 0}, - "lagVal": {"303-WIT-200(Value)": 0}, - "remStaticWin": true, - "lowLim": {}, - "uppLim": {}, - "window": 10, - "useScaler": false, - "includeAr": false, - "trainSize": 80, + "experiment_run_id": 1007, + "variable_columns": [ + "303-WIT-200(Value)" + ], + "target_variable": "03CV020/CORRENTE_N_M1_PV(Value)", + "bucket_name": "model-training", + "file_name": "training_data.csv", + "line_separator": ",", + "decimal_separator": ".", + "date_column": "timestamp", + "date_format": "yyyy-MM-dd HH:mm:ss", + "train_size": 80, "shuffle": true, - "lineSeparator": ",", - "decimalSeparator": ".", - "dateColumn": "timestamp", - "dateFormat": "yyyy-MM-dd HH:mm:ss", - "removedIntervals": [], - "degree": 1, - "interactionOnly": false, - "nanTreatment": "drop", - "startDate": null, - "endDate": null, - "scalerName": "None", - "supportFilters": {} -} + "random_state": 42, + "model_name": "Linear Regression", + "model_type": "linear_regression", + "data_model_kwargs": { + "lag_train": { + "303-WIT-200(Value)": 0 + }, + "lag_val": { + "303-WIT-200(Value)": 0 + }, + "nan_treatment": "drop", + "rem_static_win": true, + "static_threshold": null, + "start_date": null, + "end_date": null, + "support_filters": {}, + "removed_intervals": [] + }, + "model_kwargs": { + "degree": 1, + "interaction_only": false, + "scaler_name": "None" + }, + "opt_params": {} +} \ No newline at end of file diff --git a/docs/test-scenarios/08-linear-regression-with-limits.json b/docs/test-scenarios/08-linear-regression-with-limits.json index 874bfcc..f7784e8 100644 --- a/docs/test-scenarios/08-linear-regression-with-limits.json +++ b/docs/test-scenarios/08-linear-regression-with-limits.json @@ -1,30 +1,45 @@ { "_description": "Regressão linear com limites inferior e superior para variáveis", - "experimentName": "test-linear-with-limits", - "username": "bruno.domingues@aignosi.com.br", - "modelName": "Linear Regression", - "targetVariable": "03CV020/CORRENTE_N_M1_PV(Value)", - "variableColumns": ["303-WIT-200(Value)"], - "lagTrain": {"303-WIT-200(Value)": 0}, - "lagVal": {"303-WIT-200(Value)": 0}, - "remStaticWin": false, - "lowLim": {"303-WIT-200(Value)": 0.0}, - "uppLim": {"303-WIT-200(Value)": 1000.0}, - "window": 0, - "useScaler": false, - "includeAr": false, - "trainSize": 80, + "experiment_run_id": 1008, + "variable_columns": [ + "303-WIT-200(Value)" + ], + "target_variable": "03CV020/CORRENTE_N_M1_PV(Value)", + "bucket_name": "model-training", + "file_name": "training_data.csv", + "line_separator": ",", + "decimal_separator": ".", + "date_column": "timestamp", + "date_format": "yyyy-MM-dd HH:mm:ss", + "train_size": 80, "shuffle": true, - "lineSeparator": ",", - "decimalSeparator": ".", - "dateColumn": "timestamp", - "dateFormat": "yyyy-MM-dd HH:mm:ss", - "removedIntervals": [], - "degree": 1, - "interactionOnly": false, - "nanTreatment": "drop", - "startDate": null, - "endDate": null, - "scalerName": "None", - "supportFilters": {} -} + "random_state": 42, + "model_name": "Linear Regression", + "model_type": "linear_regression", + "data_model_kwargs": { + "lag_train": { + "303-WIT-200(Value)": 0 + }, + "lag_val": { + "303-WIT-200(Value)": 0 + }, + "nan_treatment": "drop", + "rem_static_win": false, + "static_threshold": null, + "start_date": null, + "end_date": null, + "support_filters": { + "303-WIT-200(Value)": { + "min": 0.0, + "max": 1000.0 + } + }, + "removed_intervals": [] + }, + "model_kwargs": { + "degree": 1, + "interaction_only": false, + "scaler_name": "None" + }, + "opt_params": {} +} \ No newline at end of file diff --git a/docs/test-scenarios/09-polynomial-degree2-with-scaler-and-lags.json b/docs/test-scenarios/09-polynomial-degree2-with-scaler-and-lags.json index 86a8880..e3ef6c0 100644 --- a/docs/test-scenarios/09-polynomial-degree2-with-scaler-and-lags.json +++ b/docs/test-scenarios/09-polynomial-degree2-with-scaler-and-lags.json @@ -1,30 +1,40 @@ { "_description": "Cenário completo: regressão polinomial grau 2 com scaler e lags", - "experimentName": "test-polynomial-complete", - "username": "bruno.domingues@aignosi.com.br", - "modelName": "Polynomial Regression", - "targetVariable": "03CV020/CORRENTE_N_M1_PV(Value)", - "variableColumns": ["303-WIT-200(Value)"], - "lagTrain": {"303-WIT-200(Value)": 3}, - "lagVal": {"303-WIT-200(Value)": 2}, - "remStaticWin": false, - "lowLim": {}, - "uppLim": {}, - "window": 0, - "useScaler": true, - "includeAr": false, - "trainSize": 80, + "experiment_run_id": 1009, + "variable_columns": [ + "303-WIT-200(Value)" + ], + "target_variable": "03CV020/CORRENTE_N_M1_PV(Value)", + "bucket_name": "model-training", + "file_name": "training_data.csv", + "line_separator": ",", + "decimal_separator": ".", + "date_column": "timestamp", + "date_format": "yyyy-MM-dd HH:mm:ss", + "train_size": 80, "shuffle": true, - "lineSeparator": ",", - "decimalSeparator": ".", - "dateColumn": "timestamp", - "dateFormat": "yyyy-MM-dd HH:mm:ss", - "removedIntervals": [], - "degree": 2, - "interactionOnly": false, - "nanTreatment": "drop", - "startDate": null, - "endDate": null, - "scalerName": "Standard Scaler", - "supportFilters": {} -} + "random_state": 42, + "model_name": "Polynomial Regression", + "model_type": "polynomial_regression", + "data_model_kwargs": { + "lag_train": { + "303-WIT-200(Value)": 3 + }, + "lag_val": { + "303-WIT-200(Value)": 2 + }, + "nan_treatment": "drop", + "rem_static_win": false, + "static_threshold": null, + "start_date": null, + "end_date": null, + "support_filters": {}, + "removed_intervals": [] + }, + "model_kwargs": { + "degree": 2, + "interaction_only": false, + "scaler_name": "Standard Scaler" + }, + "opt_params": {} +} \ No newline at end of file diff --git a/docs/test-scenarios/10-linear-regression-with-ar.json b/docs/test-scenarios/10-linear-regression-with-ar.json index 930ac3c..a2adfbe 100644 --- a/docs/test-scenarios/10-linear-regression-with-ar.json +++ b/docs/test-scenarios/10-linear-regression-with-ar.json @@ -1,30 +1,40 @@ { "_description": "Regressão linear com variável autoregressiva (AR)", - "experimentName": "test-linear-with-ar", - "username": "bruno.domingues@aignosi.com.br", - "modelName": "Linear Regression", - "targetVariable": "03CV020/CORRENTE_N_M1_PV(Value)", - "variableColumns": ["303-WIT-200(Value)"], - "lagTrain": {"303-WIT-200(Value)": 0}, - "lagVal": {"303-WIT-200(Value)": 0}, - "remStaticWin": false, - "lowLim": {}, - "uppLim": {}, - "window": 0, - "useScaler": false, - "includeAr": true, - "trainSize": 80, + "experiment_run_id": 1010, + "variable_columns": [ + "303-WIT-200(Value)" + ], + "target_variable": "03CV020/CORRENTE_N_M1_PV(Value)", + "bucket_name": "model-training", + "file_name": "training_data.csv", + "line_separator": ",", + "decimal_separator": ".", + "date_column": "timestamp", + "date_format": "yyyy-MM-dd HH:mm:ss", + "train_size": 80, "shuffle": true, - "lineSeparator": ",", - "decimalSeparator": ".", - "dateColumn": "timestamp", - "dateFormat": "yyyy-MM-dd HH:mm:ss", - "removedIntervals": [], - "degree": 1, - "interactionOnly": false, - "nanTreatment": "drop", - "startDate": null, - "endDate": null, - "scalerName": "None", - "supportFilters": {} -} + "random_state": 42, + "model_name": "Linear Regression", + "model_type": "linear_regression", + "data_model_kwargs": { + "lag_train": { + "303-WIT-200(Value)": 0 + }, + "lag_val": { + "303-WIT-200(Value)": 0 + }, + "nan_treatment": "drop", + "rem_static_win": false, + "static_threshold": null, + "start_date": null, + "end_date": null, + "support_filters": {}, + "removed_intervals": [] + }, + "model_kwargs": { + "degree": 1, + "interaction_only": false, + "scaler_name": "None" + }, + "opt_params": {} +} \ No newline at end of file diff --git a/docs/test-scenarios/11-linear-regression-static-threshold-custom.json b/docs/test-scenarios/11-linear-regression-static-threshold-custom.json index c9330c3..879f699 100644 --- a/docs/test-scenarios/11-linear-regression-static-threshold-custom.json +++ b/docs/test-scenarios/11-linear-regression-static-threshold-custom.json @@ -1,31 +1,40 @@ { "_description": "Regressão linear com remoção de janelas estáticas e static_threshold customizado", - "experimentName": "test-linear-static-threshold", - "username": "bruno.domingues@aignosi.com.br", - "modelName": "Linear Regression", - "targetVariable": "03CV020/CORRENTE_N_M1_PV(Value)", - "variableColumns": ["303-WIT-200(Value)"], - "lagTrain": {"303-WIT-200(Value)": 0}, - "lagVal": {"303-WIT-200(Value)": 0}, - "remStaticWin": true, - "staticThreshold": 100, - "lowLim": {}, - "uppLim": {}, - "window": 10, - "useScaler": false, - "includeAr": false, - "trainSize": 80, + "experiment_run_id": 1011, + "variable_columns": [ + "303-WIT-200(Value)" + ], + "target_variable": "03CV020/CORRENTE_N_M1_PV(Value)", + "bucket_name": "model-training", + "file_name": "training_data.csv", + "line_separator": ",", + "decimal_separator": ".", + "date_column": "timestamp", + "date_format": "yyyy-MM-dd HH:mm:ss", + "train_size": 80, "shuffle": true, - "lineSeparator": ",", - "decimalSeparator": ".", - "dateColumn": "timestamp", - "dateFormat": "yyyy-MM-dd HH:mm:ss", - "removedIntervals": [], - "degree": 1, - "interactionOnly": false, - "nanTreatment": "drop", - "startDate": null, - "endDate": null, - "scalerName": "None", - "supportFilters": {} -} + "random_state": 42, + "model_name": "Linear Regression", + "model_type": "linear_regression", + "data_model_kwargs": { + "lag_train": { + "303-WIT-200(Value)": 0 + }, + "lag_val": { + "303-WIT-200(Value)": 0 + }, + "nan_treatment": "drop", + "rem_static_win": true, + "static_threshold": 100, + "start_date": null, + "end_date": null, + "support_filters": {}, + "removed_intervals": [] + }, + "model_kwargs": { + "degree": 1, + "interaction_only": false, + "scaler_name": "None" + }, + "opt_params": {} +} \ No newline at end of file diff --git a/docs/test-scenarios/12-angular-test-date-format.json b/docs/test-scenarios/12-angular-test-date-format.json index 096284e..2e70b77 100644 --- a/docs/test-scenarios/12-angular-test-date-format.json +++ b/docs/test-scenarios/12-angular-test-date-format.json @@ -1,31 +1,40 @@ { "_description": "Cenário angular-test-01: CV022 WIT230 com lag e intervalo de datas", - "experimentName": "angular-test-01", - "username": "lucas.kou@aignosi.com.br", - "modelName": "Linear Regression", - "targetVariable": "03CV022/CORRENTE_N_M1_PV(Value)", - "variableColumns": ["303-WIT-230(Value)"], - "lagTrain": {"303-WIT-230(Value)": 3}, - "lagVal": {"303-WIT-230(Value)": 0}, - "remStaticWin": false, - "lowLim": {}, - "uppLim": {}, - "window": 0, - "useScaler": false, - "includeAr": false, - "trainSize": 80, + "experiment_run_id": 1012, + "variable_columns": [ + "303-WIT-230(Value)" + ], + "target_variable": "03CV022/CORRENTE_N_M1_PV(Value)", + "bucket_name": "model-training", + "file_name": "training_data.csv", + "line_separator": ",", + "decimal_separator": ".", + "date_column": "DATA", + "date_format": "dd/MM/yyyy HH:mm:ss", + "train_size": 80, "shuffle": true, - "lineSeparator": ",", - "decimalSeparator": ".", - "dateColumn": "DATA", - "dateFormat": "dd/MM/yyyy HH:mm:ss", - "removedIntervals": [], - "degree": 1, - "interactionOnly": false, - "nanTreatment": "drop", - "startDate": "01/05/2022", - "endDate": "31/07/2022", - "scalerName": "None", - "supportFilters": {}, - "staticThreshold": null -} + "random_state": 42, + "model_name": "Linear Regression", + "model_type": "linear_regression", + "data_model_kwargs": { + "lag_train": { + "303-WIT-230(Value)": 3 + }, + "lag_val": { + "303-WIT-230(Value)": 0 + }, + "nan_treatment": "drop", + "rem_static_win": false, + "static_threshold": null, + "start_date": "01/05/2022", + "end_date": "31/07/2022", + "support_filters": {}, + "removed_intervals": [] + }, + "model_kwargs": { + "degree": 1, + "interaction_only": false, + "scaler_name": "None" + }, + "opt_params": {} +} \ No newline at end of file diff --git a/docs/test-scenarios/13-angular-test-double-date-column.json b/docs/test-scenarios/13-angular-test-double-date-column.json index e94e224..efa44ff 100644 --- a/docs/test-scenarios/13-angular-test-double-date-column.json +++ b/docs/test-scenarios/13-angular-test-double-date-column.json @@ -1,31 +1,40 @@ { "_description": "Cenário angular-test: CV022 WIT230 com ficheiro double date column e intervalo curto (00:00 a 00:05)", - "experimentName": "angular-test", - "username": "lucas.kou@aignosi.com.br", - "modelName": "Linear Regression", - "targetVariable": "03CV022/CORRENTE_N_M1_PV(Value)", - "variableColumns": ["303-WIT-230(Value)"], - "lagTrain": {"303-WIT-230(Value)": 0}, - "lagVal": {"303-WIT-230(Value)": 0}, - "remStaticWin": false, - "lowLim": {}, - "uppLim": {}, - "window": 0, - "useScaler": false, - "includeAr": false, - "trainSize": 80, + "experiment_run_id": 1013, + "variable_columns": [ + "303-WIT-230(Value)" + ], + "target_variable": "03CV022/CORRENTE_N_M1_PV(Value)", + "bucket_name": "model-training", + "file_name": "training_data.csv", + "line_separator": ",", + "decimal_separator": ".", + "date_column": "DATA", + "date_format": "dd/MM/yyyy HH:mm:ss", + "train_size": 80, "shuffle": true, - "lineSeparator": ",", - "decimalSeparator": ".", - "dateColumn": "DATA", - "dateFormat": "dd/MM/yyyy HH:mm:ss", - "removedIntervals": [], - "degree": 1, - "interactionOnly": false, - "nanTreatment": "drop", - "startDate": "01/05/2022 00:00:00", - "endDate": "01/05/2022 00:05:10", - "scalerName": "None", - "supportFilters": {}, - "staticThreshold": null -} + "random_state": 42, + "model_name": "Linear Regression", + "model_type": "linear_regression", + "data_model_kwargs": { + "lag_train": { + "303-WIT-230(Value)": 0 + }, + "lag_val": { + "303-WIT-230(Value)": 0 + }, + "nan_treatment": "drop", + "rem_static_win": false, + "static_threshold": null, + "start_date": "01/05/2022 00:00:00", + "end_date": "01/05/2022 00:05:10", + "support_filters": {}, + "removed_intervals": [] + }, + "model_kwargs": { + "degree": 1, + "interaction_only": false, + "scaler_name": "None" + }, + "opt_params": {} +} \ No newline at end of file diff --git a/docs/test-scenarios/14-angular-test-polynomial-support-filters.json b/docs/test-scenarios/14-angular-test-polynomial-support-filters.json index d4af9cd..1618f2e 100644 --- a/docs/test-scenarios/14-angular-test-polynomial-support-filters.json +++ b/docs/test-scenarios/14-angular-test-polynomial-support-filters.json @@ -1,42 +1,51 @@ { "_description": "Cenário angular-test-01: regressão polinomial degree 4, scaler, support filters em 303-WIT-200", - "experimentName": "angular-test-01", - "username": "lucas.kou@aignosi.com.br", - "modelName": "Polynomial Regression", - "targetVariable": "03CV020/CORRENTE_N_M1_PV(Value)", - "variableColumns": ["303-WIT-200(Value)"], - "lagTrain": {"303-WIT-200(Value)": 0}, - "lagVal": {"303-WIT-200(Value)": 0}, - "remStaticWin": false, - "lowLim": {}, - "uppLim": {}, - "window": 0, - "useScaler": true, - "includeAr": false, - "trainSize": 80, + "experiment_run_id": 1014, + "variable_columns": [ + "303-WIT-200(Value)" + ], + "target_variable": "03CV020/CORRENTE_N_M1_PV(Value)", + "bucket_name": "model-training", + "file_name": "training_data.csv", + "line_separator": ",", + "decimal_separator": ".", + "date_column": "timestamp", + "date_format": "yyyy-MM-dd HH:mm:ss", + "train_size": 80, "shuffle": true, - "lineSeparator": ",", - "decimalSeparator": ".", - "dateColumn": "timestamp", - "dateFormat": "yyyy-MM-dd HH:mm:ss", - "removedIntervals": [], - "degree": 4, - "interactionOnly": false, - "nanTreatment": "drop", - "startDate": "2025-06-02 00:00:05", - "endDate": "2025-06-06 15:02:01", - "scalerName": "Standard Scaler", - "supportFilters": { - "303-WIT-200(Value)": { - "upper_line": { - "intercept": 40.400002, - "angle": 0 - }, - "lower_line": { - "intercept": 30.5, - "angle": 0 + "random_state": 42, + "model_name": "Polynomial Regression", + "model_type": "polynomial_regression", + "data_model_kwargs": { + "lag_train": { + "303-WIT-200(Value)": 0 + }, + "lag_val": { + "303-WIT-200(Value)": 0 + }, + "nan_treatment": "drop", + "rem_static_win": false, + "static_threshold": null, + "start_date": "2025-06-02 00:00:05", + "end_date": "2025-06-06 15:02:01", + "support_filters": { + "303-WIT-200(Value)": { + "upper_line": { + "intercept": 40.400002, + "angle": 0 + }, + "lower_line": { + "intercept": 30.5, + "angle": 0 + } } - } + }, + "removed_intervals": [] }, - "staticThreshold": null -} + "model_kwargs": { + "degree": 4, + "interaction_only": false, + "scaler_name": "Standard Scaler" + }, + "opt_params": {} +} \ No newline at end of file diff --git a/e2e/__init__.py b/e2e/__init__.py new file mode 100644 index 0000000..d02a73a --- /dev/null +++ b/e2e/__init__.py @@ -0,0 +1,3 @@ +""" +End-to-end tests for the Model Manager Temporal workflows. +""" diff --git a/e2e/conftest.py b/e2e/conftest.py new file mode 100644 index 0000000..0213782 --- /dev/null +++ b/e2e/conftest.py @@ -0,0 +1,581 @@ +""" +Pytest configuration and fixtures for E2E tests. + +All external dependencies use real services: +- PostgreSQL: testcontainers (postgres:15) +- MinIO: testcontainers (minio) +- MongoDB: testcontainers (mongo:7) +- MLflow: local filesystem tracking (no network) +- Gitea: testcontainers generic container (gitea/gitea:latest), + seeded with model-plugin-warehouse files via REST API +- Temporal: in-memory WorkflowEnvironment (time-skipping) +""" + +from concurrent.futures import ThreadPoolExecutor +import base64 +import csv +import io +import os +import shutil +import tempfile +import time +import uuid +from pathlib import Path +from unittest.mock import MagicMock + +import mlflow +import pytest +import pytest_asyncio +import requests +from minio import Minio +from sqlalchemy import create_engine, text +from testcontainers.core.container import DockerContainer +from testcontainers.minio import MinioContainer +from testcontainers.mongodb import MongoDbContainer +from testcontainers.postgres import PostgresContainer +from temporalio.testing import WorkflowEnvironment +from temporalio.worker import Worker + +from model_manager.activities.activities import Activities +from model_manager.workflows.cleanup_files import CleanupFiles +from model_manager.workflows.train_model import TrainModel +from sientia_do.notifications.handlers import CoreNotificationHandler +from sientia_do.observability.metrics_controller import MetricsController +from sientia_model.model_repository.plugin_store import PluginStore + +# --------------------------------------------------------------------------- +# Paths +# --------------------------------------------------------------------------- +_WAREHOUSE_ROOT = Path( + '/home/grezewave/Documents/projects/sientia/model-plugin-warehouse' +) + +# CSV training data: columns must match the variable_columns and target_variable +# used across all test scenarios. +_TRAIN_CSV_COLUMNS = [ + 'timestamp', + '303-WIT-200(Value)', + '03CV020/CORRENTE_N_M1_PV(Value)', + '303-WIT-230(Value)', + '03CV022/CORRENTE_N_M1_PV(Value)', +] +_MINIO_BUCKET = 'model-training' +_MINIO_OBJECT = 'training_data.csv' + +# --------------------------------------------------------------------------- +# Helpers – CSV generation +# --------------------------------------------------------------------------- + +def _build_training_csv() -> bytes: + """ + Generate a 150-row CSV with all columns needed by test scenarios. + + The numeric values cycle deterministically so lags and static-window + removal always find enough rows in both train and validation splits. + """ + output = io.StringIO() + writer = csv.writer(output) + writer.writerow(_TRAIN_CSV_COLUMNS) + for i in range(150): + ts = f'2025-06-{(i // 24) + 2:02d} {i % 24:02d}:00:00' + wit200 = round(30.0 + (i % 20) * 0.5, 2) + cv020 = round(100.0 + (i % 15) * 0.3, 2) + wit230 = round(25.0 + (i % 18) * 0.4, 2) + cv022 = round(90.0 + (i % 12) * 0.25, 2) + writer.writerow([ts, wit200, cv020, wit230, cv022]) + return output.getvalue().encode('utf-8') + + +def _build_training_csv_dd_mm_yyyy() -> bytes: + """ + Generate a 150-row CSV with dd/MM/yyyy HH:mm:ss timestamps and + a DATA column header, for scenarios 12/13 that use a different date format. + """ + output = io.StringIO() + writer = csv.writer(output) + writer.writerow([ + 'DATA', + '303-WIT-230(Value)', + '03CV022/CORRENTE_N_M1_PV(Value)', + ]) + for i in range(150): + day = (i % 30) + 1 + ts = f'{day:02d}/05/2022 {i % 24:02d}:00:00' + wit230 = round(25.0 + (i % 18) * 0.4, 2) + cv022 = round(90.0 + (i % 12) * 0.25, 2) + writer.writerow([ts, wit230, cv022]) + return output.getvalue().encode('utf-8') + + +# --------------------------------------------------------------------------- +# Helpers – Gitea seed +# --------------------------------------------------------------------------- + +def _wait_for_gitea(base_url: str, timeout: int = 120) -> None: + """Poll Gitea until it responds to HTTP requests.""" + deadline = time.time() + timeout + last_err = None + while time.time() < deadline: + try: + resp = requests.get(f'{base_url}/', timeout=3) + if resp.status_code in (200, 404, 302): + return + except Exception as e: + last_err = e + time.sleep(2) + raise TimeoutError(f'Gitea did not start within {timeout}s at {base_url}. Last error: {last_err}') + + +def _gitea_api(method: str, url: str, auth: tuple, **kwargs) -> requests.Response: + resp = requests.request(method, url, auth=auth, timeout=30, **kwargs) + try: + resp.raise_for_status() + except requests.exceptions.HTTPError as e: + raise RuntimeError(f"Gitea API error {resp.status_code}: {resp.text}") from e + return resp + + +def _seed_gitea(base_url: str, admin_user: str, admin_pass: str) -> None: + """ + Create a fictitious model-store repository with dummy models. + """ + auth = (admin_user, admin_pass) + api = f'{base_url}/api/v1' + + # Create repository + _gitea_api( + 'POST', f'{api}/user/repos', auth, + json={'name': 'model-store', 'private': False, 'auto_init': False}, + ) + + # Root index.yaml + root_index = """ +store_name: "E2E Test Store" +version: 1 +models: + - name: "linear_regression" + version: 1 + runtime: "basic" + - name: "polynomial_regression" + version: 1 + runtime: "basic" +runtimes: + basic: + version: "1.0.0" + libraries: + - name: "pandas" + - name: "numpy" +""" + + # Model index.yaml (shared for all dummies) + model_index = """ +name: "{model_name}" +version: 1 +runtime: "basic" +path: "wrapper.py" +class: "DummyWrapper" +model: + class: "DummyModel" + path: "model_logic.py" + external: false +data_model: + class: "DummyTransformer" + path: "model_logic.py" + external: false +""" + + # schemas.yaml + schemas_yaml = """ +model: + type: object + properties: {} +data_model: + type: object + properties: {} +opt_params: + type: object + properties: {} +""" + + # wrapper.py + wrapper_py = """ +from sientia_model.wrappers.sientia_model import SientiaModel +import pandas as pd +import numpy as np +from typing import Any + +class DummyWrapper(SientiaModel): + def _predict(self, data: pd.DataFrame) -> tuple[pd.DataFrame, dict[str, Any]]: + self._log("info", f"Predicting dummy model for {self.model_type}") + # Return a simple prediction (mean or 0.5) to allow metrics computation + preds = pd.DataFrame({self.target: [0.5] * len(data)}, index=data.index) + return preds, {} + + def _transform(self, data: pd.DataFrame) -> tuple[pd.DataFrame, dict[str, Any]]: + return data, {} + + def _train_transformer(self, train_data: pd.DataFrame, val_data: pd.DataFrame) -> None: + pass + + def _train_model(self, x: pd.DataFrame, y: pd.DataFrame, x_val: pd.DataFrame | None = None, y_val: pd.DataFrame | None = None) -> None: + self.target = y.columns[0] + + def _retrain_transformer(self, data: pd.DataFrame) -> None: + pass + + def _retrain_model(self, x: pd.DataFrame, y: pd.DataFrame | None) -> None: + pass +""" + + # model_logic.py + model_logic_py = """ +class DummyModel: + def __init__(self, **kwargs): + pass + +class DummyTransformer: + def __init__(self, **kwargs): + pass +""" + + def push_file(path: str, content: str): + encoded = base64.b64encode(content.encode()).decode() + _gitea_api( + 'POST', + f'{api}/repos/{admin_user}/model-store/contents/{path}', + auth, + json={'message': f'seed: {path}', 'content': encoded}, + ) + + # Push root index + push_file('index.yaml', root_index) + + # Push files for both models used in tests + for model_name in ['linear_regression', 'polynomial_regression']: + prefix = f'models/{model_name}' + push_file(f'{prefix}/index.yaml', model_index.format(model_name=model_name)) + push_file(f'{prefix}/schemas.yaml', schemas_yaml) + push_file(f'{prefix}/wrapper.py', wrapper_py) + push_file(f'{prefix}/model_logic.py', model_logic_py) + push_file(f'{prefix}/__init__.py', "") + + # Push runtime + push_file('runtime/basic.yaml', 'name: basic\nversion: "1.0.0"\nlibraries: []') + + +# --------------------------------------------------------------------------- +# Session-scoped containers +# --------------------------------------------------------------------------- + +@pytest_asyncio.fixture(scope='session') +def postgres_container(): + """PostgreSQL 15 container for experiment_run table.""" + container = PostgresContainer('postgres:15') + container.start() + yield container + container.stop() + + +@pytest_asyncio.fixture(scope='session') +def minio_container(): + """MinIO container for training CSV storage.""" + container = MinioContainer() + container.start() + yield container + container.stop() + + +@pytest_asyncio.fixture(scope='session') +def mongodb_container(): + """MongoDB container for CoreNotificationHandler.""" + container = MongoDbContainer('mongo:7') + container.start() + yield container + container.stop() + + +@pytest_asyncio.fixture(scope='session') +def gitea_container(): + """ + Gitea container seeded with the model-plugin-warehouse files. + + The container starts with INSTALL_LOCK so no setup wizard is needed. + An admin user is created via Gitea's CLI before the HTTP API is used. + """ + admin_user = 'gitea_admin' + admin_pass = 'gitea_admin_pass' # noqa: S105 + + container = ( + DockerContainer('gitea/gitea:latest') + .with_env('GITEA__security__INSTALL_LOCK', 'true') + .with_env('GITEA__server__HTTP_PORT', '3000') + .with_env('GITEA__log__LEVEL', 'Warn') + .with_exposed_ports(3000) + ) + container.start() + + port = container.get_exposed_port(3000) + base_url = f'http://localhost:{port}' + + _wait_for_gitea(base_url) + import time + time.sleep(5) # Wait a bit for DB to fully initialize after HTTP is up + + # Create admin user via Gitea CLI inside the container + # Must run after Gitea is fully initialized + gitea_cmd = ( + f'gitea admin user create ' + f'--username {admin_user} ' + f'--password {admin_pass} ' + f'--email admin@test.local ' + f'--admin ' + f'--must-change-password=false' + ) + exec_result = container.exec(f"su git -c '{gitea_cmd}'") + if exec_result.exit_code != 0: + raise RuntimeError(f"Failed to create Gitea admin user: {exec_result.output.decode('utf-8')}") + + _seed_gitea(base_url, admin_user, admin_pass) + + yield { + 'container': container, + 'base_url': base_url, + 'admin_user': admin_user, + 'admin_pass': admin_pass, + } + + container.stop() + + +@pytest_asyncio.fixture(scope='session') +def mlflow_tracking_dir(): + """Local MLflow filesystem tracking directory (no network needed).""" + tmpdir = tempfile.mkdtemp(prefix='mlflow-e2e-') + mlflow.set_tracking_uri(f'file://{tmpdir}') + yield tmpdir + shutil.rmtree(tmpdir, ignore_errors=True) + + +# --------------------------------------------------------------------------- +# Session-scoped: seed MinIO with training CSV +# --------------------------------------------------------------------------- + +@pytest_asyncio.fixture(scope='session', autouse=True) +def upload_training_csv(minio_container, mlflow_tracking_dir): # noqa: ARG001 + """ + Upload training CSV files to the MinIO container before any test runs. + Depends on mlflow_tracking_dir to ensure the MLflow URI is set at session start. + """ + port = minio_container.get_exposed_port(9000) + client = Minio( + f'localhost:{port}', + access_key='minioadmin', + secret_key='minioadmin', + secure=False, + ) + + if not client.bucket_exists(_MINIO_BUCKET): + client.make_bucket(_MINIO_BUCKET) + + # Standard training CSV + csv_bytes = _build_training_csv() + client.put_object( + _MINIO_BUCKET, + _MINIO_OBJECT, + io.BytesIO(csv_bytes), + length=len(csv_bytes), + content_type='text/csv', + ) + + # dd/MM/yyyy format CSV for scenarios 12/13 + alt_csv_bytes = _build_training_csv_dd_mm_yyyy() + client.put_object( + _MINIO_BUCKET, + 'training_data_dd_mm_yyyy.csv', + io.BytesIO(alt_csv_bytes), + length=len(alt_csv_bytes), + content_type='text/csv', + ) + + +# --------------------------------------------------------------------------- +# Function-scoped: database engine + schema setup +# --------------------------------------------------------------------------- + +@pytest_asyncio.fixture +def postgres_engine(postgres_container): + """SQLAlchemy engine connected to the test PostgreSQL container.""" + engine = create_engine(postgres_container.get_connection_url()) + yield engine + engine.dispose() + + +@pytest_asyncio.fixture(autouse=True) +def setup_experiment_run_table(postgres_engine): + """ + Create the experiment_run table before each test and drop it afterwards + to guarantee full isolation between tests. + """ + with postgres_engine.begin() as conn: + conn.execute(text(""" + CREATE TABLE IF NOT EXISTS public.experiment_run ( + id INT PRIMARY KEY, + experiment_name TEXT NOT NULL, + run_name TEXT, + username TEXT, + status TEXT NOT NULL DEFAULT 'ORCHESTRATOR_WAITING_PROC', + error_message TEXT, + created_at TIMESTAMPTZ NOT NULL DEFAULT NOW(), + updated_at TIMESTAMPTZ NOT NULL DEFAULT NOW(), + bucket_name TEXT, + file_name TEXT + ) + """)) + yield + with postgres_engine.begin() as conn: + conn.execute(text('DROP TABLE IF EXISTS public.experiment_run')) + + +# --------------------------------------------------------------------------- +# Mock-only fixtures (no external service equivalent) +# --------------------------------------------------------------------------- + +@pytest_asyncio.fixture +def mock_logger(): + """Minimal logger that prints to stdout (no external observability needed).""" + def _log(msg, *args, **kwargs): # noqa: ARG001 + print(f'[LOG] {msg}') + + logger = MagicMock() + for method in ('info', 'debug', 'error', 'warning', 'critical', + 'custom_info', 'custom_debug', 'custom_error', + 'custom_warning', 'custom_critical'): + setattr(logger, method, MagicMock(side_effect=_log)) + logger.base_logger = MagicMock() + return logger + + +@pytest_asyncio.fixture +def mock_metrics_controller(mock_logger): + """Real MetricsController backed by the mock logger.""" + return MetricsController(logger=mock_logger) + + +# --------------------------------------------------------------------------- +# Real application fixtures +# --------------------------------------------------------------------------- + +@pytest_asyncio.fixture +def notification_handler(mongodb_container, mock_logger): + """ + Real CoreNotificationHandler connected to the MongoDB testcontainer. + """ + connection_url = mongodb_container.get_connection_url() + handler = CoreNotificationHandler( + connection_string=connection_url, + database='test_notifications', + logger=mock_logger, + project_name='model-manager-e2e', + ) + yield handler + handler.shutdown() + + +@pytest_asyncio.fixture +def plugin_store(gitea_container, mock_logger, mock_metrics_controller, notification_handler): + """ + Real PluginStore pointed at the Gitea testcontainer. + cache_ttl_seconds=0 forces a fresh download every test. + """ + store = PluginStore( + base_url=gitea_container['base_url'], + owner=gitea_container['admin_user'], + repo='model-store', + username=gitea_container['admin_user'], + password=gitea_container['admin_pass'], + cache_ttl_seconds=0, + logger=mock_logger, + notification_handler=notification_handler, + metrics_controller=mock_metrics_controller, + ) + yield store + + +@pytest_asyncio.fixture +def test_activities( + postgres_container, + minio_container, + mlflow_tracking_dir, # noqa: ARG001 – ensures MLflow URI is set + plugin_store, + mock_logger, + notification_handler, + mock_metrics_controller, +): + """ + Real Activities instance wired to all testcontainers. + """ + pg_port = postgres_container.get_exposed_port(5432) + minio_port = minio_container.get_exposed_port(9000) + + activities = Activities( + postgres_config={ + 'host': 'localhost', + 'port': int(pg_port), + 'user': 'test', + 'password': 'test', + 'dbname': 'test', + 'min_connections': 1, + 'max_connections': 5, + }, + mlflow_config={ + 'url': mlflow.get_tracking_uri(), + 'username': None, + 'password': None, + }, + minio_config={ + 'endpoint_url': f'http://localhost:{minio_port}', + 'access_key': 'minioadmin', + 'secret_key': 'minioadmin', + 'use_ssl': False, + 'default_bucket': _MINIO_BUCKET, + }, + plugin_store=plugin_store, + logger=mock_logger, + notification_handler=notification_handler, + metrics_controller=mock_metrics_controller, + ) + yield activities + activities.shutdown() + + +def _activity_list(activities: Activities) -> list: + return [ + activities.update_experiment_run, + activities.load_model_metadata, + activities.validate_train_params, + activities.train_model, + activities.cleanup_resources, + activities.cleanup_temp_directories, + ] + + +@pytest_asyncio.fixture(scope='function') +async def temporal_test_env(): + """In-memory Temporal environment with time-skipping.""" + env = await WorkflowEnvironment.start_time_skipping() + async with env: + yield env + + +@pytest_asyncio.fixture(scope='function') +async def temporal_worker(temporal_test_env, test_activities): + """Temporal worker registered with all workflows and activities.""" + with ThreadPoolExecutor() as executor: + async with Worker( + temporal_test_env.client, + task_queue='test-queue', + workflows=[TrainModel, CleanupFiles], + activities=_activity_list(test_activities), + activity_executor=executor, + ) as worker: + yield worker diff --git a/e2e/helpers.py b/e2e/helpers.py new file mode 100644 index 0000000..7d7cf47 --- /dev/null +++ b/e2e/helpers.py @@ -0,0 +1,196 @@ +""" +Shared helpers for E2E tests (Temporal workflows + PostgreSQL). +""" + +import asyncio +from datetime import datetime +from typing import Any + +import pytest +from sqlalchemy import text +from sqlalchemy.engine import Engine + + +async def start_and_await_workflow( + client, + workflow_run, + input_data: dict, + workflow_id: str, + timeout: float = 120.0, +): + """ + Start a Temporal workflow and wait for its result. + + Args: + client: Temporal client from WorkflowEnvironment. + workflow_run: Workflow run method (e.g. TrainModel.run). + input_data: Workflow input payload. + workflow_id: Unique workflow id. + timeout: Max seconds to wait for completion. + + Returns: + Workflow result value. + """ + handle = await client.start_workflow( + workflow_run, + input_data, + id=workflow_id, + task_queue='test-queue', + ) + return await asyncio.wait_for(handle.result(), timeout=timeout) + + +def make_workflow_id(prefix: str) -> str: + """Build a unique workflow id using a prefix and current timestamp.""" + return f'{prefix}-{datetime.now().timestamp()}' + + +def insert_experiment_run( + engine: Engine, + experiment_run_id: int, + experiment_name: str = 'test_experiment', + status: str = 'ORCHESTRATOR_WAITING_PROC', + bucket_name: str = 'model-training', + file_name: str = 'training_data.csv', +) -> None: + """ + Insert a minimal experiment_run row to satisfy foreign-key-style lookups. + + Args: + engine: SQLAlchemy engine connected to the test database. + experiment_run_id: Primary key for the row. + experiment_name: Human-readable experiment name. + status: Initial status string. + bucket_name: MinIO bucket name. + file_name: Training file name inside the bucket. + """ + with engine.begin() as conn: + conn.execute( + text(""" + INSERT INTO public.experiment_run + (id, experiment_name, status, bucket_name, file_name) + VALUES + (:id, :experiment_name, :status, :bucket_name, :file_name) + ON CONFLICT (id) DO NOTHING + """), + { + 'id': experiment_run_id, + 'experiment_name': experiment_name, + 'status': status, + 'bucket_name': bucket_name, + 'file_name': file_name, + }, + ) + + +def assert_experiment_status( + engine: Engine, + experiment_run_id: int, + expected_status: str, +) -> None: + """ + Assert the final status of an experiment_run row. + + Args: + engine: SQLAlchemy engine. + experiment_run_id: Row primary key. + expected_status: Expected status string. + """ + with engine.connect() as conn: + row = conn.execute( + text('SELECT status FROM public.experiment_run WHERE id = :id'), + {'id': experiment_run_id}, + ).fetchone() + + assert row is not None, ( + f'No experiment_run row found for id={experiment_run_id}' + ) + assert row[0] == expected_status, ( + f'Expected status={expected_status!r}, got {row[0]!r} ' + f'for experiment_run id={experiment_run_id}' + ) + + +def assert_experiment_run_name_set( + engine: Engine, + experiment_run_id: int, +) -> None: + """Assert that run_name is not null/empty after a successful training.""" + with engine.connect() as conn: + row = conn.execute( + text('SELECT run_name FROM public.experiment_run WHERE id = :id'), + {'id': experiment_run_id}, + ).fetchone() + + assert row is not None, ( + f'No experiment_run row found for id={experiment_run_id}' + ) + assert row[0] is not None and row[0].strip() != '', ( + f'Expected run_name to be set for experiment_run id={experiment_run_id}, got {row[0]!r}' + ) + + +def assert_experiment_error( + engine: Engine, + experiment_run_id: int, + expected_status: str, + error_substr: str, +) -> None: + """ + Assert status and that error_message contains a given substring. + + Args: + engine: SQLAlchemy engine. + experiment_run_id: Row primary key. + expected_status: Expected status string. + error_substr: Substring that must appear in error_message. + """ + with engine.connect() as conn: + row = conn.execute( + text( + 'SELECT status, error_message FROM public.experiment_run WHERE id = :id' + ), + {'id': experiment_run_id}, + ).fetchone() + + assert row is not None, ( + f'No experiment_run row found for id={experiment_run_id}' + ) + assert row[0] == expected_status, ( + f'Expected status={expected_status!r}, got {row[0]!r}' + ) + assert row[1] is not None and error_substr.lower() in row[1].lower(), ( + f'Expected error_message to contain {error_substr!r}, got {row[1]!r}' + ) + + +def assert_no_experiment_row(engine: Engine, experiment_run_id: int) -> None: + """Assert that no experiment_run row exists for the given id.""" + with engine.connect() as conn: + count = conn.execute( + text('SELECT COUNT(*) FROM public.experiment_run WHERE id = :id'), + {'id': experiment_run_id}, + ).scalar() + assert count == 0, ( + f'Expected no experiment_run row for id={experiment_run_id}, found {count}' + ) + + +def load_scenario(scenario_filename: str) -> dict[str, Any]: + """ + Load a test scenario JSON file from docs/test-scenarios/. + + Args: + scenario_filename: Filename without path (e.g. '01-linear-regression-basic.json'). + + Returns: + dict: Parsed scenario payload. + """ + import json + from pathlib import Path + + scenario_path = ( + Path(__file__).parent.parent / 'docs' / 'test-scenarios' / scenario_filename + ) + with open(scenario_path) as f: + return json.load(f) diff --git a/e2e/scenarios.md b/e2e/scenarios.md new file mode 100644 index 0000000..282842d --- /dev/null +++ b/e2e/scenarios.md @@ -0,0 +1,47 @@ +# E2E Test Scenarios + +This document maps the workflow scenarios tested in the E2E suite to their corresponding JSON input files and expected behaviors. + +## 1. TrainModel Workflow (`test_train_model_workflow.py`) + +### 1.1 Happy Paths (Successful execution) + +| Test Function | Input JSON | Expected Status | Description | +|---|---|---|---| +| `test_scenario_1_1_1_linear_regression_basic` | `01-linear-regression-basic.json` | `TRAINING_SUCCESS` | Basic linear regression without scaler. Verifies end-to-end pipeline. | +| `test_scenario_1_1_2_polynomial_regression_degree2_with_scaler` | `03-polynomial-regression-degree2.json` | `TRAINING_SUCCESS` | Polynomial regression (degree 2) with Standard Scaler. | +| `test_scenario_1_1_3_linear_regression_with_lags` | `05-linear-regression-with-lags.json` | `TRAINING_SUCCESS` | Linear regression with `lag_train`/`lag_val` per variable. | +| `test_scenario_1_1_4_linear_regression_nan_interpolation` | `06-linear-regression-nan-interpolation.json` | `TRAINING_SUCCESS` | Linear regression with `nan_treatment='linear interpolation'`. | +| `test_scenario_1_1_5_linear_regression_with_limits` | `08-linear-regression-with-limits.json` | `TRAINING_SUCCESS` | Linear regression with `support_filters` (min/max limits per variable). | +| `test_scenario_1_1_6_polynomial_degree2_scaler_and_lags` | `09-polynomial-degree2-with-scaler-and-lags.json` | `TRAINING_SUCCESS` | Polynomial regression (degree 2), Standard Scaler, and lags. | +| `test_scenario_1_1_7_static_window_removal` | `11-linear-regression-static-threshold-custom.json` | `TRAINING_SUCCESS` | Linear regression with `rem_static_win=true`, `window`, and `static_threshold`. | +| `test_scenario_1_1_8_polynomial_with_support_filters` | `14-angular-test-polynomial-support-filters.json` | `TRAINING_SUCCESS` | Polynomial regression (degree 4), Standard Scaler, and support filters. | + +### 1.2 Error Paths + +| Test Function | Input JSON | Expected Status | Description | +|---|---|---|---| +| `test_scenario_1_2_1_minio_file_not_found` | `01-linear-regression-basic.json` | `TRAINING_ERROR` | MinIO file does not exist. Workflow fails during file download. | +| `test_scenario_1_2_2_experiment_run_id_not_in_db` | `01-linear-regression-basic.json` | N/A (raises Exception) | `experiment_run_id` does not exist in DB. Workflow fails immediately on status update attempt. | + +## 2. Parameter Validation (`test_train_model_validation.py`) + +These scenarios test the business rule validations inside `validate_train_params`. All are expected to terminate with `ORCHESTRATOR_VALIDATION_ERROR`. + +| Test Function | Modification | Expected Error Substring | +|---|---|---| +| `test_scenario_2_1_1_train_size_out_of_range` | `train_size = 5` | `'train_size'` | +| `test_scenario_2_1_2_empty_variable_columns` | `variable_columns = []` | `'variable_columns'` | +| `test_scenario_2_1_3_invalid_date_format` | `date_format = 'INVALID'` | `'date_format'` | +| `test_scenario_2_1_4_whitespace_only_model_name` | `model_name = ' '` | `'model_name'` | +| `test_scenario_2_1_5_unknown_model_type` | `model_type = 'totally_unknown_model'` | `'totally_unknown_model'` | +| `test_scenario_2_1_6_missing_target_variable` | `target_variable = ''` | `'target_variable'` | +| `test_scenario_2_1_7_missing_experiment_run_id` | Missing `experiment_run_id` | N/A (raises ValueError immediately) | + +## 3. CleanupFiles Workflow (`test_cleanup_files_workflow.py`) + +| Test Function | Description | +|---|---| +| `test_scenario_3_1_1_cleanup_with_no_temp_dirs` | Temp directory is empty. Activity completes without error. | +| `test_scenario_3_1_2_cleanup_removes_old_temp_dirs` | Two stale timestamped directories are removed. | +| `test_scenario_3_1_3_cleanup_nonexistent_temp_path` | Target path does not exist. Handled gracefully without error. | diff --git a/e2e/test_cleanup_files_workflow.py b/e2e/test_cleanup_files_workflow.py new file mode 100644 index 0000000..4f37932 --- /dev/null +++ b/e2e/test_cleanup_files_workflow.py @@ -0,0 +1,109 @@ +""" +End-to-end tests for CleanupFiles workflow. + +Covers scenarios 3.x: cleanup of temporary local directories. +""" + +import os +import shutil +import tempfile + +import pytest +import pytest_asyncio +from temporalio.testing import WorkflowEnvironment +from temporalio.worker import Worker + +from e2e.helpers import make_workflow_id, start_and_await_workflow +from model_manager.workflows.cleanup_files import CleanupFiles + + +@pytest.mark.asyncio +@pytest.mark.integration +async def test_scenario_3_1_1_cleanup_with_no_temp_dirs( + temporal_test_env: WorkflowEnvironment, + temporal_worker: Worker, + tmp_path, +): + """Scenario 3.1.1 – Cleanup when the temp directory is empty. + + The cleanup_temp_directories activity should complete without error + and the workflow should finish successfully. + """ + # Use an empty temp directory as the reports path + empty_dir = tmp_path / 'reports_temp' + empty_dir.mkdir() + + result = await start_and_await_workflow( + temporal_test_env.client, + CleanupFiles.run, + {'temp_path': str(empty_dir)}, + make_workflow_id('test-s3-1-1'), + ) + + # Workflow returns None on success + assert result is None + + +@pytest.mark.asyncio +@pytest.mark.integration +async def test_scenario_3_1_2_cleanup_removes_old_temp_dirs( + temporal_test_env: WorkflowEnvironment, + temporal_worker: Worker, + tmp_path, +): + """Scenario 3.1.2 – Cleanup removes stale subdirectories from the temp dir. + + Creates two subdirectories with timestamp suffixes inside the reports + temp directory and verifies the activity removes them. + """ + reports_dir = tmp_path / 'reports_temp' + reports_dir.mkdir() + + # Create two stale run directories + stale1 = reports_dir / 'run-1234567890' + stale2 = reports_dir / 'run-9876543210' + stale1.mkdir() + stale2.mkdir() + (stale1 / 'model.pkl').write_bytes(b'fake-model-data') + (stale2 / 'report.json').write_bytes(b'{"status": "old"}') + + result = await start_and_await_workflow( + temporal_test_env.client, + CleanupFiles.run, + {'temp_path': str(reports_dir)}, + make_workflow_id('test-s3-1-2'), + ) + + assert result is None + + # The activity should have cleaned up the stale directories + remaining = list(reports_dir.iterdir()) + assert len(remaining) == 0, ( + f'Expected all stale dirs to be removed, but found: {remaining}' + ) + + +@pytest.mark.asyncio +@pytest.mark.integration +async def test_scenario_3_1_3_cleanup_nonexistent_temp_path( + temporal_test_env: WorkflowEnvironment, + temporal_worker: Worker, + tmp_path, +): + """Scenario 3.1.3 – Cleanup with a temp_path that does not exist. + + The activity must handle a missing directory gracefully without + raising an unhandled exception, since the directory may have already + been cleaned by a previous run. + """ + nonexistent = str(tmp_path / 'does_not_exist' / 'reports') + + # Should not raise — the activity is expected to handle a missing path + result = await start_and_await_workflow( + temporal_test_env.client, + CleanupFiles.run, + {'temp_path': nonexistent}, + make_workflow_id('test-s3-1-3'), + ) + + assert result is None diff --git a/e2e/test_train_model_validation.py b/e2e/test_train_model_validation.py new file mode 100644 index 0000000..b2a7002 --- /dev/null +++ b/e2e/test_train_model_validation.py @@ -0,0 +1,232 @@ +""" +End-to-end tests for TrainModel parameter validation paths. + +Covers scenarios 2.1.x: workflows that must terminate with +ORCHESTRATOR_VALIDATION_ERROR due to invalid parameter values. +""" + +import pytest +import pytest_asyncio +from temporalio.testing import WorkflowEnvironment +from temporalio.worker import Worker + +from e2e.helpers import ( + assert_experiment_error, + insert_experiment_run, + load_scenario, + make_workflow_id, + start_and_await_workflow, +) +from model_manager.workflows.train_model import TrainModel + +# Base experiment_run ids for validation test scenarios (offset to avoid collision) +_VALIDATION_ID_BASE = 3000 + + +@pytest.mark.asyncio +@pytest.mark.integration +async def test_scenario_2_1_1_train_size_out_of_range( + temporal_test_env: WorkflowEnvironment, + temporal_worker: Worker, + postgres_engine, +): + """Scenario 2.1.1 – train_size=5 violates the 10–100 business rule. + + Expected: workflow updates status → ORCHESTRATOR_VALIDATION_ERROR + and error_message references 'train_size'. + """ + experiment_run_id = _VALIDATION_ID_BASE + 1 + scenario = load_scenario('01-linear-regression-basic.json') + scenario = {**scenario, 'experiment_run_id': experiment_run_id, 'train_size': 5} + insert_experiment_run(postgres_engine, experiment_run_id) + + with pytest.raises(Exception): + await start_and_await_workflow( + temporal_test_env.client, + TrainModel.run, + scenario, + make_workflow_id('test-s2-1-1'), + ) + + assert_experiment_error( + postgres_engine, + experiment_run_id, + expected_status='ORCHESTRATOR_VALIDATION_ERROR', + error_substr='train_size', + ) + + +@pytest.mark.asyncio +@pytest.mark.integration +async def test_scenario_2_1_2_empty_variable_columns( + temporal_test_env: WorkflowEnvironment, + temporal_worker: Worker, + postgres_engine, +): + """Scenario 2.1.2 – variable_columns=[] → ORCHESTRATOR_VALIDATION_ERROR.""" + experiment_run_id = _VALIDATION_ID_BASE + 2 + scenario = load_scenario('01-linear-regression-basic.json') + scenario = {**scenario, 'experiment_run_id': experiment_run_id, 'variable_columns': []} + insert_experiment_run(postgres_engine, experiment_run_id) + + with pytest.raises(Exception): + await start_and_await_workflow( + temporal_test_env.client, + TrainModel.run, + scenario, + make_workflow_id('test-s2-1-2'), + ) + + assert_experiment_error( + postgres_engine, + experiment_run_id, + expected_status='ORCHESTRATOR_VALIDATION_ERROR', + error_substr='variable_columns', + ) + + +@pytest.mark.asyncio +@pytest.mark.integration +async def test_scenario_2_1_3_invalid_date_format( + temporal_test_env: WorkflowEnvironment, + temporal_worker: Worker, + postgres_engine, +): + """Scenario 2.1.3 – date_format='INVALID' is not in the allowed list.""" + experiment_run_id = _VALIDATION_ID_BASE + 3 + scenario = load_scenario('01-linear-regression-basic.json') + scenario = {**scenario, 'experiment_run_id': experiment_run_id, 'date_format': 'INVALID'} + insert_experiment_run(postgres_engine, experiment_run_id) + + with pytest.raises(Exception): + await start_and_await_workflow( + temporal_test_env.client, + TrainModel.run, + scenario, + make_workflow_id('test-s2-1-3'), + ) + + assert_experiment_error( + postgres_engine, + experiment_run_id, + expected_status='ORCHESTRATOR_VALIDATION_ERROR', + error_substr='date_format', + ) + + +@pytest.mark.asyncio +@pytest.mark.integration +async def test_scenario_2_1_4_whitespace_only_model_name( + temporal_test_env: WorkflowEnvironment, + temporal_worker: Worker, + postgres_engine, +): + """Scenario 2.1.4 – model_name=' ' (whitespace) → ORCHESTRATOR_VALIDATION_ERROR.""" + experiment_run_id = _VALIDATION_ID_BASE + 4 + scenario = load_scenario('01-linear-regression-basic.json') + scenario = {**scenario, 'experiment_run_id': experiment_run_id, 'model_name': ' '} + insert_experiment_run(postgres_engine, experiment_run_id) + + with pytest.raises(Exception): + await start_and_await_workflow( + temporal_test_env.client, + TrainModel.run, + scenario, + make_workflow_id('test-s2-1-4'), + ) + + assert_experiment_error( + postgres_engine, + experiment_run_id, + expected_status='ORCHESTRATOR_VALIDATION_ERROR', + error_substr='model_name', + ) + + +@pytest.mark.asyncio +@pytest.mark.integration +async def test_scenario_2_1_5_unknown_model_type( + temporal_test_env: WorkflowEnvironment, + temporal_worker: Worker, + postgres_engine, +): + """Scenario 2.1.5 – model_type='totally_unknown' → ORCHESTRATOR_VALIDATION_ERROR. + + The PluginStore will not find this model in the Gitea repo, causing + load_model_metadata to fail before validate_train_params is even called. + """ + experiment_run_id = _VALIDATION_ID_BASE + 5 + scenario = load_scenario('01-linear-regression-basic.json') + scenario = { + **scenario, + 'experiment_run_id': experiment_run_id, + 'model_type': 'totally_unknown_model', + } + insert_experiment_run(postgres_engine, experiment_run_id) + + with pytest.raises(Exception): + await start_and_await_workflow( + temporal_test_env.client, + TrainModel.run, + scenario, + make_workflow_id('test-s2-1-5'), + ) + + assert_experiment_error( + postgres_engine, + experiment_run_id, + expected_status='ORCHESTRATOR_VALIDATION_ERROR', + error_substr='totally_unknown_model', + ) + + +@pytest.mark.asyncio +@pytest.mark.integration +async def test_scenario_2_1_6_missing_target_variable( + temporal_test_env: WorkflowEnvironment, + temporal_worker: Worker, + postgres_engine, +): + """Scenario 2.1.6 – target_variable='' (empty string) → ORCHESTRATOR_VALIDATION_ERROR.""" + experiment_run_id = _VALIDATION_ID_BASE + 6 + scenario = load_scenario('01-linear-regression-basic.json') + scenario = {**scenario, 'experiment_run_id': experiment_run_id, 'target_variable': ''} + insert_experiment_run(postgres_engine, experiment_run_id) + + with pytest.raises(Exception): + await start_and_await_workflow( + temporal_test_env.client, + TrainModel.run, + scenario, + make_workflow_id('test-s2-1-6'), + ) + + assert_experiment_error( + postgres_engine, + experiment_run_id, + expected_status='ORCHESTRATOR_VALIDATION_ERROR', + error_substr='target_variable', + ) + + +@pytest.mark.asyncio +@pytest.mark.integration +async def test_scenario_2_1_7_missing_experiment_run_id( + temporal_test_env: WorkflowEnvironment, + temporal_worker: Worker, +): + """Scenario 2.1.7 – experiment_run_id missing → workflow raises ValueError immediately. + + No DB row is inserted because experiment_run_id is mandatory to even + know which row to update. The workflow should raise before any DB call. + """ + scenario = load_scenario('01-linear-regression-basic.json') + scenario = {k: v for k, v in scenario.items() if k != 'experiment_run_id'} + + with pytest.raises(Exception, match='experiment_run_id'): + await start_and_await_workflow( + temporal_test_env.client, + TrainModel.run, + scenario, + make_workflow_id('test-s2-1-7'), + ) diff --git a/e2e/test_train_model_workflow.py b/e2e/test_train_model_workflow.py new file mode 100644 index 0000000..329e156 --- /dev/null +++ b/e2e/test_train_model_workflow.py @@ -0,0 +1,272 @@ +""" +End-to-end tests for TrainModel workflow – main workflow scenarios. + +Covers: + 1.1.x – Happy-path training (various scenarios from docs/test-scenarios/) + 1.2.x – Error paths (MinIO failure, missing DB row) +""" + +import pytest +import pytest_asyncio +from temporalio.testing import WorkflowEnvironment +from temporalio.worker import Worker + +from e2e.helpers import ( + assert_experiment_error, + assert_experiment_run_name_set, + assert_experiment_status, + insert_experiment_run, + load_scenario, + make_workflow_id, + start_and_await_workflow, +) +from model_manager.workflows.train_model import TrainModel + + +# --------------------------------------------------------------------------- +# 1.1 – Happy paths +# --------------------------------------------------------------------------- + +@pytest.mark.asyncio +@pytest.mark.integration +async def test_scenario_1_1_1_linear_regression_basic( + temporal_test_env: WorkflowEnvironment, + temporal_worker: Worker, + postgres_engine, +): + """Scenario 1.1.1 – Linear Regression Basic (cenário 01). + + Validates the complete training pipeline end-to-end: + load_model_metadata → validate_train_params → train_model → + update_experiment_run (TRAINING_SUCCESS). + """ + scenario = load_scenario('01-linear-regression-basic.json') + experiment_run_id = scenario['experiment_run_id'] + insert_experiment_run(postgres_engine, experiment_run_id) + + await start_and_await_workflow( + temporal_test_env.client, + TrainModel.run, + scenario, + make_workflow_id('test-s1-1-1'), + ) + + assert_experiment_status(postgres_engine, experiment_run_id, 'TRAINING_SUCCESS') + assert_experiment_run_name_set(postgres_engine, experiment_run_id) + + +@pytest.mark.asyncio +@pytest.mark.integration +async def test_scenario_1_1_2_polynomial_regression_degree2_with_scaler( + temporal_test_env: WorkflowEnvironment, + temporal_worker: Worker, + postgres_engine, +): + """Scenario 1.1.2 – Polynomial Regression Degree 2 with Standard Scaler (cenário 03).""" + scenario = load_scenario('03-polynomial-regression-degree2.json') + experiment_run_id = scenario['experiment_run_id'] + insert_experiment_run(postgres_engine, experiment_run_id) + + await start_and_await_workflow( + temporal_test_env.client, + TrainModel.run, + scenario, + make_workflow_id('test-s1-1-2'), + ) + + assert_experiment_status(postgres_engine, experiment_run_id, 'TRAINING_SUCCESS') + assert_experiment_run_name_set(postgres_engine, experiment_run_id) + + +@pytest.mark.asyncio +@pytest.mark.integration +async def test_scenario_1_1_3_linear_regression_with_lags( + temporal_test_env: WorkflowEnvironment, + temporal_worker: Worker, + postgres_engine, +): + """Scenario 1.1.3 – Linear Regression with lag_train/lag_val per variable (cenário 05).""" + scenario = load_scenario('05-linear-regression-with-lags.json') + experiment_run_id = scenario['experiment_run_id'] + insert_experiment_run(postgres_engine, experiment_run_id) + + await start_and_await_workflow( + temporal_test_env.client, + TrainModel.run, + scenario, + make_workflow_id('test-s1-1-3'), + ) + + assert_experiment_status(postgres_engine, experiment_run_id, 'TRAINING_SUCCESS') + assert_experiment_run_name_set(postgres_engine, experiment_run_id) + + +@pytest.mark.asyncio +@pytest.mark.integration +async def test_scenario_1_1_4_linear_regression_nan_interpolation( + temporal_test_env: WorkflowEnvironment, + temporal_worker: Worker, + postgres_engine, +): + """Scenario 1.1.4 – nan_treatment='linear interpolation' (cenário 06).""" + scenario = load_scenario('06-linear-regression-nan-interpolation.json') + experiment_run_id = scenario['experiment_run_id'] + insert_experiment_run(postgres_engine, experiment_run_id) + + await start_and_await_workflow( + temporal_test_env.client, + TrainModel.run, + scenario, + make_workflow_id('test-s1-1-4'), + ) + + assert_experiment_status(postgres_engine, experiment_run_id, 'TRAINING_SUCCESS') + + +@pytest.mark.asyncio +@pytest.mark.integration +async def test_scenario_1_1_5_linear_regression_with_limits( + temporal_test_env: WorkflowEnvironment, + temporal_worker: Worker, + postgres_engine, +): + """Scenario 1.1.5 – support_filters with min/max limits per variable (cenário 08).""" + scenario = load_scenario('08-linear-regression-with-limits.json') + experiment_run_id = scenario['experiment_run_id'] + insert_experiment_run(postgres_engine, experiment_run_id) + + await start_and_await_workflow( + temporal_test_env.client, + TrainModel.run, + scenario, + make_workflow_id('test-s1-1-5'), + ) + + assert_experiment_status(postgres_engine, experiment_run_id, 'TRAINING_SUCCESS') + + +@pytest.mark.asyncio +@pytest.mark.integration +async def test_scenario_1_1_6_polynomial_degree2_scaler_and_lags( + temporal_test_env: WorkflowEnvironment, + temporal_worker: Worker, + postgres_engine, +): + """Scenario 1.1.6 – Polynomial degree 2, Standard Scaler and lags (cenário 09).""" + scenario = load_scenario('09-polynomial-degree2-with-scaler-and-lags.json') + experiment_run_id = scenario['experiment_run_id'] + insert_experiment_run(postgres_engine, experiment_run_id) + + await start_and_await_workflow( + temporal_test_env.client, + TrainModel.run, + scenario, + make_workflow_id('test-s1-1-6'), + ) + + assert_experiment_status(postgres_engine, experiment_run_id, 'TRAINING_SUCCESS') + assert_experiment_run_name_set(postgres_engine, experiment_run_id) + + +@pytest.mark.asyncio +@pytest.mark.integration +async def test_scenario_1_1_7_static_window_removal( + temporal_test_env: WorkflowEnvironment, + temporal_worker: Worker, + postgres_engine, +): + """Scenario 1.1.7 – rem_static_win=true with window and static_threshold (cenário 11).""" + scenario = load_scenario('11-linear-regression-static-threshold-custom.json') + experiment_run_id = scenario['experiment_run_id'] + insert_experiment_run(postgres_engine, experiment_run_id) + + await start_and_await_workflow( + temporal_test_env.client, + TrainModel.run, + scenario, + make_workflow_id('test-s1-1-7'), + ) + + assert_experiment_status(postgres_engine, experiment_run_id, 'TRAINING_SUCCESS') + + +@pytest.mark.asyncio +@pytest.mark.integration +async def test_scenario_1_1_8_polynomial_with_support_filters( + temporal_test_env: WorkflowEnvironment, + temporal_worker: Worker, + postgres_engine, +): + """Scenario 1.1.8 – Polynomial degree 4, Standard Scaler, upper/lower support filters (cenário 14).""" + scenario = load_scenario('14-angular-test-polynomial-support-filters.json') + # Override date range to match rows in our test CSV + scenario['data_model_kwargs']['start_date'] = '2025-06-02 00:00:00' + scenario['data_model_kwargs']['end_date'] = '2025-06-06 23:59:59' + experiment_run_id = scenario['experiment_run_id'] + insert_experiment_run(postgres_engine, experiment_run_id) + + await start_and_await_workflow( + temporal_test_env.client, + TrainModel.run, + scenario, + make_workflow_id('test-s1-1-8'), + ) + + assert_experiment_status(postgres_engine, experiment_run_id, 'TRAINING_SUCCESS') + + +# --------------------------------------------------------------------------- +# 1.2 – Error paths +# --------------------------------------------------------------------------- + +@pytest.mark.asyncio +@pytest.mark.integration +async def test_scenario_1_2_1_minio_file_not_found( + temporal_test_env: WorkflowEnvironment, + temporal_worker: Worker, + postgres_engine, +): + """Scenario 1.2.1 – Training file does not exist in MinIO → TRAINING_ERROR.""" + scenario = load_scenario('01-linear-regression-basic.json') + scenario = {**scenario, 'experiment_run_id': 2001, 'file_name': 'does_not_exist.csv'} + experiment_run_id = 2001 + insert_experiment_run(postgres_engine, experiment_run_id) + + with pytest.raises(Exception): + await start_and_await_workflow( + temporal_test_env.client, + TrainModel.run, + scenario, + make_workflow_id('test-s1-2-1'), + ) + + assert_experiment_error( + postgres_engine, + experiment_run_id, + expected_status='TRAINING_ERROR', + error_substr='does_not_exist', + ) + + +@pytest.mark.asyncio +@pytest.mark.integration +async def test_scenario_1_2_2_experiment_run_id_not_in_db( + temporal_test_env: WorkflowEnvironment, + temporal_worker: Worker, + postgres_engine, +): + """Scenario 1.2.2 – experiment_run_id row absent → update_experiment_run raises.""" + scenario = load_scenario('01-linear-regression-basic.json') + scenario = {**scenario, 'experiment_run_id': 9999} + # Intentionally NOT inserting the row + + with pytest.raises(Exception): + await start_and_await_workflow( + temporal_test_env.client, + TrainModel.run, + scenario, + make_workflow_id('test-s1-2-2'), + ) + + from e2e.helpers import assert_no_experiment_row + assert_no_experiment_row(postgres_engine, 9999) diff --git a/input-sample.json b/input-sample.json new file mode 100644 index 0000000..9e6bbbc --- /dev/null +++ b/input-sample.json @@ -0,0 +1,37 @@ +{ + "experiment_run_id": 1001, + "variable_columns": ["feature_a", "feature_b"], + "target_variable": "target", + "bucket_name": "model-training", + "file_name": "training_data.csv", + "line_separator": ",", + "decimal_separator": ".", + "train_size": 80, + "shuffle": true, + "random_state": 42, + "model_name": "Linear Regression", + "model_type": "linear_regression", + "data_model_kwargs": { + "lag_train": { + "feature_a": 0, + "feature_b": 0 + }, + "lag_val": { + "feature_a": 0, + "feature_b": 0 + }, + "nan_treatment": "drop", + "rem_static_win": false, + "static_threshold": null, + "start_date": null, + "end_date": null, + "support_filters": {}, + "removed_intervals": [] + }, + "model_kwargs": { + "degree": 1, + "interaction_only": false, + "scaler_name": "Standard Scaler" + }, + "opt_params": {} +} diff --git a/model_manager/utils/models/train_model_params.py b/model_manager/utils/models/train_model_params.py index 40dc033..33a8fc9 100644 --- a/model_manager/utils/models/train_model_params.py +++ b/model_manager/utils/models/train_model_params.py @@ -138,7 +138,7 @@ class TrainModelParams: random_state=cls._check_none(data.get('random_state', 42), int, 'random_state'), experiment_run_id=cls._coerce_experiment_run_id(data.get('experiment_run_id')), model_name=model_name, - experiment_name=model_name + '_experiment', + experiment_name=model_name, val_file_name=data.get('val_file_name'), data_model_kwargs=cls._check_none( data.get('data_model_kwargs'), dict, 'data_model_kwargs' diff --git a/model_manager/utils/repository/data_manager_repository.py b/model_manager/utils/repository/data_manager_repository.py index 01997e3..5553bf5 100644 --- a/model_manager/utils/repository/data_manager_repository.py +++ b/model_manager/utils/repository/data_manager_repository.py @@ -200,7 +200,7 @@ class DataManagerRepository(SientiaMonitoring): metadata, ) - experiment_name = f'{params.model_name}' + experiment_name = f'{params.experiment_name}' run_name = f'{experiment_name}_{datetime.now().strftime("%Y%m%d_%H%M%S")}' return TrainModelResult( diff --git a/pyproject.toml b/pyproject.toml index 3ab4344..8eae7e4 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -58,6 +58,13 @@ ignore = [ "S106", # hardcoded passwords ok in tests "S108", # temp paths are expected in tests ] +"e2e/**/*.py" = [ + "S101", # assert allowed in tests + "S105", # hardcoded passwords ok in tests + "S106", # hardcoded passwords ok in tests + "S108", # temp paths are expected in tests + "ARG001", # unused function args in fixtures +] [tool.ruff.lint.mccabe] max-complexity = 15 @@ -130,7 +137,7 @@ module = [ ignore_errors = true [tool.pytest.ini_options] -testpaths = ["tests"] +testpaths = ["tests", "e2e"] python_files = ["test_*.py"] python_classes = ["Test*"] python_functions = ["test_*"] diff --git a/requirements-dev.txt b/requirements-dev.txt index fd3ce26..d1a62ab 100644 --- a/requirements-dev.txt +++ b/requirements-dev.txt @@ -10,9 +10,11 @@ pandas-stubs>=2.0.0 # Type stubs for pandas types-requests>=2.31.0 # Type stubs for requests # Testing -pytest>=7.4.0 # Testing framework -pytest-cov>=4.1.0 # Coverage plugin for pytest -pytest-asyncio>=0.21.0 # Async test support (already in main requirements) +pytest>=7.4.0 # Testing framework +pytest-cov>=4.1.0 # Coverage plugin for pytest +pytest-asyncio>=0.21.0 # Async test support +testcontainers[postgres,minio,mongodb]>=4.0.0 # Real containers for E2E tests +requests>=2.31.0 # HTTP client for Gitea REST API seeding (E2E) # Development Tools ipython>=8.12.0 # Enhanced Python shell