feat: enhance test scenarios and configuration for regression models

- Updated `pyproject.toml` to include new linting rules for end-to-end tests.
- Modified `requirements-dev.txt` to add dependencies for E2E testing with `testcontainers` and `requests`.
- Refactored multiple JSON test scenario files to standardize structure, including new fields for `experiment_run_id`, `bucket_name`, and `file_name`.
- Improved model training parameters in `train_model_params.py` to use `experiment_name` directly.
- Adjusted `data_manager_repository.py` to utilize the updated `experiment_name` for logging.

These changes improve the organization and clarity of regression model tests and enhance the overall testing framework.
This commit is contained in:
vitor-aignosi
2026-05-04 11:11:16 -03:00
parent 50d0ea6f32
commit 6bd30e3328
26 changed files with 2024 additions and 397 deletions

View File

@@ -1,30 +1,40 @@
{
"_description": "Cenário básico de regressão linear sem scaler",
"experimentName": "test-linear-regression-basic",
"username": "bruno.domingues@aignosi.com.br",
"modelName": "Linear Regression",
"targetVariable": "03CV020/CORRENTE_N_M1_PV(Value)",
"variableColumns": ["303-WIT-200(Value)"],
"lagTrain": {"303-WIT-200(Value)": 0},
"lagVal": {"303-WIT-200(Value)": 0},
"remStaticWin": false,
"lowLim": {},
"uppLim": {},
"window": 0,
"useScaler": false,
"includeAr": false,
"trainSize": 80,
"experiment_run_id": 1001,
"variable_columns": [
"303-WIT-200(Value)"
],
"target_variable": "03CV020/CORRENTE_N_M1_PV(Value)",
"bucket_name": "model-training",
"file_name": "training_data.csv",
"line_separator": ",",
"decimal_separator": ".",
"date_column": "timestamp",
"date_format": "yyyy-MM-dd HH:mm:ss",
"train_size": 80,
"shuffle": true,
"lineSeparator": ",",
"decimalSeparator": ".",
"dateColumn": "timestamp",
"dateFormat": "yyyy-MM-dd HH:mm:ss",
"removedIntervals": [],
"random_state": 42,
"model_name": "Linear Regression",
"model_type": "linear_regression",
"data_model_kwargs": {
"lag_train": {
"303-WIT-200(Value)": 0
},
"lag_val": {
"303-WIT-200(Value)": 0
},
"nan_treatment": "drop",
"rem_static_win": false,
"static_threshold": null,
"start_date": null,
"end_date": null,
"support_filters": {},
"removed_intervals": []
},
"model_kwargs": {
"degree": 1,
"interactionOnly": false,
"nanTreatment": "drop",
"startDate": null,
"endDate": null,
"scalerName": "None",
"supportFilters": {}
"interaction_only": false,
"scaler_name": "None"
},
"opt_params": {}
}

View File

@@ -1,30 +1,40 @@
{
"_description": "Regressão linear com Standard Scaler habilitado",
"experimentName": "test-linear-regression-scaler",
"username": "bruno.domingues@aignosi.com.br",
"modelName": "Linear Regression",
"targetVariable": "03CV020/CORRENTE_N_M1_PV(Value)",
"variableColumns": ["303-WIT-200(Value)"],
"lagTrain": {"303-WIT-200(Value)": 0},
"lagVal": {"303-WIT-200(Value)": 0},
"remStaticWin": false,
"lowLim": {},
"uppLim": {},
"window": 0,
"useScaler": true,
"includeAr": false,
"trainSize": 80,
"experiment_run_id": 1002,
"variable_columns": [
"303-WIT-200(Value)"
],
"target_variable": "03CV020/CORRENTE_N_M1_PV(Value)",
"bucket_name": "model-training",
"file_name": "training_data.csv",
"line_separator": ",",
"decimal_separator": ".",
"date_column": "timestamp",
"date_format": "yyyy-MM-dd HH:mm:ss",
"train_size": 80,
"shuffle": true,
"lineSeparator": ",",
"decimalSeparator": ".",
"dateColumn": "timestamp",
"dateFormat": "yyyy-MM-dd HH:mm:ss",
"removedIntervals": [],
"random_state": 42,
"model_name": "Linear Regression",
"model_type": "linear_regression",
"data_model_kwargs": {
"lag_train": {
"303-WIT-200(Value)": 0
},
"lag_val": {
"303-WIT-200(Value)": 0
},
"nan_treatment": "drop",
"rem_static_win": false,
"static_threshold": null,
"start_date": null,
"end_date": null,
"support_filters": {},
"removed_intervals": []
},
"model_kwargs": {
"degree": 1,
"interactionOnly": false,
"nanTreatment": "drop",
"startDate": null,
"endDate": null,
"scalerName": "Standard Scaler",
"supportFilters": {}
"interaction_only": false,
"scaler_name": "Standard Scaler"
},
"opt_params": {}
}

View File

@@ -1,30 +1,40 @@
{
"_description": "Regressão polinomial de grau 2 com scaler (obrigatório para evitar overflow)",
"experimentName": "test-polynomial-degree2",
"username": "bruno.domingues@aignosi.com.br",
"modelName": "Polynomial Regression",
"targetVariable": "03CV020/CORRENTE_N_M1_PV(Value)",
"variableColumns": ["303-WIT-200(Value)"],
"lagTrain": {"303-WIT-200(Value)": 0},
"lagVal": {"303-WIT-200(Value)": 0},
"remStaticWin": false,
"lowLim": {},
"uppLim": {},
"window": 0,
"useScaler": true,
"includeAr": false,
"trainSize": 80,
"experiment_run_id": 1003,
"variable_columns": [
"303-WIT-200(Value)"
],
"target_variable": "03CV020/CORRENTE_N_M1_PV(Value)",
"bucket_name": "model-training",
"file_name": "training_data.csv",
"line_separator": ",",
"decimal_separator": ".",
"date_column": "timestamp",
"date_format": "yyyy-MM-dd HH:mm:ss",
"train_size": 80,
"shuffle": true,
"lineSeparator": ",",
"decimalSeparator": ".",
"dateColumn": "timestamp",
"dateFormat": "yyyy-MM-dd HH:mm:ss",
"removedIntervals": [],
"random_state": 42,
"model_name": "Polynomial Regression",
"model_type": "polynomial_regression",
"data_model_kwargs": {
"lag_train": {
"303-WIT-200(Value)": 0
},
"lag_val": {
"303-WIT-200(Value)": 0
},
"nan_treatment": "drop",
"rem_static_win": false,
"static_threshold": null,
"start_date": null,
"end_date": null,
"support_filters": {},
"removed_intervals": []
},
"model_kwargs": {
"degree": 2,
"interactionOnly": false,
"nanTreatment": "drop",
"startDate": null,
"endDate": null,
"scalerName": "Standard Scaler",
"supportFilters": {}
"interaction_only": false,
"scaler_name": "Standard Scaler"
},
"opt_params": {}
}

View File

@@ -1,30 +1,40 @@
{
"_description": "Regressão polinomial de grau 3 com scaler",
"experimentName": "test-polynomial-degree3",
"username": "bruno.domingues@aignosi.com.br",
"modelName": "Polynomial Regression",
"targetVariable": "03CV020/CORRENTE_N_M1_PV(Value)",
"variableColumns": ["303-WIT-200(Value)"],
"lagTrain": {"303-WIT-200(Value)": 0},
"lagVal": {"303-WIT-200(Value)": 0},
"remStaticWin": false,
"lowLim": {},
"uppLim": {},
"window": 0,
"useScaler": true,
"includeAr": false,
"trainSize": 80,
"experiment_run_id": 1004,
"variable_columns": [
"303-WIT-200(Value)"
],
"target_variable": "03CV020/CORRENTE_N_M1_PV(Value)",
"bucket_name": "model-training",
"file_name": "training_data.csv",
"line_separator": ",",
"decimal_separator": ".",
"date_column": "timestamp",
"date_format": "yyyy-MM-dd HH:mm:ss",
"train_size": 80,
"shuffle": true,
"lineSeparator": ",",
"decimalSeparator": ".",
"dateColumn": "timestamp",
"dateFormat": "yyyy-MM-dd HH:mm:ss",
"removedIntervals": [],
"random_state": 42,
"model_name": "Polynomial Regression",
"model_type": "polynomial_regression",
"data_model_kwargs": {
"lag_train": {
"303-WIT-200(Value)": 0
},
"lag_val": {
"303-WIT-200(Value)": 0
},
"nan_treatment": "drop",
"rem_static_win": false,
"static_threshold": null,
"start_date": null,
"end_date": null,
"support_filters": {},
"removed_intervals": []
},
"model_kwargs": {
"degree": 3,
"interactionOnly": false,
"nanTreatment": "drop",
"startDate": null,
"endDate": null,
"scalerName": "Standard Scaler",
"supportFilters": {}
"interaction_only": false,
"scaler_name": "Standard Scaler"
},
"opt_params": {}
}

View File

@@ -1,30 +1,40 @@
{
"_description": "Regressão linear com lags de treino e validação",
"experimentName": "test-linear-with-lags",
"username": "bruno.domingues@aignosi.com.br",
"modelName": "Linear Regression",
"targetVariable": "03CV020/CORRENTE_N_M1_PV(Value)",
"variableColumns": ["303-WIT-200(Value)"],
"lagTrain": {"303-WIT-200(Value)": 5},
"lagVal": {"303-WIT-200(Value)": 3},
"remStaticWin": false,
"lowLim": {},
"uppLim": {},
"window": 0,
"useScaler": false,
"includeAr": false,
"trainSize": 80,
"experiment_run_id": 1005,
"variable_columns": [
"303-WIT-200(Value)"
],
"target_variable": "03CV020/CORRENTE_N_M1_PV(Value)",
"bucket_name": "model-training",
"file_name": "training_data.csv",
"line_separator": ",",
"decimal_separator": ".",
"date_column": "timestamp",
"date_format": "yyyy-MM-dd HH:mm:ss",
"train_size": 80,
"shuffle": true,
"lineSeparator": ",",
"decimalSeparator": ".",
"dateColumn": "timestamp",
"dateFormat": "yyyy-MM-dd HH:mm:ss",
"removedIntervals": [],
"random_state": 42,
"model_name": "Linear Regression",
"model_type": "linear_regression",
"data_model_kwargs": {
"lag_train": {
"303-WIT-200(Value)": 5
},
"lag_val": {
"303-WIT-200(Value)": 3
},
"nan_treatment": "drop",
"rem_static_win": false,
"static_threshold": null,
"start_date": null,
"end_date": null,
"support_filters": {},
"removed_intervals": []
},
"model_kwargs": {
"degree": 1,
"interactionOnly": false,
"nanTreatment": "drop",
"startDate": null,
"endDate": null,
"scalerName": "None",
"supportFilters": {}
"interaction_only": false,
"scaler_name": "None"
},
"opt_params": {}
}

View File

@@ -1,30 +1,40 @@
{
"_description": "Regressão linear com tratamento de NaN por interpolação linear",
"experimentName": "test-linear-nan-interpolation",
"username": "bruno.domingues@aignosi.com.br",
"modelName": "Linear Regression",
"targetVariable": "03CV020/CORRENTE_N_M1_PV(Value)",
"variableColumns": ["303-WIT-200(Value)"],
"lagTrain": {"303-WIT-200(Value)": 0},
"lagVal": {"303-WIT-200(Value)": 0},
"remStaticWin": false,
"lowLim": {},
"uppLim": {},
"window": 0,
"useScaler": false,
"includeAr": false,
"trainSize": 80,
"experiment_run_id": 1006,
"variable_columns": [
"303-WIT-200(Value)"
],
"target_variable": "03CV020/CORRENTE_N_M1_PV(Value)",
"bucket_name": "model-training",
"file_name": "training_data.csv",
"line_separator": ",",
"decimal_separator": ".",
"date_column": "timestamp",
"date_format": "yyyy-MM-dd HH:mm:ss",
"train_size": 80,
"shuffle": true,
"lineSeparator": ",",
"decimalSeparator": ".",
"dateColumn": "timestamp",
"dateFormat": "yyyy-MM-dd HH:mm:ss",
"removedIntervals": [],
"random_state": 42,
"model_name": "Linear Regression",
"model_type": "linear_regression",
"data_model_kwargs": {
"lag_train": {
"303-WIT-200(Value)": 0
},
"lag_val": {
"303-WIT-200(Value)": 0
},
"nan_treatment": "linear interpolation",
"rem_static_win": false,
"static_threshold": null,
"start_date": null,
"end_date": null,
"support_filters": {},
"removed_intervals": []
},
"model_kwargs": {
"degree": 1,
"interactionOnly": false,
"nanTreatment": "linear interpolation",
"startDate": null,
"endDate": null,
"scalerName": "None",
"supportFilters": {}
"interaction_only": false,
"scaler_name": "None"
},
"opt_params": {}
}

View File

@@ -1,30 +1,40 @@
{
"_description": "Regressão linear com remoção de janelas estáticas",
"experimentName": "test-linear-static-removal",
"username": "bruno.domingues@aignosi.com.br",
"modelName": "Linear Regression",
"targetVariable": "03CV020/CORRENTE_N_M1_PV(Value)",
"variableColumns": ["303-WIT-200(Value)"],
"lagTrain": {"303-WIT-200(Value)": 0},
"lagVal": {"303-WIT-200(Value)": 0},
"remStaticWin": true,
"lowLim": {},
"uppLim": {},
"window": 10,
"useScaler": false,
"includeAr": false,
"trainSize": 80,
"experiment_run_id": 1007,
"variable_columns": [
"303-WIT-200(Value)"
],
"target_variable": "03CV020/CORRENTE_N_M1_PV(Value)",
"bucket_name": "model-training",
"file_name": "training_data.csv",
"line_separator": ",",
"decimal_separator": ".",
"date_column": "timestamp",
"date_format": "yyyy-MM-dd HH:mm:ss",
"train_size": 80,
"shuffle": true,
"lineSeparator": ",",
"decimalSeparator": ".",
"dateColumn": "timestamp",
"dateFormat": "yyyy-MM-dd HH:mm:ss",
"removedIntervals": [],
"random_state": 42,
"model_name": "Linear Regression",
"model_type": "linear_regression",
"data_model_kwargs": {
"lag_train": {
"303-WIT-200(Value)": 0
},
"lag_val": {
"303-WIT-200(Value)": 0
},
"nan_treatment": "drop",
"rem_static_win": true,
"static_threshold": null,
"start_date": null,
"end_date": null,
"support_filters": {},
"removed_intervals": []
},
"model_kwargs": {
"degree": 1,
"interactionOnly": false,
"nanTreatment": "drop",
"startDate": null,
"endDate": null,
"scalerName": "None",
"supportFilters": {}
"interaction_only": false,
"scaler_name": "None"
},
"opt_params": {}
}

View File

@@ -1,30 +1,45 @@
{
"_description": "Regressão linear com limites inferior e superior para variáveis",
"experimentName": "test-linear-with-limits",
"username": "bruno.domingues@aignosi.com.br",
"modelName": "Linear Regression",
"targetVariable": "03CV020/CORRENTE_N_M1_PV(Value)",
"variableColumns": ["303-WIT-200(Value)"],
"lagTrain": {"303-WIT-200(Value)": 0},
"lagVal": {"303-WIT-200(Value)": 0},
"remStaticWin": false,
"lowLim": {"303-WIT-200(Value)": 0.0},
"uppLim": {"303-WIT-200(Value)": 1000.0},
"window": 0,
"useScaler": false,
"includeAr": false,
"trainSize": 80,
"experiment_run_id": 1008,
"variable_columns": [
"303-WIT-200(Value)"
],
"target_variable": "03CV020/CORRENTE_N_M1_PV(Value)",
"bucket_name": "model-training",
"file_name": "training_data.csv",
"line_separator": ",",
"decimal_separator": ".",
"date_column": "timestamp",
"date_format": "yyyy-MM-dd HH:mm:ss",
"train_size": 80,
"shuffle": true,
"lineSeparator": ",",
"decimalSeparator": ".",
"dateColumn": "timestamp",
"dateFormat": "yyyy-MM-dd HH:mm:ss",
"removedIntervals": [],
"random_state": 42,
"model_name": "Linear Regression",
"model_type": "linear_regression",
"data_model_kwargs": {
"lag_train": {
"303-WIT-200(Value)": 0
},
"lag_val": {
"303-WIT-200(Value)": 0
},
"nan_treatment": "drop",
"rem_static_win": false,
"static_threshold": null,
"start_date": null,
"end_date": null,
"support_filters": {
"303-WIT-200(Value)": {
"min": 0.0,
"max": 1000.0
}
},
"removed_intervals": []
},
"model_kwargs": {
"degree": 1,
"interactionOnly": false,
"nanTreatment": "drop",
"startDate": null,
"endDate": null,
"scalerName": "None",
"supportFilters": {}
"interaction_only": false,
"scaler_name": "None"
},
"opt_params": {}
}

View File

@@ -1,30 +1,40 @@
{
"_description": "Cenário completo: regressão polinomial grau 2 com scaler e lags",
"experimentName": "test-polynomial-complete",
"username": "bruno.domingues@aignosi.com.br",
"modelName": "Polynomial Regression",
"targetVariable": "03CV020/CORRENTE_N_M1_PV(Value)",
"variableColumns": ["303-WIT-200(Value)"],
"lagTrain": {"303-WIT-200(Value)": 3},
"lagVal": {"303-WIT-200(Value)": 2},
"remStaticWin": false,
"lowLim": {},
"uppLim": {},
"window": 0,
"useScaler": true,
"includeAr": false,
"trainSize": 80,
"experiment_run_id": 1009,
"variable_columns": [
"303-WIT-200(Value)"
],
"target_variable": "03CV020/CORRENTE_N_M1_PV(Value)",
"bucket_name": "model-training",
"file_name": "training_data.csv",
"line_separator": ",",
"decimal_separator": ".",
"date_column": "timestamp",
"date_format": "yyyy-MM-dd HH:mm:ss",
"train_size": 80,
"shuffle": true,
"lineSeparator": ",",
"decimalSeparator": ".",
"dateColumn": "timestamp",
"dateFormat": "yyyy-MM-dd HH:mm:ss",
"removedIntervals": [],
"random_state": 42,
"model_name": "Polynomial Regression",
"model_type": "polynomial_regression",
"data_model_kwargs": {
"lag_train": {
"303-WIT-200(Value)": 3
},
"lag_val": {
"303-WIT-200(Value)": 2
},
"nan_treatment": "drop",
"rem_static_win": false,
"static_threshold": null,
"start_date": null,
"end_date": null,
"support_filters": {},
"removed_intervals": []
},
"model_kwargs": {
"degree": 2,
"interactionOnly": false,
"nanTreatment": "drop",
"startDate": null,
"endDate": null,
"scalerName": "Standard Scaler",
"supportFilters": {}
"interaction_only": false,
"scaler_name": "Standard Scaler"
},
"opt_params": {}
}

View File

@@ -1,30 +1,40 @@
{
"_description": "Regressão linear com variável autoregressiva (AR)",
"experimentName": "test-linear-with-ar",
"username": "bruno.domingues@aignosi.com.br",
"modelName": "Linear Regression",
"targetVariable": "03CV020/CORRENTE_N_M1_PV(Value)",
"variableColumns": ["303-WIT-200(Value)"],
"lagTrain": {"303-WIT-200(Value)": 0},
"lagVal": {"303-WIT-200(Value)": 0},
"remStaticWin": false,
"lowLim": {},
"uppLim": {},
"window": 0,
"useScaler": false,
"includeAr": true,
"trainSize": 80,
"experiment_run_id": 1010,
"variable_columns": [
"303-WIT-200(Value)"
],
"target_variable": "03CV020/CORRENTE_N_M1_PV(Value)",
"bucket_name": "model-training",
"file_name": "training_data.csv",
"line_separator": ",",
"decimal_separator": ".",
"date_column": "timestamp",
"date_format": "yyyy-MM-dd HH:mm:ss",
"train_size": 80,
"shuffle": true,
"lineSeparator": ",",
"decimalSeparator": ".",
"dateColumn": "timestamp",
"dateFormat": "yyyy-MM-dd HH:mm:ss",
"removedIntervals": [],
"random_state": 42,
"model_name": "Linear Regression",
"model_type": "linear_regression",
"data_model_kwargs": {
"lag_train": {
"303-WIT-200(Value)": 0
},
"lag_val": {
"303-WIT-200(Value)": 0
},
"nan_treatment": "drop",
"rem_static_win": false,
"static_threshold": null,
"start_date": null,
"end_date": null,
"support_filters": {},
"removed_intervals": []
},
"model_kwargs": {
"degree": 1,
"interactionOnly": false,
"nanTreatment": "drop",
"startDate": null,
"endDate": null,
"scalerName": "None",
"supportFilters": {}
"interaction_only": false,
"scaler_name": "None"
},
"opt_params": {}
}

View File

@@ -1,31 +1,40 @@
{
"_description": "Regressão linear com remoção de janelas estáticas e static_threshold customizado",
"experimentName": "test-linear-static-threshold",
"username": "bruno.domingues@aignosi.com.br",
"modelName": "Linear Regression",
"targetVariable": "03CV020/CORRENTE_N_M1_PV(Value)",
"variableColumns": ["303-WIT-200(Value)"],
"lagTrain": {"303-WIT-200(Value)": 0},
"lagVal": {"303-WIT-200(Value)": 0},
"remStaticWin": true,
"staticThreshold": 100,
"lowLim": {},
"uppLim": {},
"window": 10,
"useScaler": false,
"includeAr": false,
"trainSize": 80,
"experiment_run_id": 1011,
"variable_columns": [
"303-WIT-200(Value)"
],
"target_variable": "03CV020/CORRENTE_N_M1_PV(Value)",
"bucket_name": "model-training",
"file_name": "training_data.csv",
"line_separator": ",",
"decimal_separator": ".",
"date_column": "timestamp",
"date_format": "yyyy-MM-dd HH:mm:ss",
"train_size": 80,
"shuffle": true,
"lineSeparator": ",",
"decimalSeparator": ".",
"dateColumn": "timestamp",
"dateFormat": "yyyy-MM-dd HH:mm:ss",
"removedIntervals": [],
"random_state": 42,
"model_name": "Linear Regression",
"model_type": "linear_regression",
"data_model_kwargs": {
"lag_train": {
"303-WIT-200(Value)": 0
},
"lag_val": {
"303-WIT-200(Value)": 0
},
"nan_treatment": "drop",
"rem_static_win": true,
"static_threshold": 100,
"start_date": null,
"end_date": null,
"support_filters": {},
"removed_intervals": []
},
"model_kwargs": {
"degree": 1,
"interactionOnly": false,
"nanTreatment": "drop",
"startDate": null,
"endDate": null,
"scalerName": "None",
"supportFilters": {}
"interaction_only": false,
"scaler_name": "None"
},
"opt_params": {}
}

View File

@@ -1,31 +1,40 @@
{
"_description": "Cenário angular-test-01: CV022 WIT230 com lag e intervalo de datas",
"experimentName": "angular-test-01",
"username": "lucas.kou@aignosi.com.br",
"modelName": "Linear Regression",
"targetVariable": "03CV022/CORRENTE_N_M1_PV(Value)",
"variableColumns": ["303-WIT-230(Value)"],
"lagTrain": {"303-WIT-230(Value)": 3},
"lagVal": {"303-WIT-230(Value)": 0},
"remStaticWin": false,
"lowLim": {},
"uppLim": {},
"window": 0,
"useScaler": false,
"includeAr": false,
"trainSize": 80,
"experiment_run_id": 1012,
"variable_columns": [
"303-WIT-230(Value)"
],
"target_variable": "03CV022/CORRENTE_N_M1_PV(Value)",
"bucket_name": "model-training",
"file_name": "training_data.csv",
"line_separator": ",",
"decimal_separator": ".",
"date_column": "DATA",
"date_format": "dd/MM/yyyy HH:mm:ss",
"train_size": 80,
"shuffle": true,
"lineSeparator": ",",
"decimalSeparator": ".",
"dateColumn": "DATA",
"dateFormat": "dd/MM/yyyy HH:mm:ss",
"removedIntervals": [],
"random_state": 42,
"model_name": "Linear Regression",
"model_type": "linear_regression",
"data_model_kwargs": {
"lag_train": {
"303-WIT-230(Value)": 3
},
"lag_val": {
"303-WIT-230(Value)": 0
},
"nan_treatment": "drop",
"rem_static_win": false,
"static_threshold": null,
"start_date": "01/05/2022",
"end_date": "31/07/2022",
"support_filters": {},
"removed_intervals": []
},
"model_kwargs": {
"degree": 1,
"interactionOnly": false,
"nanTreatment": "drop",
"startDate": "01/05/2022",
"endDate": "31/07/2022",
"scalerName": "None",
"supportFilters": {},
"staticThreshold": null
"interaction_only": false,
"scaler_name": "None"
},
"opt_params": {}
}

View File

@@ -1,31 +1,40 @@
{
"_description": "Cenário angular-test: CV022 WIT230 com ficheiro double date column e intervalo curto (00:00 a 00:05)",
"experimentName": "angular-test",
"username": "lucas.kou@aignosi.com.br",
"modelName": "Linear Regression",
"targetVariable": "03CV022/CORRENTE_N_M1_PV(Value)",
"variableColumns": ["303-WIT-230(Value)"],
"lagTrain": {"303-WIT-230(Value)": 0},
"lagVal": {"303-WIT-230(Value)": 0},
"remStaticWin": false,
"lowLim": {},
"uppLim": {},
"window": 0,
"useScaler": false,
"includeAr": false,
"trainSize": 80,
"experiment_run_id": 1013,
"variable_columns": [
"303-WIT-230(Value)"
],
"target_variable": "03CV022/CORRENTE_N_M1_PV(Value)",
"bucket_name": "model-training",
"file_name": "training_data.csv",
"line_separator": ",",
"decimal_separator": ".",
"date_column": "DATA",
"date_format": "dd/MM/yyyy HH:mm:ss",
"train_size": 80,
"shuffle": true,
"lineSeparator": ",",
"decimalSeparator": ".",
"dateColumn": "DATA",
"dateFormat": "dd/MM/yyyy HH:mm:ss",
"removedIntervals": [],
"random_state": 42,
"model_name": "Linear Regression",
"model_type": "linear_regression",
"data_model_kwargs": {
"lag_train": {
"303-WIT-230(Value)": 0
},
"lag_val": {
"303-WIT-230(Value)": 0
},
"nan_treatment": "drop",
"rem_static_win": false,
"static_threshold": null,
"start_date": "01/05/2022 00:00:00",
"end_date": "01/05/2022 00:05:10",
"support_filters": {},
"removed_intervals": []
},
"model_kwargs": {
"degree": 1,
"interactionOnly": false,
"nanTreatment": "drop",
"startDate": "01/05/2022 00:00:00",
"endDate": "01/05/2022 00:05:10",
"scalerName": "None",
"supportFilters": {},
"staticThreshold": null
"interaction_only": false,
"scaler_name": "None"
},
"opt_params": {}
}

View File

@@ -1,32 +1,34 @@
{
"_description": "Cenário angular-test-01: regressão polinomial degree 4, scaler, support filters em 303-WIT-200",
"experimentName": "angular-test-01",
"username": "lucas.kou@aignosi.com.br",
"modelName": "Polynomial Regression",
"targetVariable": "03CV020/CORRENTE_N_M1_PV(Value)",
"variableColumns": ["303-WIT-200(Value)"],
"lagTrain": {"303-WIT-200(Value)": 0},
"lagVal": {"303-WIT-200(Value)": 0},
"remStaticWin": false,
"lowLim": {},
"uppLim": {},
"window": 0,
"useScaler": true,
"includeAr": false,
"trainSize": 80,
"experiment_run_id": 1014,
"variable_columns": [
"303-WIT-200(Value)"
],
"target_variable": "03CV020/CORRENTE_N_M1_PV(Value)",
"bucket_name": "model-training",
"file_name": "training_data.csv",
"line_separator": ",",
"decimal_separator": ".",
"date_column": "timestamp",
"date_format": "yyyy-MM-dd HH:mm:ss",
"train_size": 80,
"shuffle": true,
"lineSeparator": ",",
"decimalSeparator": ".",
"dateColumn": "timestamp",
"dateFormat": "yyyy-MM-dd HH:mm:ss",
"removedIntervals": [],
"degree": 4,
"interactionOnly": false,
"nanTreatment": "drop",
"startDate": "2025-06-02 00:00:05",
"endDate": "2025-06-06 15:02:01",
"scalerName": "Standard Scaler",
"supportFilters": {
"random_state": 42,
"model_name": "Polynomial Regression",
"model_type": "polynomial_regression",
"data_model_kwargs": {
"lag_train": {
"303-WIT-200(Value)": 0
},
"lag_val": {
"303-WIT-200(Value)": 0
},
"nan_treatment": "drop",
"rem_static_win": false,
"static_threshold": null,
"start_date": "2025-06-02 00:00:05",
"end_date": "2025-06-06 15:02:01",
"support_filters": {
"303-WIT-200(Value)": {
"upper_line": {
"intercept": 40.400002,
@@ -38,5 +40,12 @@
}
}
},
"staticThreshold": null
"removed_intervals": []
},
"model_kwargs": {
"degree": 4,
"interaction_only": false,
"scaler_name": "Standard Scaler"
},
"opt_params": {}
}

3
e2e/__init__.py Normal file
View File

@@ -0,0 +1,3 @@
"""
End-to-end tests for the Model Manager Temporal workflows.
"""

581
e2e/conftest.py Normal file
View File

@@ -0,0 +1,581 @@
"""
Pytest configuration and fixtures for E2E tests.
All external dependencies use real services:
- PostgreSQL: testcontainers (postgres:15)
- MinIO: testcontainers (minio)
- MongoDB: testcontainers (mongo:7)
- MLflow: local filesystem tracking (no network)
- Gitea: testcontainers generic container (gitea/gitea:latest),
seeded with model-plugin-warehouse files via REST API
- Temporal: in-memory WorkflowEnvironment (time-skipping)
"""
from concurrent.futures import ThreadPoolExecutor
import base64
import csv
import io
import os
import shutil
import tempfile
import time
import uuid
from pathlib import Path
from unittest.mock import MagicMock
import mlflow
import pytest
import pytest_asyncio
import requests
from minio import Minio
from sqlalchemy import create_engine, text
from testcontainers.core.container import DockerContainer
from testcontainers.minio import MinioContainer
from testcontainers.mongodb import MongoDbContainer
from testcontainers.postgres import PostgresContainer
from temporalio.testing import WorkflowEnvironment
from temporalio.worker import Worker
from model_manager.activities.activities import Activities
from model_manager.workflows.cleanup_files import CleanupFiles
from model_manager.workflows.train_model import TrainModel
from sientia_do.notifications.handlers import CoreNotificationHandler
from sientia_do.observability.metrics_controller import MetricsController
from sientia_model.model_repository.plugin_store import PluginStore
# ---------------------------------------------------------------------------
# Paths
# ---------------------------------------------------------------------------
_WAREHOUSE_ROOT = Path(
'/home/grezewave/Documents/projects/sientia/model-plugin-warehouse'
)
# CSV training data: columns must match the variable_columns and target_variable
# used across all test scenarios.
_TRAIN_CSV_COLUMNS = [
'timestamp',
'303-WIT-200(Value)',
'03CV020/CORRENTE_N_M1_PV(Value)',
'303-WIT-230(Value)',
'03CV022/CORRENTE_N_M1_PV(Value)',
]
_MINIO_BUCKET = 'model-training'
_MINIO_OBJECT = 'training_data.csv'
# ---------------------------------------------------------------------------
# Helpers CSV generation
# ---------------------------------------------------------------------------
def _build_training_csv() -> bytes:
"""
Generate a 150-row CSV with all columns needed by test scenarios.
The numeric values cycle deterministically so lags and static-window
removal always find enough rows in both train and validation splits.
"""
output = io.StringIO()
writer = csv.writer(output)
writer.writerow(_TRAIN_CSV_COLUMNS)
for i in range(150):
ts = f'2025-06-{(i // 24) + 2:02d} {i % 24:02d}:00:00'
wit200 = round(30.0 + (i % 20) * 0.5, 2)
cv020 = round(100.0 + (i % 15) * 0.3, 2)
wit230 = round(25.0 + (i % 18) * 0.4, 2)
cv022 = round(90.0 + (i % 12) * 0.25, 2)
writer.writerow([ts, wit200, cv020, wit230, cv022])
return output.getvalue().encode('utf-8')
def _build_training_csv_dd_mm_yyyy() -> bytes:
"""
Generate a 150-row CSV with dd/MM/yyyy HH:mm:ss timestamps and
a DATA column header, for scenarios 12/13 that use a different date format.
"""
output = io.StringIO()
writer = csv.writer(output)
writer.writerow([
'DATA',
'303-WIT-230(Value)',
'03CV022/CORRENTE_N_M1_PV(Value)',
])
for i in range(150):
day = (i % 30) + 1
ts = f'{day:02d}/05/2022 {i % 24:02d}:00:00'
wit230 = round(25.0 + (i % 18) * 0.4, 2)
cv022 = round(90.0 + (i % 12) * 0.25, 2)
writer.writerow([ts, wit230, cv022])
return output.getvalue().encode('utf-8')
# ---------------------------------------------------------------------------
# Helpers Gitea seed
# ---------------------------------------------------------------------------
def _wait_for_gitea(base_url: str, timeout: int = 120) -> None:
"""Poll Gitea until it responds to HTTP requests."""
deadline = time.time() + timeout
last_err = None
while time.time() < deadline:
try:
resp = requests.get(f'{base_url}/', timeout=3)
if resp.status_code in (200, 404, 302):
return
except Exception as e:
last_err = e
time.sleep(2)
raise TimeoutError(f'Gitea did not start within {timeout}s at {base_url}. Last error: {last_err}')
def _gitea_api(method: str, url: str, auth: tuple, **kwargs) -> requests.Response:
resp = requests.request(method, url, auth=auth, timeout=30, **kwargs)
try:
resp.raise_for_status()
except requests.exceptions.HTTPError as e:
raise RuntimeError(f"Gitea API error {resp.status_code}: {resp.text}") from e
return resp
def _seed_gitea(base_url: str, admin_user: str, admin_pass: str) -> None:
"""
Create a fictitious model-store repository with dummy models.
"""
auth = (admin_user, admin_pass)
api = f'{base_url}/api/v1'
# Create repository
_gitea_api(
'POST', f'{api}/user/repos', auth,
json={'name': 'model-store', 'private': False, 'auto_init': False},
)
# Root index.yaml
root_index = """
store_name: "E2E Test Store"
version: 1
models:
- name: "linear_regression"
version: 1
runtime: "basic"
- name: "polynomial_regression"
version: 1
runtime: "basic"
runtimes:
basic:
version: "1.0.0"
libraries:
- name: "pandas"
- name: "numpy"
"""
# Model index.yaml (shared for all dummies)
model_index = """
name: "{model_name}"
version: 1
runtime: "basic"
path: "wrapper.py"
class: "DummyWrapper"
model:
class: "DummyModel"
path: "model_logic.py"
external: false
data_model:
class: "DummyTransformer"
path: "model_logic.py"
external: false
"""
# schemas.yaml
schemas_yaml = """
model:
type: object
properties: {}
data_model:
type: object
properties: {}
opt_params:
type: object
properties: {}
"""
# wrapper.py
wrapper_py = """
from sientia_model.wrappers.sientia_model import SientiaModel
import pandas as pd
import numpy as np
from typing import Any
class DummyWrapper(SientiaModel):
def _predict(self, data: pd.DataFrame) -> tuple[pd.DataFrame, dict[str, Any]]:
self._log("info", f"Predicting dummy model for {self.model_type}")
# Return a simple prediction (mean or 0.5) to allow metrics computation
preds = pd.DataFrame({self.target: [0.5] * len(data)}, index=data.index)
return preds, {}
def _transform(self, data: pd.DataFrame) -> tuple[pd.DataFrame, dict[str, Any]]:
return data, {}
def _train_transformer(self, train_data: pd.DataFrame, val_data: pd.DataFrame) -> None:
pass
def _train_model(self, x: pd.DataFrame, y: pd.DataFrame, x_val: pd.DataFrame | None = None, y_val: pd.DataFrame | None = None) -> None:
self.target = y.columns[0]
def _retrain_transformer(self, data: pd.DataFrame) -> None:
pass
def _retrain_model(self, x: pd.DataFrame, y: pd.DataFrame | None) -> None:
pass
"""
# model_logic.py
model_logic_py = """
class DummyModel:
def __init__(self, **kwargs):
pass
class DummyTransformer:
def __init__(self, **kwargs):
pass
"""
def push_file(path: str, content: str):
encoded = base64.b64encode(content.encode()).decode()
_gitea_api(
'POST',
f'{api}/repos/{admin_user}/model-store/contents/{path}',
auth,
json={'message': f'seed: {path}', 'content': encoded},
)
# Push root index
push_file('index.yaml', root_index)
# Push files for both models used in tests
for model_name in ['linear_regression', 'polynomial_regression']:
prefix = f'models/{model_name}'
push_file(f'{prefix}/index.yaml', model_index.format(model_name=model_name))
push_file(f'{prefix}/schemas.yaml', schemas_yaml)
push_file(f'{prefix}/wrapper.py', wrapper_py)
push_file(f'{prefix}/model_logic.py', model_logic_py)
push_file(f'{prefix}/__init__.py', "")
# Push runtime
push_file('runtime/basic.yaml', 'name: basic\nversion: "1.0.0"\nlibraries: []')
# ---------------------------------------------------------------------------
# Session-scoped containers
# ---------------------------------------------------------------------------
@pytest_asyncio.fixture(scope='session')
def postgres_container():
"""PostgreSQL 15 container for experiment_run table."""
container = PostgresContainer('postgres:15')
container.start()
yield container
container.stop()
@pytest_asyncio.fixture(scope='session')
def minio_container():
"""MinIO container for training CSV storage."""
container = MinioContainer()
container.start()
yield container
container.stop()
@pytest_asyncio.fixture(scope='session')
def mongodb_container():
"""MongoDB container for CoreNotificationHandler."""
container = MongoDbContainer('mongo:7')
container.start()
yield container
container.stop()
@pytest_asyncio.fixture(scope='session')
def gitea_container():
"""
Gitea container seeded with the model-plugin-warehouse files.
The container starts with INSTALL_LOCK so no setup wizard is needed.
An admin user is created via Gitea's CLI before the HTTP API is used.
"""
admin_user = 'gitea_admin'
admin_pass = 'gitea_admin_pass' # noqa: S105
container = (
DockerContainer('gitea/gitea:latest')
.with_env('GITEA__security__INSTALL_LOCK', 'true')
.with_env('GITEA__server__HTTP_PORT', '3000')
.with_env('GITEA__log__LEVEL', 'Warn')
.with_exposed_ports(3000)
)
container.start()
port = container.get_exposed_port(3000)
base_url = f'http://localhost:{port}'
_wait_for_gitea(base_url)
import time
time.sleep(5) # Wait a bit for DB to fully initialize after HTTP is up
# Create admin user via Gitea CLI inside the container
# Must run after Gitea is fully initialized
gitea_cmd = (
f'gitea admin user create '
f'--username {admin_user} '
f'--password {admin_pass} '
f'--email admin@test.local '
f'--admin '
f'--must-change-password=false'
)
exec_result = container.exec(f"su git -c '{gitea_cmd}'")
if exec_result.exit_code != 0:
raise RuntimeError(f"Failed to create Gitea admin user: {exec_result.output.decode('utf-8')}")
_seed_gitea(base_url, admin_user, admin_pass)
yield {
'container': container,
'base_url': base_url,
'admin_user': admin_user,
'admin_pass': admin_pass,
}
container.stop()
@pytest_asyncio.fixture(scope='session')
def mlflow_tracking_dir():
"""Local MLflow filesystem tracking directory (no network needed)."""
tmpdir = tempfile.mkdtemp(prefix='mlflow-e2e-')
mlflow.set_tracking_uri(f'file://{tmpdir}')
yield tmpdir
shutil.rmtree(tmpdir, ignore_errors=True)
# ---------------------------------------------------------------------------
# Session-scoped: seed MinIO with training CSV
# ---------------------------------------------------------------------------
@pytest_asyncio.fixture(scope='session', autouse=True)
def upload_training_csv(minio_container, mlflow_tracking_dir): # noqa: ARG001
"""
Upload training CSV files to the MinIO container before any test runs.
Depends on mlflow_tracking_dir to ensure the MLflow URI is set at session start.
"""
port = minio_container.get_exposed_port(9000)
client = Minio(
f'localhost:{port}',
access_key='minioadmin',
secret_key='minioadmin',
secure=False,
)
if not client.bucket_exists(_MINIO_BUCKET):
client.make_bucket(_MINIO_BUCKET)
# Standard training CSV
csv_bytes = _build_training_csv()
client.put_object(
_MINIO_BUCKET,
_MINIO_OBJECT,
io.BytesIO(csv_bytes),
length=len(csv_bytes),
content_type='text/csv',
)
# dd/MM/yyyy format CSV for scenarios 12/13
alt_csv_bytes = _build_training_csv_dd_mm_yyyy()
client.put_object(
_MINIO_BUCKET,
'training_data_dd_mm_yyyy.csv',
io.BytesIO(alt_csv_bytes),
length=len(alt_csv_bytes),
content_type='text/csv',
)
# ---------------------------------------------------------------------------
# Function-scoped: database engine + schema setup
# ---------------------------------------------------------------------------
@pytest_asyncio.fixture
def postgres_engine(postgres_container):
"""SQLAlchemy engine connected to the test PostgreSQL container."""
engine = create_engine(postgres_container.get_connection_url())
yield engine
engine.dispose()
@pytest_asyncio.fixture(autouse=True)
def setup_experiment_run_table(postgres_engine):
"""
Create the experiment_run table before each test and drop it afterwards
to guarantee full isolation between tests.
"""
with postgres_engine.begin() as conn:
conn.execute(text("""
CREATE TABLE IF NOT EXISTS public.experiment_run (
id INT PRIMARY KEY,
experiment_name TEXT NOT NULL,
run_name TEXT,
username TEXT,
status TEXT NOT NULL DEFAULT 'ORCHESTRATOR_WAITING_PROC',
error_message TEXT,
created_at TIMESTAMPTZ NOT NULL DEFAULT NOW(),
updated_at TIMESTAMPTZ NOT NULL DEFAULT NOW(),
bucket_name TEXT,
file_name TEXT
)
"""))
yield
with postgres_engine.begin() as conn:
conn.execute(text('DROP TABLE IF EXISTS public.experiment_run'))
# ---------------------------------------------------------------------------
# Mock-only fixtures (no external service equivalent)
# ---------------------------------------------------------------------------
@pytest_asyncio.fixture
def mock_logger():
"""Minimal logger that prints to stdout (no external observability needed)."""
def _log(msg, *args, **kwargs): # noqa: ARG001
print(f'[LOG] {msg}')
logger = MagicMock()
for method in ('info', 'debug', 'error', 'warning', 'critical',
'custom_info', 'custom_debug', 'custom_error',
'custom_warning', 'custom_critical'):
setattr(logger, method, MagicMock(side_effect=_log))
logger.base_logger = MagicMock()
return logger
@pytest_asyncio.fixture
def mock_metrics_controller(mock_logger):
"""Real MetricsController backed by the mock logger."""
return MetricsController(logger=mock_logger)
# ---------------------------------------------------------------------------
# Real application fixtures
# ---------------------------------------------------------------------------
@pytest_asyncio.fixture
def notification_handler(mongodb_container, mock_logger):
"""
Real CoreNotificationHandler connected to the MongoDB testcontainer.
"""
connection_url = mongodb_container.get_connection_url()
handler = CoreNotificationHandler(
connection_string=connection_url,
database='test_notifications',
logger=mock_logger,
project_name='model-manager-e2e',
)
yield handler
handler.shutdown()
@pytest_asyncio.fixture
def plugin_store(gitea_container, mock_logger, mock_metrics_controller, notification_handler):
"""
Real PluginStore pointed at the Gitea testcontainer.
cache_ttl_seconds=0 forces a fresh download every test.
"""
store = PluginStore(
base_url=gitea_container['base_url'],
owner=gitea_container['admin_user'],
repo='model-store',
username=gitea_container['admin_user'],
password=gitea_container['admin_pass'],
cache_ttl_seconds=0,
logger=mock_logger,
notification_handler=notification_handler,
metrics_controller=mock_metrics_controller,
)
yield store
@pytest_asyncio.fixture
def test_activities(
postgres_container,
minio_container,
mlflow_tracking_dir, # noqa: ARG001 ensures MLflow URI is set
plugin_store,
mock_logger,
notification_handler,
mock_metrics_controller,
):
"""
Real Activities instance wired to all testcontainers.
"""
pg_port = postgres_container.get_exposed_port(5432)
minio_port = minio_container.get_exposed_port(9000)
activities = Activities(
postgres_config={
'host': 'localhost',
'port': int(pg_port),
'user': 'test',
'password': 'test',
'dbname': 'test',
'min_connections': 1,
'max_connections': 5,
},
mlflow_config={
'url': mlflow.get_tracking_uri(),
'username': None,
'password': None,
},
minio_config={
'endpoint_url': f'http://localhost:{minio_port}',
'access_key': 'minioadmin',
'secret_key': 'minioadmin',
'use_ssl': False,
'default_bucket': _MINIO_BUCKET,
},
plugin_store=plugin_store,
logger=mock_logger,
notification_handler=notification_handler,
metrics_controller=mock_metrics_controller,
)
yield activities
activities.shutdown()
def _activity_list(activities: Activities) -> list:
return [
activities.update_experiment_run,
activities.load_model_metadata,
activities.validate_train_params,
activities.train_model,
activities.cleanup_resources,
activities.cleanup_temp_directories,
]
@pytest_asyncio.fixture(scope='function')
async def temporal_test_env():
"""In-memory Temporal environment with time-skipping."""
env = await WorkflowEnvironment.start_time_skipping()
async with env:
yield env
@pytest_asyncio.fixture(scope='function')
async def temporal_worker(temporal_test_env, test_activities):
"""Temporal worker registered with all workflows and activities."""
with ThreadPoolExecutor() as executor:
async with Worker(
temporal_test_env.client,
task_queue='test-queue',
workflows=[TrainModel, CleanupFiles],
activities=_activity_list(test_activities),
activity_executor=executor,
) as worker:
yield worker

196
e2e/helpers.py Normal file
View File

@@ -0,0 +1,196 @@
"""
Shared helpers for E2E tests (Temporal workflows + PostgreSQL).
"""
import asyncio
from datetime import datetime
from typing import Any
import pytest
from sqlalchemy import text
from sqlalchemy.engine import Engine
async def start_and_await_workflow(
client,
workflow_run,
input_data: dict,
workflow_id: str,
timeout: float = 120.0,
):
"""
Start a Temporal workflow and wait for its result.
Args:
client: Temporal client from WorkflowEnvironment.
workflow_run: Workflow run method (e.g. TrainModel.run).
input_data: Workflow input payload.
workflow_id: Unique workflow id.
timeout: Max seconds to wait for completion.
Returns:
Workflow result value.
"""
handle = await client.start_workflow(
workflow_run,
input_data,
id=workflow_id,
task_queue='test-queue',
)
return await asyncio.wait_for(handle.result(), timeout=timeout)
def make_workflow_id(prefix: str) -> str:
"""Build a unique workflow id using a prefix and current timestamp."""
return f'{prefix}-{datetime.now().timestamp()}'
def insert_experiment_run(
engine: Engine,
experiment_run_id: int,
experiment_name: str = 'test_experiment',
status: str = 'ORCHESTRATOR_WAITING_PROC',
bucket_name: str = 'model-training',
file_name: str = 'training_data.csv',
) -> None:
"""
Insert a minimal experiment_run row to satisfy foreign-key-style lookups.
Args:
engine: SQLAlchemy engine connected to the test database.
experiment_run_id: Primary key for the row.
experiment_name: Human-readable experiment name.
status: Initial status string.
bucket_name: MinIO bucket name.
file_name: Training file name inside the bucket.
"""
with engine.begin() as conn:
conn.execute(
text("""
INSERT INTO public.experiment_run
(id, experiment_name, status, bucket_name, file_name)
VALUES
(:id, :experiment_name, :status, :bucket_name, :file_name)
ON CONFLICT (id) DO NOTHING
"""),
{
'id': experiment_run_id,
'experiment_name': experiment_name,
'status': status,
'bucket_name': bucket_name,
'file_name': file_name,
},
)
def assert_experiment_status(
engine: Engine,
experiment_run_id: int,
expected_status: str,
) -> None:
"""
Assert the final status of an experiment_run row.
Args:
engine: SQLAlchemy engine.
experiment_run_id: Row primary key.
expected_status: Expected status string.
"""
with engine.connect() as conn:
row = conn.execute(
text('SELECT status FROM public.experiment_run WHERE id = :id'),
{'id': experiment_run_id},
).fetchone()
assert row is not None, (
f'No experiment_run row found for id={experiment_run_id}'
)
assert row[0] == expected_status, (
f'Expected status={expected_status!r}, got {row[0]!r} '
f'for experiment_run id={experiment_run_id}'
)
def assert_experiment_run_name_set(
engine: Engine,
experiment_run_id: int,
) -> None:
"""Assert that run_name is not null/empty after a successful training."""
with engine.connect() as conn:
row = conn.execute(
text('SELECT run_name FROM public.experiment_run WHERE id = :id'),
{'id': experiment_run_id},
).fetchone()
assert row is not None, (
f'No experiment_run row found for id={experiment_run_id}'
)
assert row[0] is not None and row[0].strip() != '', (
f'Expected run_name to be set for experiment_run id={experiment_run_id}, got {row[0]!r}'
)
def assert_experiment_error(
engine: Engine,
experiment_run_id: int,
expected_status: str,
error_substr: str,
) -> None:
"""
Assert status and that error_message contains a given substring.
Args:
engine: SQLAlchemy engine.
experiment_run_id: Row primary key.
expected_status: Expected status string.
error_substr: Substring that must appear in error_message.
"""
with engine.connect() as conn:
row = conn.execute(
text(
'SELECT status, error_message FROM public.experiment_run WHERE id = :id'
),
{'id': experiment_run_id},
).fetchone()
assert row is not None, (
f'No experiment_run row found for id={experiment_run_id}'
)
assert row[0] == expected_status, (
f'Expected status={expected_status!r}, got {row[0]!r}'
)
assert row[1] is not None and error_substr.lower() in row[1].lower(), (
f'Expected error_message to contain {error_substr!r}, got {row[1]!r}'
)
def assert_no_experiment_row(engine: Engine, experiment_run_id: int) -> None:
"""Assert that no experiment_run row exists for the given id."""
with engine.connect() as conn:
count = conn.execute(
text('SELECT COUNT(*) FROM public.experiment_run WHERE id = :id'),
{'id': experiment_run_id},
).scalar()
assert count == 0, (
f'Expected no experiment_run row for id={experiment_run_id}, found {count}'
)
def load_scenario(scenario_filename: str) -> dict[str, Any]:
"""
Load a test scenario JSON file from docs/test-scenarios/.
Args:
scenario_filename: Filename without path (e.g. '01-linear-regression-basic.json').
Returns:
dict: Parsed scenario payload.
"""
import json
from pathlib import Path
scenario_path = (
Path(__file__).parent.parent / 'docs' / 'test-scenarios' / scenario_filename
)
with open(scenario_path) as f:
return json.load(f)

47
e2e/scenarios.md Normal file
View File

@@ -0,0 +1,47 @@
# E2E Test Scenarios
This document maps the workflow scenarios tested in the E2E suite to their corresponding JSON input files and expected behaviors.
## 1. TrainModel Workflow (`test_train_model_workflow.py`)
### 1.1 Happy Paths (Successful execution)
| Test Function | Input JSON | Expected Status | Description |
|---|---|---|---|
| `test_scenario_1_1_1_linear_regression_basic` | `01-linear-regression-basic.json` | `TRAINING_SUCCESS` | Basic linear regression without scaler. Verifies end-to-end pipeline. |
| `test_scenario_1_1_2_polynomial_regression_degree2_with_scaler` | `03-polynomial-regression-degree2.json` | `TRAINING_SUCCESS` | Polynomial regression (degree 2) with Standard Scaler. |
| `test_scenario_1_1_3_linear_regression_with_lags` | `05-linear-regression-with-lags.json` | `TRAINING_SUCCESS` | Linear regression with `lag_train`/`lag_val` per variable. |
| `test_scenario_1_1_4_linear_regression_nan_interpolation` | `06-linear-regression-nan-interpolation.json` | `TRAINING_SUCCESS` | Linear regression with `nan_treatment='linear interpolation'`. |
| `test_scenario_1_1_5_linear_regression_with_limits` | `08-linear-regression-with-limits.json` | `TRAINING_SUCCESS` | Linear regression with `support_filters` (min/max limits per variable). |
| `test_scenario_1_1_6_polynomial_degree2_scaler_and_lags` | `09-polynomial-degree2-with-scaler-and-lags.json` | `TRAINING_SUCCESS` | Polynomial regression (degree 2), Standard Scaler, and lags. |
| `test_scenario_1_1_7_static_window_removal` | `11-linear-regression-static-threshold-custom.json` | `TRAINING_SUCCESS` | Linear regression with `rem_static_win=true`, `window`, and `static_threshold`. |
| `test_scenario_1_1_8_polynomial_with_support_filters` | `14-angular-test-polynomial-support-filters.json` | `TRAINING_SUCCESS` | Polynomial regression (degree 4), Standard Scaler, and support filters. |
### 1.2 Error Paths
| Test Function | Input JSON | Expected Status | Description |
|---|---|---|---|
| `test_scenario_1_2_1_minio_file_not_found` | `01-linear-regression-basic.json` | `TRAINING_ERROR` | MinIO file does not exist. Workflow fails during file download. |
| `test_scenario_1_2_2_experiment_run_id_not_in_db` | `01-linear-regression-basic.json` | N/A (raises Exception) | `experiment_run_id` does not exist in DB. Workflow fails immediately on status update attempt. |
## 2. Parameter Validation (`test_train_model_validation.py`)
These scenarios test the business rule validations inside `validate_train_params`. All are expected to terminate with `ORCHESTRATOR_VALIDATION_ERROR`.
| Test Function | Modification | Expected Error Substring |
|---|---|---|
| `test_scenario_2_1_1_train_size_out_of_range` | `train_size = 5` | `'train_size'` |
| `test_scenario_2_1_2_empty_variable_columns` | `variable_columns = []` | `'variable_columns'` |
| `test_scenario_2_1_3_invalid_date_format` | `date_format = 'INVALID'` | `'date_format'` |
| `test_scenario_2_1_4_whitespace_only_model_name` | `model_name = ' '` | `'model_name'` |
| `test_scenario_2_1_5_unknown_model_type` | `model_type = 'totally_unknown_model'` | `'totally_unknown_model'` |
| `test_scenario_2_1_6_missing_target_variable` | `target_variable = ''` | `'target_variable'` |
| `test_scenario_2_1_7_missing_experiment_run_id` | Missing `experiment_run_id` | N/A (raises ValueError immediately) |
## 3. CleanupFiles Workflow (`test_cleanup_files_workflow.py`)
| Test Function | Description |
|---|---|
| `test_scenario_3_1_1_cleanup_with_no_temp_dirs` | Temp directory is empty. Activity completes without error. |
| `test_scenario_3_1_2_cleanup_removes_old_temp_dirs` | Two stale timestamped directories are removed. |
| `test_scenario_3_1_3_cleanup_nonexistent_temp_path` | Target path does not exist. Handled gracefully without error. |

View File

@@ -0,0 +1,109 @@
"""
End-to-end tests for CleanupFiles workflow.
Covers scenarios 3.x: cleanup of temporary local directories.
"""
import os
import shutil
import tempfile
import pytest
import pytest_asyncio
from temporalio.testing import WorkflowEnvironment
from temporalio.worker import Worker
from e2e.helpers import make_workflow_id, start_and_await_workflow
from model_manager.workflows.cleanup_files import CleanupFiles
@pytest.mark.asyncio
@pytest.mark.integration
async def test_scenario_3_1_1_cleanup_with_no_temp_dirs(
temporal_test_env: WorkflowEnvironment,
temporal_worker: Worker,
tmp_path,
):
"""Scenario 3.1.1 Cleanup when the temp directory is empty.
The cleanup_temp_directories activity should complete without error
and the workflow should finish successfully.
"""
# Use an empty temp directory as the reports path
empty_dir = tmp_path / 'reports_temp'
empty_dir.mkdir()
result = await start_and_await_workflow(
temporal_test_env.client,
CleanupFiles.run,
{'temp_path': str(empty_dir)},
make_workflow_id('test-s3-1-1'),
)
# Workflow returns None on success
assert result is None
@pytest.mark.asyncio
@pytest.mark.integration
async def test_scenario_3_1_2_cleanup_removes_old_temp_dirs(
temporal_test_env: WorkflowEnvironment,
temporal_worker: Worker,
tmp_path,
):
"""Scenario 3.1.2 Cleanup removes stale subdirectories from the temp dir.
Creates two subdirectories with timestamp suffixes inside the reports
temp directory and verifies the activity removes them.
"""
reports_dir = tmp_path / 'reports_temp'
reports_dir.mkdir()
# Create two stale run directories
stale1 = reports_dir / 'run-1234567890'
stale2 = reports_dir / 'run-9876543210'
stale1.mkdir()
stale2.mkdir()
(stale1 / 'model.pkl').write_bytes(b'fake-model-data')
(stale2 / 'report.json').write_bytes(b'{"status": "old"}')
result = await start_and_await_workflow(
temporal_test_env.client,
CleanupFiles.run,
{'temp_path': str(reports_dir)},
make_workflow_id('test-s3-1-2'),
)
assert result is None
# The activity should have cleaned up the stale directories
remaining = list(reports_dir.iterdir())
assert len(remaining) == 0, (
f'Expected all stale dirs to be removed, but found: {remaining}'
)
@pytest.mark.asyncio
@pytest.mark.integration
async def test_scenario_3_1_3_cleanup_nonexistent_temp_path(
temporal_test_env: WorkflowEnvironment,
temporal_worker: Worker,
tmp_path,
):
"""Scenario 3.1.3 Cleanup with a temp_path that does not exist.
The activity must handle a missing directory gracefully without
raising an unhandled exception, since the directory may have already
been cleaned by a previous run.
"""
nonexistent = str(tmp_path / 'does_not_exist' / 'reports')
# Should not raise — the activity is expected to handle a missing path
result = await start_and_await_workflow(
temporal_test_env.client,
CleanupFiles.run,
{'temp_path': nonexistent},
make_workflow_id('test-s3-1-3'),
)
assert result is None

View File

@@ -0,0 +1,232 @@
"""
End-to-end tests for TrainModel parameter validation paths.
Covers scenarios 2.1.x: workflows that must terminate with
ORCHESTRATOR_VALIDATION_ERROR due to invalid parameter values.
"""
import pytest
import pytest_asyncio
from temporalio.testing import WorkflowEnvironment
from temporalio.worker import Worker
from e2e.helpers import (
assert_experiment_error,
insert_experiment_run,
load_scenario,
make_workflow_id,
start_and_await_workflow,
)
from model_manager.workflows.train_model import TrainModel
# Base experiment_run ids for validation test scenarios (offset to avoid collision)
_VALIDATION_ID_BASE = 3000
@pytest.mark.asyncio
@pytest.mark.integration
async def test_scenario_2_1_1_train_size_out_of_range(
temporal_test_env: WorkflowEnvironment,
temporal_worker: Worker,
postgres_engine,
):
"""Scenario 2.1.1 train_size=5 violates the 10100 business rule.
Expected: workflow updates status → ORCHESTRATOR_VALIDATION_ERROR
and error_message references 'train_size'.
"""
experiment_run_id = _VALIDATION_ID_BASE + 1
scenario = load_scenario('01-linear-regression-basic.json')
scenario = {**scenario, 'experiment_run_id': experiment_run_id, 'train_size': 5}
insert_experiment_run(postgres_engine, experiment_run_id)
with pytest.raises(Exception):
await start_and_await_workflow(
temporal_test_env.client,
TrainModel.run,
scenario,
make_workflow_id('test-s2-1-1'),
)
assert_experiment_error(
postgres_engine,
experiment_run_id,
expected_status='ORCHESTRATOR_VALIDATION_ERROR',
error_substr='train_size',
)
@pytest.mark.asyncio
@pytest.mark.integration
async def test_scenario_2_1_2_empty_variable_columns(
temporal_test_env: WorkflowEnvironment,
temporal_worker: Worker,
postgres_engine,
):
"""Scenario 2.1.2 variable_columns=[] → ORCHESTRATOR_VALIDATION_ERROR."""
experiment_run_id = _VALIDATION_ID_BASE + 2
scenario = load_scenario('01-linear-regression-basic.json')
scenario = {**scenario, 'experiment_run_id': experiment_run_id, 'variable_columns': []}
insert_experiment_run(postgres_engine, experiment_run_id)
with pytest.raises(Exception):
await start_and_await_workflow(
temporal_test_env.client,
TrainModel.run,
scenario,
make_workflow_id('test-s2-1-2'),
)
assert_experiment_error(
postgres_engine,
experiment_run_id,
expected_status='ORCHESTRATOR_VALIDATION_ERROR',
error_substr='variable_columns',
)
@pytest.mark.asyncio
@pytest.mark.integration
async def test_scenario_2_1_3_invalid_date_format(
temporal_test_env: WorkflowEnvironment,
temporal_worker: Worker,
postgres_engine,
):
"""Scenario 2.1.3 date_format='INVALID' is not in the allowed list."""
experiment_run_id = _VALIDATION_ID_BASE + 3
scenario = load_scenario('01-linear-regression-basic.json')
scenario = {**scenario, 'experiment_run_id': experiment_run_id, 'date_format': 'INVALID'}
insert_experiment_run(postgres_engine, experiment_run_id)
with pytest.raises(Exception):
await start_and_await_workflow(
temporal_test_env.client,
TrainModel.run,
scenario,
make_workflow_id('test-s2-1-3'),
)
assert_experiment_error(
postgres_engine,
experiment_run_id,
expected_status='ORCHESTRATOR_VALIDATION_ERROR',
error_substr='date_format',
)
@pytest.mark.asyncio
@pytest.mark.integration
async def test_scenario_2_1_4_whitespace_only_model_name(
temporal_test_env: WorkflowEnvironment,
temporal_worker: Worker,
postgres_engine,
):
"""Scenario 2.1.4 model_name=' ' (whitespace) → ORCHESTRATOR_VALIDATION_ERROR."""
experiment_run_id = _VALIDATION_ID_BASE + 4
scenario = load_scenario('01-linear-regression-basic.json')
scenario = {**scenario, 'experiment_run_id': experiment_run_id, 'model_name': ' '}
insert_experiment_run(postgres_engine, experiment_run_id)
with pytest.raises(Exception):
await start_and_await_workflow(
temporal_test_env.client,
TrainModel.run,
scenario,
make_workflow_id('test-s2-1-4'),
)
assert_experiment_error(
postgres_engine,
experiment_run_id,
expected_status='ORCHESTRATOR_VALIDATION_ERROR',
error_substr='model_name',
)
@pytest.mark.asyncio
@pytest.mark.integration
async def test_scenario_2_1_5_unknown_model_type(
temporal_test_env: WorkflowEnvironment,
temporal_worker: Worker,
postgres_engine,
):
"""Scenario 2.1.5 model_type='totally_unknown' → ORCHESTRATOR_VALIDATION_ERROR.
The PluginStore will not find this model in the Gitea repo, causing
load_model_metadata to fail before validate_train_params is even called.
"""
experiment_run_id = _VALIDATION_ID_BASE + 5
scenario = load_scenario('01-linear-regression-basic.json')
scenario = {
**scenario,
'experiment_run_id': experiment_run_id,
'model_type': 'totally_unknown_model',
}
insert_experiment_run(postgres_engine, experiment_run_id)
with pytest.raises(Exception):
await start_and_await_workflow(
temporal_test_env.client,
TrainModel.run,
scenario,
make_workflow_id('test-s2-1-5'),
)
assert_experiment_error(
postgres_engine,
experiment_run_id,
expected_status='ORCHESTRATOR_VALIDATION_ERROR',
error_substr='totally_unknown_model',
)
@pytest.mark.asyncio
@pytest.mark.integration
async def test_scenario_2_1_6_missing_target_variable(
temporal_test_env: WorkflowEnvironment,
temporal_worker: Worker,
postgres_engine,
):
"""Scenario 2.1.6 target_variable='' (empty string) → ORCHESTRATOR_VALIDATION_ERROR."""
experiment_run_id = _VALIDATION_ID_BASE + 6
scenario = load_scenario('01-linear-regression-basic.json')
scenario = {**scenario, 'experiment_run_id': experiment_run_id, 'target_variable': ''}
insert_experiment_run(postgres_engine, experiment_run_id)
with pytest.raises(Exception):
await start_and_await_workflow(
temporal_test_env.client,
TrainModel.run,
scenario,
make_workflow_id('test-s2-1-6'),
)
assert_experiment_error(
postgres_engine,
experiment_run_id,
expected_status='ORCHESTRATOR_VALIDATION_ERROR',
error_substr='target_variable',
)
@pytest.mark.asyncio
@pytest.mark.integration
async def test_scenario_2_1_7_missing_experiment_run_id(
temporal_test_env: WorkflowEnvironment,
temporal_worker: Worker,
):
"""Scenario 2.1.7 experiment_run_id missing → workflow raises ValueError immediately.
No DB row is inserted because experiment_run_id is mandatory to even
know which row to update. The workflow should raise before any DB call.
"""
scenario = load_scenario('01-linear-regression-basic.json')
scenario = {k: v for k, v in scenario.items() if k != 'experiment_run_id'}
with pytest.raises(Exception, match='experiment_run_id'):
await start_and_await_workflow(
temporal_test_env.client,
TrainModel.run,
scenario,
make_workflow_id('test-s2-1-7'),
)

View File

@@ -0,0 +1,272 @@
"""
End-to-end tests for TrainModel workflow main workflow scenarios.
Covers:
1.1.x Happy-path training (various scenarios from docs/test-scenarios/)
1.2.x Error paths (MinIO failure, missing DB row)
"""
import pytest
import pytest_asyncio
from temporalio.testing import WorkflowEnvironment
from temporalio.worker import Worker
from e2e.helpers import (
assert_experiment_error,
assert_experiment_run_name_set,
assert_experiment_status,
insert_experiment_run,
load_scenario,
make_workflow_id,
start_and_await_workflow,
)
from model_manager.workflows.train_model import TrainModel
# ---------------------------------------------------------------------------
# 1.1 Happy paths
# ---------------------------------------------------------------------------
@pytest.mark.asyncio
@pytest.mark.integration
async def test_scenario_1_1_1_linear_regression_basic(
temporal_test_env: WorkflowEnvironment,
temporal_worker: Worker,
postgres_engine,
):
"""Scenario 1.1.1 Linear Regression Basic (cenário 01).
Validates the complete training pipeline end-to-end:
load_model_metadata → validate_train_params → train_model →
update_experiment_run (TRAINING_SUCCESS).
"""
scenario = load_scenario('01-linear-regression-basic.json')
experiment_run_id = scenario['experiment_run_id']
insert_experiment_run(postgres_engine, experiment_run_id)
await start_and_await_workflow(
temporal_test_env.client,
TrainModel.run,
scenario,
make_workflow_id('test-s1-1-1'),
)
assert_experiment_status(postgres_engine, experiment_run_id, 'TRAINING_SUCCESS')
assert_experiment_run_name_set(postgres_engine, experiment_run_id)
@pytest.mark.asyncio
@pytest.mark.integration
async def test_scenario_1_1_2_polynomial_regression_degree2_with_scaler(
temporal_test_env: WorkflowEnvironment,
temporal_worker: Worker,
postgres_engine,
):
"""Scenario 1.1.2 Polynomial Regression Degree 2 with Standard Scaler (cenário 03)."""
scenario = load_scenario('03-polynomial-regression-degree2.json')
experiment_run_id = scenario['experiment_run_id']
insert_experiment_run(postgres_engine, experiment_run_id)
await start_and_await_workflow(
temporal_test_env.client,
TrainModel.run,
scenario,
make_workflow_id('test-s1-1-2'),
)
assert_experiment_status(postgres_engine, experiment_run_id, 'TRAINING_SUCCESS')
assert_experiment_run_name_set(postgres_engine, experiment_run_id)
@pytest.mark.asyncio
@pytest.mark.integration
async def test_scenario_1_1_3_linear_regression_with_lags(
temporal_test_env: WorkflowEnvironment,
temporal_worker: Worker,
postgres_engine,
):
"""Scenario 1.1.3 Linear Regression with lag_train/lag_val per variable (cenário 05)."""
scenario = load_scenario('05-linear-regression-with-lags.json')
experiment_run_id = scenario['experiment_run_id']
insert_experiment_run(postgres_engine, experiment_run_id)
await start_and_await_workflow(
temporal_test_env.client,
TrainModel.run,
scenario,
make_workflow_id('test-s1-1-3'),
)
assert_experiment_status(postgres_engine, experiment_run_id, 'TRAINING_SUCCESS')
assert_experiment_run_name_set(postgres_engine, experiment_run_id)
@pytest.mark.asyncio
@pytest.mark.integration
async def test_scenario_1_1_4_linear_regression_nan_interpolation(
temporal_test_env: WorkflowEnvironment,
temporal_worker: Worker,
postgres_engine,
):
"""Scenario 1.1.4 nan_treatment='linear interpolation' (cenário 06)."""
scenario = load_scenario('06-linear-regression-nan-interpolation.json')
experiment_run_id = scenario['experiment_run_id']
insert_experiment_run(postgres_engine, experiment_run_id)
await start_and_await_workflow(
temporal_test_env.client,
TrainModel.run,
scenario,
make_workflow_id('test-s1-1-4'),
)
assert_experiment_status(postgres_engine, experiment_run_id, 'TRAINING_SUCCESS')
@pytest.mark.asyncio
@pytest.mark.integration
async def test_scenario_1_1_5_linear_regression_with_limits(
temporal_test_env: WorkflowEnvironment,
temporal_worker: Worker,
postgres_engine,
):
"""Scenario 1.1.5 support_filters with min/max limits per variable (cenário 08)."""
scenario = load_scenario('08-linear-regression-with-limits.json')
experiment_run_id = scenario['experiment_run_id']
insert_experiment_run(postgres_engine, experiment_run_id)
await start_and_await_workflow(
temporal_test_env.client,
TrainModel.run,
scenario,
make_workflow_id('test-s1-1-5'),
)
assert_experiment_status(postgres_engine, experiment_run_id, 'TRAINING_SUCCESS')
@pytest.mark.asyncio
@pytest.mark.integration
async def test_scenario_1_1_6_polynomial_degree2_scaler_and_lags(
temporal_test_env: WorkflowEnvironment,
temporal_worker: Worker,
postgres_engine,
):
"""Scenario 1.1.6 Polynomial degree 2, Standard Scaler and lags (cenário 09)."""
scenario = load_scenario('09-polynomial-degree2-with-scaler-and-lags.json')
experiment_run_id = scenario['experiment_run_id']
insert_experiment_run(postgres_engine, experiment_run_id)
await start_and_await_workflow(
temporal_test_env.client,
TrainModel.run,
scenario,
make_workflow_id('test-s1-1-6'),
)
assert_experiment_status(postgres_engine, experiment_run_id, 'TRAINING_SUCCESS')
assert_experiment_run_name_set(postgres_engine, experiment_run_id)
@pytest.mark.asyncio
@pytest.mark.integration
async def test_scenario_1_1_7_static_window_removal(
temporal_test_env: WorkflowEnvironment,
temporal_worker: Worker,
postgres_engine,
):
"""Scenario 1.1.7 rem_static_win=true with window and static_threshold (cenário 11)."""
scenario = load_scenario('11-linear-regression-static-threshold-custom.json')
experiment_run_id = scenario['experiment_run_id']
insert_experiment_run(postgres_engine, experiment_run_id)
await start_and_await_workflow(
temporal_test_env.client,
TrainModel.run,
scenario,
make_workflow_id('test-s1-1-7'),
)
assert_experiment_status(postgres_engine, experiment_run_id, 'TRAINING_SUCCESS')
@pytest.mark.asyncio
@pytest.mark.integration
async def test_scenario_1_1_8_polynomial_with_support_filters(
temporal_test_env: WorkflowEnvironment,
temporal_worker: Worker,
postgres_engine,
):
"""Scenario 1.1.8 Polynomial degree 4, Standard Scaler, upper/lower support filters (cenário 14)."""
scenario = load_scenario('14-angular-test-polynomial-support-filters.json')
# Override date range to match rows in our test CSV
scenario['data_model_kwargs']['start_date'] = '2025-06-02 00:00:00'
scenario['data_model_kwargs']['end_date'] = '2025-06-06 23:59:59'
experiment_run_id = scenario['experiment_run_id']
insert_experiment_run(postgres_engine, experiment_run_id)
await start_and_await_workflow(
temporal_test_env.client,
TrainModel.run,
scenario,
make_workflow_id('test-s1-1-8'),
)
assert_experiment_status(postgres_engine, experiment_run_id, 'TRAINING_SUCCESS')
# ---------------------------------------------------------------------------
# 1.2 Error paths
# ---------------------------------------------------------------------------
@pytest.mark.asyncio
@pytest.mark.integration
async def test_scenario_1_2_1_minio_file_not_found(
temporal_test_env: WorkflowEnvironment,
temporal_worker: Worker,
postgres_engine,
):
"""Scenario 1.2.1 Training file does not exist in MinIO → TRAINING_ERROR."""
scenario = load_scenario('01-linear-regression-basic.json')
scenario = {**scenario, 'experiment_run_id': 2001, 'file_name': 'does_not_exist.csv'}
experiment_run_id = 2001
insert_experiment_run(postgres_engine, experiment_run_id)
with pytest.raises(Exception):
await start_and_await_workflow(
temporal_test_env.client,
TrainModel.run,
scenario,
make_workflow_id('test-s1-2-1'),
)
assert_experiment_error(
postgres_engine,
experiment_run_id,
expected_status='TRAINING_ERROR',
error_substr='does_not_exist',
)
@pytest.mark.asyncio
@pytest.mark.integration
async def test_scenario_1_2_2_experiment_run_id_not_in_db(
temporal_test_env: WorkflowEnvironment,
temporal_worker: Worker,
postgres_engine,
):
"""Scenario 1.2.2 experiment_run_id row absent → update_experiment_run raises."""
scenario = load_scenario('01-linear-regression-basic.json')
scenario = {**scenario, 'experiment_run_id': 9999}
# Intentionally NOT inserting the row
with pytest.raises(Exception):
await start_and_await_workflow(
temporal_test_env.client,
TrainModel.run,
scenario,
make_workflow_id('test-s1-2-2'),
)
from e2e.helpers import assert_no_experiment_row
assert_no_experiment_row(postgres_engine, 9999)

37
input-sample.json Normal file
View File

@@ -0,0 +1,37 @@
{
"experiment_run_id": 1001,
"variable_columns": ["feature_a", "feature_b"],
"target_variable": "target",
"bucket_name": "model-training",
"file_name": "training_data.csv",
"line_separator": ",",
"decimal_separator": ".",
"train_size": 80,
"shuffle": true,
"random_state": 42,
"model_name": "Linear Regression",
"model_type": "linear_regression",
"data_model_kwargs": {
"lag_train": {
"feature_a": 0,
"feature_b": 0
},
"lag_val": {
"feature_a": 0,
"feature_b": 0
},
"nan_treatment": "drop",
"rem_static_win": false,
"static_threshold": null,
"start_date": null,
"end_date": null,
"support_filters": {},
"removed_intervals": []
},
"model_kwargs": {
"degree": 1,
"interaction_only": false,
"scaler_name": "Standard Scaler"
},
"opt_params": {}
}

View File

@@ -138,7 +138,7 @@ class TrainModelParams:
random_state=cls._check_none(data.get('random_state', 42), int, 'random_state'),
experiment_run_id=cls._coerce_experiment_run_id(data.get('experiment_run_id')),
model_name=model_name,
experiment_name=model_name + '_experiment',
experiment_name=model_name,
val_file_name=data.get('val_file_name'),
data_model_kwargs=cls._check_none(
data.get('data_model_kwargs'), dict, 'data_model_kwargs'

View File

@@ -200,7 +200,7 @@ class DataManagerRepository(SientiaMonitoring):
metadata,
)
experiment_name = f'{params.model_name}'
experiment_name = f'{params.experiment_name}'
run_name = f'{experiment_name}_{datetime.now().strftime("%Y%m%d_%H%M%S")}'
return TrainModelResult(

View File

@@ -58,6 +58,13 @@ ignore = [
"S106", # hardcoded passwords ok in tests
"S108", # temp paths are expected in tests
]
"e2e/**/*.py" = [
"S101", # assert allowed in tests
"S105", # hardcoded passwords ok in tests
"S106", # hardcoded passwords ok in tests
"S108", # temp paths are expected in tests
"ARG001", # unused function args in fixtures
]
[tool.ruff.lint.mccabe]
max-complexity = 15
@@ -130,7 +137,7 @@ module = [
ignore_errors = true
[tool.pytest.ini_options]
testpaths = ["tests"]
testpaths = ["tests", "e2e"]
python_files = ["test_*.py"]
python_classes = ["Test*"]
python_functions = ["test_*"]

View File

@@ -12,7 +12,9 @@ types-requests>=2.31.0 # Type stubs for requests
# Testing
pytest>=7.4.0 # Testing framework
pytest-cov>=4.1.0 # Coverage plugin for pytest
pytest-asyncio>=0.21.0 # Async test support (already in main requirements)
pytest-asyncio>=0.21.0 # Async test support
testcontainers[postgres,minio,mongodb]>=4.0.0 # Real containers for E2E tests
requests>=2.31.0 # HTTP client for Gitea REST API seeding (E2E)
# Development Tools
ipython>=8.12.0 # Enhanced Python shell