From aa45fc8b99248d89195d7750648bce1cd930da26 Mon Sep 17 00:00:00 2001 From: vitor-aignosi Date: Fri, 29 Aug 2025 15:59:46 -0300 Subject: [PATCH] SIENTIAPDE-1182 Update README.md to reflect changes in prediction workflow configuration - Revised input parameters for the prediction process, including updates to schedule name, model ID, and workflow name. - Enhanced structure of input filters and MLflow filter policies for improved clarity and functionality. - Introduced new fields for model retention and output configuration, while maintaining backward compatibility with existing parameters. --- README.md | 170 +++++++++++++++++++++++++++++------------------------- 1 file changed, 91 insertions(+), 79 deletions(-) diff --git a/README.md b/README.md index 3886d8b..e9562b0 100644 --- a/README.md +++ b/README.md @@ -314,71 +314,34 @@ The **PredictionProcess** workflow implements the core prediction pipeline for M #### Input Parameters ```json { - "schedule_name": "laborious-orchestrated-pipeline", - "model_id": "1", - "workflow_type": "predictions_batch", - "frequency": "30s", # Workflow execution frequency - "max_retry_policy": 1, # Maximum number of retries for the workflow - "query": "select * from sientia_data.laborious_data where model_id = 1 and \"timestamp\" > NOW() - INTERVAL '5 minutes' order by \"timestamp\" desc limit 30;", - "retention_time": 60, # Retention time for models in minutes - "write_tags": [ - { - "server_id": "1", - "type": "prediction", # Type of tag to write, can be prediction or confidence - "addr": "ns=2;i=5", - "data_type": "double" - }, - { - "server_id": "1", - "type": "confidence", - "addr": "ns=2;i=5", - "data_type": "double" + "metadata": { + "schedule_name": "hourly_predictions", + "model_name": "temperature_prediction_model", + "model_id": "temp_pred_001", + "workflow_name": "predictions_batch" + }, + "data": {...}, + "schema": {...}, + "table_name": "predictions", + "model_id": "temp_pred_001", + "model_name": "temperature_prediction_model", + "input_filters": { + "EMPTY_DATA": {"POLICY": "STOP"}, + "SPECIFIC_VARIABLES_NULL_VALUES": { + "POLICY": "STOP", + "config": {"variables": ["temperature", "humidity"]} } - ], - "input_filters": [ - { - "filter_name": "EMPTY_DATA", # Required filter - "policy": "STOP" - }, - { - "filter_name": "SPECIFIC_VARIABLES_NULL_VALUES", - "policy": "CONTINUE", - "config": { - "variables": [ - "Counter" - ] - } - } - ], - "mlflow_transform_filters": [ - { - "filter_name": "API_ERROR", # Required filter - "policy": "REPEAT" - }, - { - "filter_name": "NAN_VALUES", - "policy": "STOP" - } - ], - "mlflow_predict_filters": [ - { - "filter_name": "API_ERROR", # Required filter - "policy": "CONTINUE" - } - ], - "path_priority": [ # In case of multiple filters catch problems, this will determine the path to take - "STOP", - "CONTINUE", - "REPEAT" - ], - "active": true, - "datetime_columns": [ # Columns in data comming from query that are datetime - "timestamp", - "created_at" - ], - "updated_at": { - "$date": "2025-08-27T18:35:01.600Z" - } + }, + "mlflow_transform_filters": { + "API_ERROR": {"POLICY": "STOP"} + }, + "mlflow_predict_filters": { + "API_ERROR": {"POLICY": "STOP"}, + "NAN_VALUES": {"POLICY": "STOP"} + }, + "model_retention": 60, + "path_priority": ["STOP", "CONTINUE", "REPEAT"], + "opc_output_config": {...} } ``` @@ -659,22 +622,71 @@ Workflows are configured through input parameters and filter policies: ```json { - "input_filters": { - "EMPTY_DATA": {"POLICY": "STOP"}, - "SPECIFIC_VARIABLES_NULL_VALUES": { - "POLICY": "STOP", - "config": {"variables": ["temperature", "humidity"]} + "schedule_name": "laborious-orchestrated-pipeline", + "model_id": "1", + "workflow_type": "predictions_batch", + "frequency": "30s", # Workflow execution frequency + "max_retry_policy": 1, # Maximum number of retries for the workflow + "query": "select * from sientia_data.laborious_data where model_id = 1 and \"timestamp\" > NOW() - INTERVAL '5 minutes' order by \"timestamp\" desc limit 30;", + "retention_time": 60, # Retention time for models in minutes + "write_tags": [ + { + "server_id": "1", + "type": "prediction", # Type of tag to write, can be prediction or confidence + "addr": "ns=2;i=5", + "data_type": "double" + }, + { + "server_id": "1", + "type": "confidence", + "addr": "ns=2;i=5", + "data_type": "double" } - }, - "mlflow_transform_filters": { - "API_ERROR": {"POLICY": "STOP"} - }, - "mlflow_predict_filters": { - "API_ERROR": {"POLICY": "STOP"}, - "NAN_VALUES": {"POLICY": "STOP"} - }, - "path_priority": ["STOP", "CONTINUE", "REPEAT"], - "model_retention": 60 + ], + "input_filters": [ + { + "filter_name": "EMPTY_DATA", # Required filter + "policy": "STOP" + }, + { + "filter_name": "SPECIFIC_VARIABLES_NULL_VALUES", + "policy": "CONTINUE", + "config": { + "variables": [ + "Counter" + ] + } + } + ], + "mlflow_transform_filters": [ + { + "filter_name": "API_ERROR", # Required filter + "policy": "REPEAT" + }, + { + "filter_name": "NAN_VALUES", + "policy": "STOP" + } + ], + "mlflow_predict_filters": [ + { + "filter_name": "API_ERROR", # Required filter + "policy": "CONTINUE" + } + ], + "path_priority": [ # In case of multiple filters catch problems, this will determine the path to take + "STOP", + "CONTINUE", + "REPEAT" + ], + "active": true, + "datetime_columns": [ # Columns in data comming from query that are datetime + "timestamp", + "created_at" + ], + "updated_at": { + "$date": "2025-08-27T18:35:01.600Z" + } } ```