diff --git a/README.md b/README.md index 7d57f09..3886d8b 100644 --- a/README.md +++ b/README.md @@ -314,34 +314,71 @@ The **PredictionProcess** workflow implements the core prediction pipeline for M #### Input Parameters ```json { - "metadata": { - "schedule_name": "hourly_predictions", - "model_name": "temperature_prediction_model", - "model_id": "temp_pred_001", - "workflow_name": "predictions_batch" - }, - "data": {...}, - "schema": {...}, - "table_name": "predictions", - "model_id": "temp_pred_001", - "model_name": "temperature_prediction_model", - "input_filters": { - "EMPTY_DATA": {"POLICY": "STOP"}, - "SPECIFIC_VARIABLES_NULL_VALUES": { - "POLICY": "STOP", - "config": {"variables": ["temperature", "humidity"]} + "schedule_name": "laborious-orchestrated-pipeline", + "model_id": "1", + "workflow_type": "predictions_batch", + "frequency": "30s", # Workflow execution frequency + "max_retry_policy": 1, # Maximum number of retries for the workflow + "query": "select * from sientia_data.laborious_data where model_id = 1 and \"timestamp\" > NOW() - INTERVAL '5 minutes' order by \"timestamp\" desc limit 30;", + "retention_time": 60, # Retention time for models in minutes + "write_tags": [ + { + "server_id": "1", + "type": "prediction", # Type of tag to write, can be prediction or confidence + "addr": "ns=2;i=5", + "data_type": "double" + }, + { + "server_id": "1", + "type": "confidence", + "addr": "ns=2;i=5", + "data_type": "double" } - }, - "mlflow_transform_filters": { - "API_ERROR": {"POLICY": "STOP"} - }, - "mlflow_predict_filters": { - "API_ERROR": {"POLICY": "STOP"}, - "NAN_VALUES": {"POLICY": "STOP"} - }, - "model_retention": 60, - "path_priority": ["STOP", "CONTINUE", "REPEAT"], - "opc_output_config": {...} + ], + "input_filters": [ + { + "filter_name": "EMPTY_DATA", # Required filter + "policy": "STOP" + }, + { + "filter_name": "SPECIFIC_VARIABLES_NULL_VALUES", + "policy": "CONTINUE", + "config": { + "variables": [ + "Counter" + ] + } + } + ], + "mlflow_transform_filters": [ + { + "filter_name": "API_ERROR", # Required filter + "policy": "REPEAT" + }, + { + "filter_name": "NAN_VALUES", + "policy": "STOP" + } + ], + "mlflow_predict_filters": [ + { + "filter_name": "API_ERROR", # Required filter + "policy": "CONTINUE" + } + ], + "path_priority": [ # In case of multiple filters catch problems, this will determine the path to take + "STOP", + "CONTINUE", + "REPEAT" + ], + "active": true, + "datetime_columns": [ # Columns in data comming from query that are datetime + "timestamp", + "created_at" + ], + "updated_at": { + "$date": "2025-08-27T18:35:01.600Z" + } } ```