SIENTIAPDE-1182

Update README.md to reflect new prediction workflow configuration

- Revised input parameters for the prediction process, including changes to schedule name, model ID, and workflow type.
- Introduced new fields for workflow execution frequency, maximum retry policy, and query for data retrieval.
- Updated input and MLflow filter structures to enhance clarity and functionality.
- Added support for datetime columns and updated retention time for models.
This commit is contained in:
vitor-aignosi
2025-08-29 15:28:51 -03:00
parent 995ba7900a
commit cda94a3850

View File

@@ -314,34 +314,71 @@ The **PredictionProcess** workflow implements the core prediction pipeline for M
#### Input Parameters #### Input Parameters
```json ```json
{ {
"metadata": { "schedule_name": "laborious-orchestrated-pipeline",
"schedule_name": "hourly_predictions", "model_id": "1",
"model_name": "temperature_prediction_model", "workflow_type": "predictions_batch",
"model_id": "temp_pred_001", "frequency": "30s", # Workflow execution frequency
"workflow_name": "predictions_batch" "max_retry_policy": 1, # Maximum number of retries for the workflow
"query": "select * from sientia_data.laborious_data where model_id = 1 and \"timestamp\" > NOW() - INTERVAL '5 minutes' order by \"timestamp\" desc limit 30;",
"retention_time": 60, # Retention time for models in minutes
"write_tags": [
{
"server_id": "1",
"type": "prediction", # Type of tag to write, can be prediction or confidence
"addr": "ns=2;i=5",
"data_type": "double"
}, },
"data": {...}, {
"schema": {...}, "server_id": "1",
"table_name": "predictions", "type": "confidence",
"model_id": "temp_pred_001", "addr": "ns=2;i=5",
"model_name": "temperature_prediction_model", "data_type": "double"
"input_filters": {
"EMPTY_DATA": {"POLICY": "STOP"},
"SPECIFIC_VARIABLES_NULL_VALUES": {
"POLICY": "STOP",
"config": {"variables": ["temperature", "humidity"]}
} }
],
"input_filters": [
{
"filter_name": "EMPTY_DATA", # Required filter
"policy": "STOP"
}, },
"mlflow_transform_filters": { {
"API_ERROR": {"POLICY": "STOP"} "filter_name": "SPECIFIC_VARIABLES_NULL_VALUES",
"policy": "CONTINUE",
"config": {
"variables": [
"Counter"
]
}
}
],
"mlflow_transform_filters": [
{
"filter_name": "API_ERROR", # Required filter
"policy": "REPEAT"
}, },
"mlflow_predict_filters": { {
"API_ERROR": {"POLICY": "STOP"}, "filter_name": "NAN_VALUES",
"NAN_VALUES": {"POLICY": "STOP"} "policy": "STOP"
}, }
"model_retention": 60, ],
"path_priority": ["STOP", "CONTINUE", "REPEAT"], "mlflow_predict_filters": [
"opc_output_config": {...} {
"filter_name": "API_ERROR", # Required filter
"policy": "CONTINUE"
}
],
"path_priority": [ # In case of multiple filters catch problems, this will determine the path to take
"STOP",
"CONTINUE",
"REPEAT"
],
"active": true,
"datetime_columns": [ # Columns in data comming from query that are datetime
"timestamp",
"created_at"
],
"updated_at": {
"$date": "2025-08-27T18:35:01.600Z"
}
} }
``` ```