SIENTIAPDE-1273

Enhance resource management and configuration in Laborious worker

- Updated `values.yaml` to define resource limits and requests for better performance tuning.
- Modified environment variables in `worker.py` to support resource-based scaling and improved task queue management.
- Introduced new functions for creating resource tuners and poller behaviors, enhancing scalability and efficiency in handling workloads.
This commit is contained in:
vitor-aignosi
2025-11-19 11:53:50 -03:00
parent 7e4e048ca3
commit 8f6a3cd93d
2 changed files with 101 additions and 38 deletions

View File

@@ -5,12 +5,15 @@ This module provides the main worker implementation for the Sientia DataOps Labo
It orchestrates Temporal workers, manages task queues, and handles the lifecycle of It orchestrates Temporal workers, manages task queues, and handles the lifecycle of
prediction and retraining workflows. prediction and retraining workflows.
The worker supports two main task queues: The worker supports multiple task queues:
- predictions_batch-queue: Handles batch prediction workflows - predictions_batch-queue: Handles batch prediction workflows (heavy workload)
- minimal_retrain-queue: Handles model retraining workflows - minimal_retrain-queue: Handles model retraining workflows
- drift-queue: Handles drift detection workflows
- simple_metrics-queue: Handles simple metrics calculation workflows
Key Features: Key Features:
- Automatic scaling with PollerBehaviorAutoscaling - Resource-based scaling with WorkerTuner (CPU and memory aware)
- Automatic polling scaling with PollerBehaviorAutoscaling
- Prometheus metrics integration - Prometheus metrics integration
- Comprehensive error handling and logging - Comprehensive error handling and logging
- Graceful shutdown with cleanup - Graceful shutdown with cleanup
@@ -23,16 +26,38 @@ Environment Variables:
- HTTP_METRICS_PORT: Prometheus metrics server port (default: 9090) - HTTP_METRICS_PORT: Prometheus metrics server port (default: 9090)
- HTTP_SDK_METRICS_PORT: Temporal SDK metrics port (default: 9091) - HTTP_SDK_METRICS_PORT: Temporal SDK metrics port (default: 9091)
- PROJECT_NAME: Project name for notifications (default: laborious) - PROJECT_NAME: Project name for notifications (default: laborious)
Tuner Configuration (Resource-based scaling):
- TUNER_TARGET_MEMORY_USAGE: Target memory usage (0.0-1.0, default: 0.75)
- TUNER_TARGET_CPU_USAGE: Target CPU usage (0.0-1.0, default: 0.80)
- TUNER_WORKFLOW_MIN_SLOTS: Minimum workflow slots (default: 5)
- TUNER_WORKFLOW_MAX_SLOTS: Maximum workflow slots (default: 50)
- TUNER_ACTIVITY_MIN_SLOTS: Minimum activity slots (default: 5)
- TUNER_ACTIVITY_MAX_SLOTS: Maximum activity slots (default: 50)
- TUNER_WORKFLOW_RAMP_THROTTLE_MS: Workflow ramp throttle in ms (default: 100)
- TUNER_ACTIVITY_RAMP_THROTTLE_MS: Activity ramp throttle in ms (default: 50)
Poller Configuration:
- POLLER_MINIMUM: Minimum number of pollers (default: 1)
- POLLER_MAXIMUM: Maximum number of pollers (default: 10)
- POLLER_INITIAL: Initial number of pollers (default: 2)
""" """
import os
from temporalio import client, workflow from temporalio import client, workflow
from temporalio.runtime import PrometheusConfig, Runtime, TelemetryConfig from temporalio.runtime import PrometheusConfig, Runtime, TelemetryConfig
from temporalio.worker import PollerBehaviorAutoscaling, Worker from temporalio.worker import (
PollerBehaviorAutoscaling,
ResourceBasedSlotConfig,
Worker,
WorkerTuner,
)
with workflow.unsafe.imports_passed_through(): with workflow.unsafe.imports_passed_through():
import asyncio import asyncio
import os
import sys import sys
from datetime import timedelta
from prometheus_client import start_http_server from prometheus_client import start_http_server
from sientia_do.notifications.handlers import CoreNotificationHandler as NotificationHandler from sientia_do.notifications.handlers import CoreNotificationHandler as NotificationHandler
@@ -60,6 +85,49 @@ POD_ID = os.getenv('POD_ID')
SDK_METRICS_PORT = int(os.getenv('HTTP_SDK_METRICS_PORT', '9091')) SDK_METRICS_PORT = int(os.getenv('HTTP_SDK_METRICS_PORT', '9091'))
def create_resource_tuner() -> WorkerTuner:
"""Create a resource-based tuner from environment variables."""
target_memory = float(os.getenv('TUNER_TARGET_MEMORY_USAGE', '0.75'))
target_cpu = float(os.getenv('TUNER_TARGET_CPU_USAGE', '0.50'))
workflow_min = int(os.getenv('TUNER_WORKFLOW_MIN_SLOTS', '5'))
workflow_max = int(os.getenv('TUNER_WORKFLOW_MAX_SLOTS', '50'))
activity_min = int(os.getenv('TUNER_ACTIVITY_MIN_SLOTS', '5'))
activity_max = int(os.getenv('TUNER_ACTIVITY_MAX_SLOTS', '50'))
local_activity_min = int(os.getenv('TUNER_LOCAL_ACTIVITY_MIN_SLOTS', '1'))
local_activity_max = int(os.getenv('TUNER_LOCAL_ACTIVITY_MAX_SLOTS', '30'))
workflow_ramp = int(os.getenv('TUNER_WORKFLOW_RAMP_THROTTLE_MS', '100'))
activity_ramp = int(os.getenv('TUNER_ACTIVITY_RAMP_THROTTLE_MS', '50'))
local_activity_ramp = int(os.getenv('TUNER_LOCAL_ACTIVITY_RAMP_THROTTLE_MS', '50'))
return WorkerTuner.create_resource_based(
target_memory_usage=target_memory,
target_cpu_usage=target_cpu,
workflow_config=ResourceBasedSlotConfig(
minimum_slots=workflow_min,
maximum_slots=workflow_max,
ramp_throttle=timedelta(milliseconds=workflow_ramp),
),
activity_config=ResourceBasedSlotConfig(
minimum_slots=activity_min,
maximum_slots=activity_max,
ramp_throttle=timedelta(milliseconds=activity_ramp),
),
local_activity_config=ResourceBasedSlotConfig(
minimum_slots=local_activity_min,
maximum_slots=local_activity_max,
ramp_throttle=timedelta(milliseconds=local_activity_ramp),
),
)
def create_poller_behavior() -> PollerBehaviorAutoscaling:
"""Create poller behavior from environment variables."""
minimum = int(os.getenv('POLLER_MINIMUM', '1'))
maximum = int(os.getenv('POLLER_MAXIMUM', '10'))
initial = int(os.getenv('POLLER_INITIAL', '2'))
return PollerBehaviorAutoscaling(minimum=minimum, maximum=maximum, initial=initial)
async def main(): async def main():
""" """
Main entry point for the Laborious worker application. Main entry point for the Laborious worker application.
@@ -138,6 +206,9 @@ async def main():
logger.custom_info('Starting Workers...', metadata) logger.custom_info('Starting Workers...', metadata)
tuner = create_resource_tuner()
poller = create_poller_behavior()
workers = [ workers = [
Worker( Worker(
temporal_client, temporal_client,
@@ -151,12 +222,10 @@ async def main():
activities.format_retrain_report, activities.format_retrain_report,
activities.export_data_to_postgres, activities.export_data_to_postgres,
], ],
max_concurrent_workflow_tasks=50, tuner=tuner,
max_concurrent_activities=50,
max_concurrent_local_activities=50,
max_cached_workflows=2, max_cached_workflows=2,
workflow_task_poller_behavior=PollerBehaviorAutoscaling(), workflow_task_poller_behavior=poller,
activity_task_poller_behavior=PollerBehaviorAutoscaling(), activity_task_poller_behavior=poller,
), ),
Worker( Worker(
temporal_client, temporal_client,
@@ -168,12 +237,10 @@ async def main():
activities.calculate_drift, activities.calculate_drift,
activities.export_data_to_postgres, activities.export_data_to_postgres,
], ],
max_concurrent_workflow_tasks=50, tuner=tuner,
max_concurrent_activities=50,
max_concurrent_local_activities=50,
max_cached_workflows=2, max_cached_workflows=2,
workflow_task_poller_behavior=PollerBehaviorAutoscaling(), workflow_task_poller_behavior=poller,
activity_task_poller_behavior=PollerBehaviorAutoscaling(), activity_task_poller_behavior=poller,
), ),
Worker( Worker(
temporal_client, temporal_client,
@@ -184,12 +251,10 @@ async def main():
activities.calculate_simple_metrics, activities.calculate_simple_metrics,
activities.export_data_to_postgres, activities.export_data_to_postgres,
], ],
max_concurrent_workflow_tasks=50, tuner=tuner,
max_concurrent_activities=50,
max_concurrent_local_activities=50,
max_cached_workflows=2, max_cached_workflows=2,
workflow_task_poller_behavior=PollerBehaviorAutoscaling(), workflow_task_poller_behavior=poller,
activity_task_poller_behavior=PollerBehaviorAutoscaling(), activity_task_poller_behavior=poller,
), ),
Worker( Worker(
temporal_client, temporal_client,
@@ -215,12 +280,10 @@ async def main():
activities.export_data_to_postgres, activities.export_data_to_postgres,
activities.write_metrics, activities.write_metrics,
], ],
max_concurrent_workflow_tasks=50, tuner=tuner,
max_concurrent_activities=50,
max_concurrent_local_activities=50,
max_cached_workflows=200, max_cached_workflows=200,
workflow_task_poller_behavior=PollerBehaviorAutoscaling(), workflow_task_poller_behavior=poller,
activity_task_poller_behavior=PollerBehaviorAutoscaling(), activity_task_poller_behavior=poller,
), ),
] ]

View File

@@ -52,17 +52,15 @@ securityContext: {}
# runAsUser: 1000 # runAsUser: 1000
resources: {} resources:
# We usually recommend not to specify default resources and to leave this as a conscious # Resource limits and requests are important for ResourceBasedTuner to work correctly.
# choice for the user. This also increases chances charts run on environments with little # The tuner monitors system CPU and memory usage, so proper resource limits must be set.
# resources, such as Minikube. If you do want to specify resources, uncomment the following limits:
# lines, adjust them as necessary, and remove the curly braces after 'resources:'. cpu: 2000m # 2 CPU cores
# limits: memory: 20Gi # 20 GB memory
# cpu: 100m requests:
# memory: 128Mi cpu: 1000m # 1 CPU core
# requests: memory: 2Gi # 2 GB memory
# cpu: 100m
# memory: 128Mi
# This is to setup the liveness and readiness probes more information can be found here: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-startup-probes/ # This is to setup the liveness and readiness probes more information can be found here: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-startup-probes/
livenessProbe: livenessProbe:
@@ -167,9 +165,11 @@ env:
- name: POSTGRES_DBNAME - name: POSTGRES_DBNAME
value: "sientia" value: "sientia"
- name: POSTGRES_MIN_CONNECTIONS - name: POSTGRES_MIN_CONNECTIONS
value: "10" value: "20"
# max_connections = number_of_workers * max_concurrent_activities * safety_factor
# Example: 4 workers * 50 activities * 0.5 = 100 connections
- name: POSTGRES_MAX_CONNECTIONS - name: POSTGRES_MAX_CONNECTIONS
value: "30" value: "100"
- name: MLFLOW_HOST - name: MLFLOW_HOST
value: "http://sientia-tracker-mlflow-tracking.sientia-tracker.svc.cluster.local" value: "http://sientia-tracker-mlflow-tracking.sientia-tracker.svc.cluster.local"