From 7bb24b42f615d09482fe9c94be8e2b988a98f2c1 Mon Sep 17 00:00:00 2001 From: vitor-aignosi Date: Fri, 1 Aug 2025 12:01:03 -0300 Subject: [PATCH] SIENTIAPDE-1174 feat: update dependencies and enhance metrics integration - Updated sientia-dataops-library dependency version from 1.3.5 to 1.3.7 in requirements.txt. - Incremented image tag in values.yaml from 0.2.7 to 0.3.1. - Added Prometheus metrics service configuration in values.yaml, enabling metrics collection. - Implemented Prometheus client in worker.py to start a metrics server and track application health. --- orchestrator/metrics.py | 16 ++++++++++++++ orchestrator/worker/worker.py | 22 ++++++++++++++++++- requirements.txt | 3 ++- values.yaml | 40 +++++++++++++++++++++++------------ 4 files changed, 66 insertions(+), 15 deletions(-) create mode 100644 orchestrator/metrics.py diff --git a/orchestrator/metrics.py b/orchestrator/metrics.py new file mode 100644 index 0000000..c7bc20e --- /dev/null +++ b/orchestrator/metrics.py @@ -0,0 +1,16 @@ +from prometheus_client import Gauge, Counter + +APP_UP = Gauge( + "app_up", + "Indicates if the application is running (1) or shutting down (0)", + ["pod_id"], +) + +CORE_LABELS = ["pod_id", "model_name", "pipeline_name"] + + +EMAIL_SENT_COUNT = Counter( + "email_sent_count", + "Number of emails sent", + [*CORE_LABELS, "email_group"], +) diff --git a/orchestrator/worker/worker.py b/orchestrator/worker/worker.py index fcd815d..89959e3 100644 --- a/orchestrator/worker/worker.py +++ b/orchestrator/worker/worker.py @@ -22,6 +22,10 @@ with workflow.unsafe.imports_passed_through(): ) from sientia_do.notifications.handlers import CoreNotificationHandler as NotificationHandler from sientia_do.temporal.utils.logger import get_logger + from prometheus_client import start_http_server + from orchestrator import metrics + +POD_ID = os.getenv("POD_ID") async def main(): @@ -29,7 +33,10 @@ async def main(): namespace = os.getenv('TEMPORAL_NAMESPACE', 'default') logger = get_logger(__name__) - logger.info('Starting Worker...') + logger.info(f'Starting Worker with POD_ID: {POD_ID}') + + logger.info("Starting prometheus client...") + start_prometheus_server() logger.info('Starting Notification Handler...') @@ -164,7 +171,20 @@ async def main(): if activities: activities.shutdown() # Exit with a non-zero status code to indicate failure to Kubernetes + metrics.APP_UP.labels(pod_id=POD_ID).set(0) # Mark app as DOWN sys.exit(1) + +def start_prometheus_server(): + try: + port = int(os.getenv("HTTP_METRICS_PORT", 9090)) + start_http_server(port) + print(f"Prometheus server started on port {port}.") + metrics.APP_UP.labels(pod_id=POD_ID).set(1) + except Exception as e: + print(f"Failed to start Prometheus server: {e}") + os._exit(1) + + if __name__ == '__main__': asyncio.run(main()) diff --git a/requirements.txt b/requirements.txt index 166f96d..f7a80cf 100644 --- a/requirements.txt +++ b/requirements.txt @@ -5,4 +5,5 @@ redis couchbase pymongo jinja2 -git+ssh://git@github.com/Aignosi/sientia-dataops-library.git@1.3.5 +git+ssh://git@github.com/Aignosi/sientia-dataops-library.git@1.3.7 +prometheus-client diff --git a/values.yaml b/values.yaml index 570ae29..802fba4 100644 --- a/values.yaml +++ b/values.yaml @@ -11,7 +11,7 @@ image: # This sets the pull policy for images. pullPolicy: Always # Overrides the image tag whose default is the chart appVersion. - tag: "0.2.7" + tag: "0.3.1" # This is for the secrets for pulling an image from a private repository more information can be found here: https://kubernetes.io/docs/tasks/configure-pod-container/pull-image-private-registry/ imagePullSecrets: @@ -112,19 +112,31 @@ tolerations: [] affinity: {} services: - api: - enabled: false + metrics: + enabled: true type: ClusterIP - port: 4841 - targetPort: 4841 - name: api + port: 9090 + targetPort: 9090 + name: metrics - opc: - enabled: false - type: ClusterIP - port: 4840 - targetPort: 4840 - name: server +# Configuração do ServiceMonitor para o Prometheus Operator +# ref: https://github.com/prometheus-operator/prometheus-operator +serviceMonitor: + # Se true, um recurso ServiceMonitor será criado. + enabled: true + # O intervalo no qual as métricas devem ser coletadas (ex: 30s, 1m). + interval: 30s + # O path do endpoint de métricas na sua aplicação. + path: /metrics + # Labels adicionais para o recurso ServiceMonitor. + # Essencial para que o Prometheus Operator o descubra. Se você usa o helm chart kube-prometheus-stack, + # ele procura por ServiceMonitors com o label "release: kube-prometheus-stack". + additionalLabels: + release: kube-prometheus-stack + # Configurações de relabeling adicionais, se necessário. + # ref: https://prometheus.io/docs/prometheus/latest/configuration/configuration/#relabel_config + relabelings: [] + port: metrics env: @@ -132,7 +144,7 @@ env: - name: GITHUB_REPO_URL value: "git@github.com:Aignosi/sientia-dataops-orchestrator_temporal.git" - name: GITHUB_BRANCH - value: "SIENTIAPDE-1172-criar-pipeline-de-alertas-orquestrador" + value: "SIENTIAPDE-1174-mapear-e-implementar-metricas-a-serem-criadas" - name: PYTHON_APP value: "orchestrator.worker.worker" @@ -203,6 +215,8 @@ env: - name: LOG_LEVEL value: "DEBUG" + - name: HTTP_METRICS_PORT + value: "9090" - name: PROJECT_NAME value: "sientia-orchestrator"