SIENTIAPDE-1174
feat: update dependencies and enhance metrics integration - Updated sientia-dataops-library dependency version from 1.3.5 to 1.3.7 in requirements.txt. - Incremented image tag in values.yaml from 0.2.7 to 0.3.1. - Added Prometheus metrics service configuration in values.yaml, enabling metrics collection. - Implemented Prometheus client in worker.py to start a metrics server and track application health.
This commit is contained in:
16
orchestrator/metrics.py
Normal file
16
orchestrator/metrics.py
Normal file
@@ -0,0 +1,16 @@
|
||||
from prometheus_client import Gauge, Counter
|
||||
|
||||
APP_UP = Gauge(
|
||||
"app_up",
|
||||
"Indicates if the application is running (1) or shutting down (0)",
|
||||
["pod_id"],
|
||||
)
|
||||
|
||||
CORE_LABELS = ["pod_id", "model_name", "pipeline_name"]
|
||||
|
||||
|
||||
EMAIL_SENT_COUNT = Counter(
|
||||
"email_sent_count",
|
||||
"Number of emails sent",
|
||||
[*CORE_LABELS, "email_group"],
|
||||
)
|
||||
@@ -22,6 +22,10 @@ with workflow.unsafe.imports_passed_through():
|
||||
)
|
||||
from sientia_do.notifications.handlers import CoreNotificationHandler as NotificationHandler
|
||||
from sientia_do.temporal.utils.logger import get_logger
|
||||
from prometheus_client import start_http_server
|
||||
from orchestrator import metrics
|
||||
|
||||
POD_ID = os.getenv("POD_ID")
|
||||
|
||||
|
||||
async def main():
|
||||
@@ -29,7 +33,10 @@ async def main():
|
||||
namespace = os.getenv('TEMPORAL_NAMESPACE', 'default')
|
||||
logger = get_logger(__name__)
|
||||
|
||||
logger.info('Starting Worker...')
|
||||
logger.info(f'Starting Worker with POD_ID: {POD_ID}')
|
||||
|
||||
logger.info("Starting prometheus client...")
|
||||
start_prometheus_server()
|
||||
|
||||
logger.info('Starting Notification Handler...')
|
||||
|
||||
@@ -164,7 +171,20 @@ async def main():
|
||||
if activities:
|
||||
activities.shutdown()
|
||||
# Exit with a non-zero status code to indicate failure to Kubernetes
|
||||
metrics.APP_UP.labels(pod_id=POD_ID).set(0) # Mark app as DOWN
|
||||
sys.exit(1)
|
||||
|
||||
|
||||
def start_prometheus_server():
|
||||
try:
|
||||
port = int(os.getenv("HTTP_METRICS_PORT", 9090))
|
||||
start_http_server(port)
|
||||
print(f"Prometheus server started on port {port}.")
|
||||
metrics.APP_UP.labels(pod_id=POD_ID).set(1)
|
||||
except Exception as e:
|
||||
print(f"Failed to start Prometheus server: {e}")
|
||||
os._exit(1)
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
asyncio.run(main())
|
||||
|
||||
@@ -5,4 +5,5 @@ redis
|
||||
couchbase
|
||||
pymongo
|
||||
jinja2
|
||||
git+ssh://git@github.com/Aignosi/sientia-dataops-library.git@1.3.5
|
||||
git+ssh://git@github.com/Aignosi/sientia-dataops-library.git@1.3.7
|
||||
prometheus-client
|
||||
|
||||
40
values.yaml
40
values.yaml
@@ -11,7 +11,7 @@ image:
|
||||
# This sets the pull policy for images.
|
||||
pullPolicy: Always
|
||||
# Overrides the image tag whose default is the chart appVersion.
|
||||
tag: "0.2.7"
|
||||
tag: "0.3.1"
|
||||
|
||||
# This is for the secrets for pulling an image from a private repository more information can be found here: https://kubernetes.io/docs/tasks/configure-pod-container/pull-image-private-registry/
|
||||
imagePullSecrets:
|
||||
@@ -112,19 +112,31 @@ tolerations: []
|
||||
affinity: {}
|
||||
|
||||
services:
|
||||
api:
|
||||
enabled: false
|
||||
metrics:
|
||||
enabled: true
|
||||
type: ClusterIP
|
||||
port: 4841
|
||||
targetPort: 4841
|
||||
name: api
|
||||
port: 9090
|
||||
targetPort: 9090
|
||||
name: metrics
|
||||
|
||||
opc:
|
||||
enabled: false
|
||||
type: ClusterIP
|
||||
port: 4840
|
||||
targetPort: 4840
|
||||
name: server
|
||||
# Configuração do ServiceMonitor para o Prometheus Operator
|
||||
# ref: https://github.com/prometheus-operator/prometheus-operator
|
||||
serviceMonitor:
|
||||
# Se true, um recurso ServiceMonitor será criado.
|
||||
enabled: true
|
||||
# O intervalo no qual as métricas devem ser coletadas (ex: 30s, 1m).
|
||||
interval: 30s
|
||||
# O path do endpoint de métricas na sua aplicação.
|
||||
path: /metrics
|
||||
# Labels adicionais para o recurso ServiceMonitor.
|
||||
# Essencial para que o Prometheus Operator o descubra. Se você usa o helm chart kube-prometheus-stack,
|
||||
# ele procura por ServiceMonitors com o label "release: kube-prometheus-stack".
|
||||
additionalLabels:
|
||||
release: kube-prometheus-stack
|
||||
# Configurações de relabeling adicionais, se necessário.
|
||||
# ref: https://prometheus.io/docs/prometheus/latest/configuration/configuration/#relabel_config
|
||||
relabelings: []
|
||||
port: metrics
|
||||
|
||||
|
||||
env:
|
||||
@@ -132,7 +144,7 @@ env:
|
||||
- name: GITHUB_REPO_URL
|
||||
value: "git@github.com:Aignosi/sientia-dataops-orchestrator_temporal.git"
|
||||
- name: GITHUB_BRANCH
|
||||
value: "SIENTIAPDE-1172-criar-pipeline-de-alertas-orquestrador"
|
||||
value: "SIENTIAPDE-1174-mapear-e-implementar-metricas-a-serem-criadas"
|
||||
- name: PYTHON_APP
|
||||
value: "orchestrator.worker.worker"
|
||||
|
||||
@@ -203,6 +215,8 @@ env:
|
||||
|
||||
- name: LOG_LEVEL
|
||||
value: "DEBUG"
|
||||
- name: HTTP_METRICS_PORT
|
||||
value: "9090"
|
||||
- name: PROJECT_NAME
|
||||
value: "sientia-orchestrator"
|
||||
|
||||
|
||||
Reference in New Issue
Block a user