Files
sientia-dataops-model-manager/values.yaml

389 lines
12 KiB
YAML

#
# Default values for sientia-model-manager using the sientia-module chart (0.6.x).
# This is a YAML-formatted file.
# Declare variables to be passed into your templates.
#
projectName: &projectName "sientia-model-manager"
# -----------------------------------------------------------------------------
# Global configuration shared by all runtimes
# -----------------------------------------------------------------------------
global:
# Namespace used by the chart.
namespace: sientia
# -----------------------------------------------------------------------------
# Image configuration (chart-level)
# -----------------------------------------------------------------------------
# The sientia-module chart allows overriding the image used by all runtimes.
# Per requirement, we deploy using the sientia-module image v1.0.0.
image:
repository: aignosi.azurecr.io/sientia-module
pullPolicy: Always
tag: "1.0.2"
# Common labels applied to pods (can be extended per project).
commonLabels: {}
# Resources inherited by all runtimes unless overridden.
resources: {}
# We usually recommend not to specify default resources and to leave this as a conscious
# choice for the user. This also increases chances charts run on environments with little
# resources, such as Minikube. If you do want to specify resources, uncomment the following
# lines, adjust them as necessary, and remove the curly braces after 'resources:'.
# limits:
# cpu: 100m
# memory: 128Mi
# requests:
# cpu: 100m
# memory: 128Mi
# Probes inherited by all runtimes unless overridden.
# More information:
# https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-startup-probes/
livenessProbe:
exec:
command:
- python3
- -c
- "import requests; requests.get('http://localhost:9090/metrics')"
initialDelaySeconds: 20
periodSeconds: 30
readinessProbe:
exec:
command:
- python3
- -c
- "import requests; requests.get('http://localhost:9090/metrics')"
initialDelaySeconds: 10
periodSeconds: 15
# Autoscaling configuration inherited by all runtimes unless overridden.
# More information:
# https://kubernetes.io/docs/concepts/workloads/autoscaling/
autoscaling:
enabled: false
minReplicas: 1
maxReplicas: 100
targetCPUUtilizationPercentage: 80
# targetMemoryUtilizationPercentage: 80
# Environment variables shared by all runtimes.
env:
# Entrypoint variables
- name: GITHUB_REPO_URL
value: "git@github.com:Aignosi/sientia-dataops-model-manager.git"
- name: GITHUB_BRANCH
value: "release/SIENTIAPDE-1645"
- name: PYTHON_APP
value: "model_manager.worker.worker"
- name: PYPI_SERVER
value: "http://library-distribution-server.library.svc.cluster.local:5000"
- name: POSTGRES_HOST
value: "paradedb-rw.paradedb.svc.cluster.local"
- name: POSTGRES_PORT
value: "5432"
- name: POSTGRES_USER
value: "postgres"
- name: POSTGRES_PASSWORD
value: "nFqc81y6kwmr2zuAIx43DhiOosFCVPpeEfTtTWZflkNjB2j1KtEeIANkhFR9mAX3"
- name: POSTGRES_DBNAME
value: "sientia-core-mlops-bff"
- name: POSTGRES_MIN_CONNECTIONS
value: "10"
- name: POSTGRES_MAX_CONNECTIONS
value: "30"
- name: MLFLOW_URL
value: "http://sientia-tracker-mlflow-tracking.sientia-tracker.svc.cluster.local:80"
- name: MLFLOW_USERNAME
value: "aignosi"
- name: MLFLOW_PASSWORD
value: "1L0FP50j3ncp123"
- name: LOG_LEVEL
value: "DEBUG"
- name: HTTP_METRICS_PORT
value: "9090"
- name: HTTP_SDK_METRICS_PORT
value: "9091"
- name: TEMPORAL_HOST
value: "temporal-frontend.temporal.svc.cluster.local:7233"
- name: TEMPORAL_NAMESPACE
value: "model-manager"
- name: TRAIN_TASK_QUEUE
value: "train_model-queue"
- name: CLEANUP_TASK_QUEUE
value: "cleanup-queue"
- name: TEMPORAL_USE_TLS
value: "false"
- name: STORE_BASE_URL
value: "http://gitea-http.gitea.svc.cluster.local:3000"
- name: STORE_OWNER
value: "aignosi"
- name: STORE_REPO
value: "suse-model-store"
- name: STORE_USERNAME
valueFrom:
secretKeyRef:
name: sientia-plugin-store-credentials
key: username
- name: STORE_PASSWORD
valueFrom:
secretKeyRef:
name: sientia-plugin-store-credentials
key: password
- name: STORE_CACHE_TTL_SECONDS
value: "3600"
- name: MONGODB_USERNAME
value: "root"
- name: MONGODB_PASSWORD
value: "wKZDbMNU1c"
- name: MONGODB_URL
value: "my-release-mongodb.mongodb.svc.cluster.local:27017"
- name: MONGODB_DATABASE
value: "sientia"
- name: MONGODB_TTL_INDEX_HOURS
value: "1"
- name: MINIO_ENDPOINT_URL
value: "http://minio.minio.svc.cluster.local:9000"
- name: MINIO_ACCESS_KEY
value: "model-training-user"
- name: MINIO_SECRET_KEY
value: "modelTrainingUser123"
- name: MINIO_REGION
value: "us-east-1"
- name: MINIO_SECURE
value: "false"
- name: MINIO_MAX_RETRY_ATTEMPTS
value: "3"
- name: MINIO_RETRY_MODE
value: "adaptive"
- name: MINIO_CONNECT_TIMEOUT
value: "10"
- name: MINIO_READ_TIMEOUT
value: "60"
- name: MINIO_DEFAULT_BUCKET
value: "model-training"
- name: TIMEOUT_VALIDATE_PARAMS
value: "30"
- name: TIMEOUT_TRAIN_MODEL
value: "2700"
- name: TIMEOUT_DELETE_FILE
value: "120"
- name: TIMEOUT_UPDATE_DATABASE
value: "30"
- name: CLEANUP_RETENTION_HOURS
value: "24"
- name: CLEANUP_DRY_RUN
value: "false"
- name: TIMEOUT_CLEANUP_MINIO
value: "300"
- name: TIMEOUT_CLEANUP_LOCAL
value: "120"
- name: MAX_KEYS_CLEANUP
value: "1000"
- name: DEFAULT_CLEANUP_BUCKET
value: "model-training"
# Cleanup Schedule Configuration
- name: CLEANUP_SCHEDULE_ID
value: "cleanup-files-daily"
- name: CLEANUP_CRON
value: "0 0 * * *" # Midnight UTC
- name: CLEANUP_TIMEZONE
value: "UTC"
- name: CLEANUP_EXECUTION_TIMEOUT_HOURS
value: "1"
# -----------------------------------------------------------------------------
# Runtimes configuration
# -----------------------------------------------------------------------------
# Each runtime inherits settings from `global` (resources, env, probes, autoscaling)
# unless overridden here.
runtimes:
basic:
# Replicas for this runtime. Replaces the old replicaCount.
replicas: 1
xgboost:
replicas: 1
# -----------------------------------------------------------------------------
# Chart-level configuration (applies to all runtimes)
# -----------------------------------------------------------------------------
# This is for the secrets for pulling an image from a private repository more information can be found here: https://kubernetes.io/docs/tasks/configure-pod-container/pull-image-private-registry/
imagePullSecrets:
- name: docker-hub-secret
# This is to override the chart name.
nameOverride: *projectName
fullnameOverride: *projectName
# This section builds out the service account more information can be found here: https://kubernetes.io/docs/concepts/security/service-accounts/
serviceAccount:
# Specifies whether a service account should be created
create: true
# Automatically mount a ServiceAccount's API credentials?
automount: true
# Annotations to add to the service account
annotations: {}
# The name of the service account to use.
# If not set and create is true, a name is generated using the fullname template
name: *projectName
# This is for setting Kubernetes Annotations to a Pod.
# For more information checkout: https://kubernetes.io/docs/concepts/overview/working-with-objects/annotations/
podAnnotations: {}
# This is for setting Kubernetes Labels to a Pod.
# For more information checkout: https://kubernetes.io/docs/concepts/overview/working-with-objects/labels/
podLabels: {}
podSecurityContext: {}
# fsGroup: 2000
securityContext: {}
# capabilities:
# drop:
# - ALL
# readOnlyRootFilesystem: true
# runAsNonRoot: true
# runAsUser: 1000
# Additional volumes on the output Deployment definition.
volumes:
- name: model-manager-runtime
emptyDir:
sizeLimit: 1Gi
# Additional volumeMounts on the output Deployment definition.
volumeMounts:
- name: model-manager-runtime
mountPath: "/var/lib/model-manager"
# Deployment strategy configuration
# More information: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy
deploymentStrategy:
type: Recreate
# rollingUpdate:
# maxSurge: 0
# maxUnavailable: 1
# Number of old ReplicaSets to retain
revisionHistoryLimit: 2
nodeSelector: {}
tolerations: []
affinity: {}
services:
sdk-metrics:
enabled: true
type: ClusterIP
port: 9091
targetPort: 9091
name: sdk-metrics
metrics:
enabled: true
type: ClusterIP
port: 9090
targetPort: 9090
name: metrics
# Configuração do ServiceMonitor para o Prometheus Operator
# ref: https://github.com/prometheus-operator/prometheus-operator
serviceMonitor:
# Se true, um recurso ServiceMonitor será criado.
enabled: true
# O intervalo no qual as métricas devem ser coletadas (ex: 30s, 1m).
endpoints:
- port: metrics
path: /metrics
interval: 30s
relabelings: []
- port: sdk-metrics
path: /metrics
interval: 30s
relabelings: []
additionalLabels:
release: kube-prometheus-stack
ssh:
enabled: true
secretName: git-ssh-key-sientia-model-manager-worker
sshPath: /mnt/.ssh
knownHostsPath: /mnt/known_hosts
# Configuração para dashboards do Grafana
grafanaDashboard:
# Habilita a criação de ConfigMaps para dashboards
enabled: true
# Namespace onde o Grafana está instalado (ajuste conforme seu ambiente)
namespace: monitoring
# Labels para que o sidecar do Grafana encontre os dashboards
labels:
grafana_dashboard: "1"
# Lista de dashboards para importar
dashboards:
- name: sientia-dataops-model-manager
title: "Sientia DataOps Model Manager"
uid: "sientia-dataops-model-manager"
folder: "Sientia"
jsonFile: "dashboards/sientia-dataops-model-manager.json"
overwrite: true # Sobrescreve dashboard se já existir
version: "1.0.0" # Version inicial do dashboard
# Configuração para datasources do Grafana
grafanaDatasource:
# Habilita a criação de ConfigMap para datasources
enabled: false
# Namespace onde o Grafana está instalado
namespace: monitoring
# Labels para que o sidecar do Grafana encontre os datasources
labels:
grafana_datasource: "1"
# Lista de datasources para configurar
datasources: []
# Exemplo de datasource:
# - name: Prometheus
# type: prometheus
# url: http://prometheus-server.monitoring.svc.cluster.local
# isDefault: true
# jsonData:
# timeInterval: "5s"
# -----------------------------------------------------------------------------
# Helm usage examples
# -----------------------------------------------------------------------------
# kubectl create secret docker-registry docker-hub-secret --namespace sientia --docker-server=http://aignosi.azurecr.io --docker-username=aignosi --docker-password=<pwd>
#
# helm upgrade --install sientia-model-manager sientia/sientia-module -n sientia --create-namespace -f ./values.yaml --version 0.6.1
#
# Global/runtimes layout note:
# - Shared configuration lives under `global` (env, probes, autoscaling, namespace).
# - Individual runtimes are defined under `runtimes`, each with its own `name` and `replicas`.
# - Runtimes inherit `global` settings unless overridden at the runtime level.
#
# kubectl create secret generic git-ssh-key-sientia-model-manager-worker \
# --namespace sientia \
# --from-file=ssh-privatekey=git_key \
# --type=kubernetes.io/ssh-auth
# kubectl create secret generic sientia-plugin-store-credentials \
# --namespace sientia \
# --from-literal=username=<username> \
# --from-literal=password=<password>