# Default values for sientia-module. # This is a YAML-formatted file. # Declare variables to be passed into your templates. # This will set the replicaset count more information can be found here: https://kubernetes.io/docs/concepts/workloads/controllers/replicaset/ replicaCount: 1 # This sets the container image more information can be found here: https://kubernetes.io/docs/concepts/containers/images/ image: repository: aignosi.azurecr.io/sientia-dataops-model-manager # This sets the pull policy for images. pullPolicy: IfNotPresent # Overrides the image tag whose default is the chart appVersion. tag: "1.2.0" # This is for the secrets for pulling an image from a private repository more information can be found here: https://kubernetes.io/docs/tasks/configure-pod-container/pull-image-private-registry/ imagePullSecrets: - name: docker-hub-secret # This is to override the chart name. nameOverride: "sientia-dataops-model-manager" fullnameOverride: "sientia-dataops-model-manager" namespace: sientia # This section builds out the service account more information can be found here: https://kubernetes.io/docs/concepts/security/service-accounts/ serviceAccount: # Specifies whether a service account should be created create: true # Automatically mount a ServiceAccount's API credentials? automount: true # Annotations to add to the service account annotations: {} # The name of the service account to use. # If not set and create is true, a name is generated using the fullname template name: "sientia-dataops-model-manager" # This is for setting Kubernetes Annotations to a Pod. # For more information checkout: https://kubernetes.io/docs/concepts/overview/working-with-objects/annotations/ podAnnotations: {} # This is for setting Kubernetes Labels to a Pod. # For more information checkout: https://kubernetes.io/docs/concepts/overview/working-with-objects/labels/ podLabels: {} podSecurityContext: {} # fsGroup: 2000 securityContext: {} # capabilities: # drop: # - ALL # readOnlyRootFilesystem: true # runAsNonRoot: true # runAsUser: 1000 resources: {} # We usually recommend not to specify default resources and to leave this as a conscious # choice for the user. This also increases chances charts run on environments with little # resources, such as Minikube. If you do want to specify resources, uncomment the following # lines, adjust them as necessary, and remove the curly braces after 'resources:'. # limits: # cpu: 100m # memory: 128Mi # requests: # cpu: 100m # memory: 128Mi # This is to setup the liveness and readiness probes more information can be found here: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-startup-probes/ livenessProbe: exec: command: - python3 - -c - "import requests; requests.get('http://localhost:9090/metrics')" initialDelaySeconds: 20 periodSeconds: 30 readinessProbe: exec: command: - python3 - -c - "import requests; requests.get('http://localhost:9090/metrics')" initialDelaySeconds: 10 periodSeconds: 15 # This section is for setting up autoscaling more information can be found here: https://kubernetes.io/docs/concepts/workloads/autoscaling/ autoscaling: enabled: false minReplicas: 1 maxReplicas: 100 targetCPUUtilizationPercentage: 80 # targetMemoryUtilizationPercentage: 80 # Additional volumes on the output Deployment definition. volumes: - name: reports-volume emptyDir: sizeLimit: 1Gi # Additional volumeMounts on the output Deployment definition. volumeMounts: - name: reports-volume mountPath: "/app/model_manager/reports/temp" # Deployment strategy configuration # More information: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy deploymentStrategy: type: Recreate # rollingUpdate: # maxSurge: 0 # maxUnavailable: 1 # Number of old ReplicaSets to retain revisionHistoryLimit: 2 nodeSelector: {} tolerations: [] affinity: {} services: sdk-metrics: enabled: true type: ClusterIP port: 9091 targetPort: 9091 name: sdk-metrics metrics: enabled: true type: ClusterIP port: 9090 targetPort: 9090 name: metrics # Configuração do ServiceMonitor para o Prometheus Operator # ref: https://github.com/prometheus-operator/prometheus-operator serviceMonitor: # Se true, um recurso ServiceMonitor será criado. enabled: true # O intervalo no qual as métricas devem ser coletadas (ex: 30s, 1m). endpoints: - port: metrics path: /metrics interval: 30s relabelings: [] - port: sdk-metrics path: /metrics interval: 30s relabelings: [] additionalLabels: release: kube-prometheus-stack env: - name: POSTGRES_HOST value: "paradedb-rw.paradedb.svc.cluster.local" - name: POSTGRES_PORT value: "5432" - name: POSTGRES_USER value: "postgres" - name: POSTGRES_PASSWORD value: "nFqc81y6kwmr2zuAIx43DhiOosFCVPpeEfTtTWZflkNjB2j1KtEeIANkhFR9mAX3" - name: POSTGRES_DBNAME value: "sientia-core-mlops-bff" - name: POSTGRES_MIN_CONNECTIONS value: "10" - name: POSTGRES_MAX_CONNECTIONS value: "30" - name: MLFLOW_URL value: "http://sientia-tracker-mlflow-tracking.sientia-tracker.svc.cluster.local:80" - name: MLFLOW_USERNAME value: "aignosi" - name: MLFLOW_PASSWORD value: "1L0FP50j3ncp123" - name: LOG_LEVEL value: "DEBUG" - name: HTTP_METRICS_PORT value: "9090" - name: HTTP_SDK_METRICS_PORT value: "9091" - name: PROJECT_NAME value: "sientia-model-manager" - name: TEMPORAL_HOST value: "temporal-frontend.temporal.svc.cluster.local:7233" - name: TEMPORAL_NAMESPACE value: "model-manager" - name: TRAIN_TASK_QUEUE value: "train_model-queue" - name: CLEANUP_TASK_QUEUE value: "cleanup-queue" - name: TEMPORAL_USE_TLS value: "false" - name: MONGODB_USERNAME value: "root" - name: MONGODB_PASSWORD value: "wKZDbMNU1c" - name: MONGODB_URL value: "my-release-mongodb.mongodb.svc.cluster.local:27017" - name: MONGODB_DATABASE value: "sientia" - name: MONGODB_TTL_INDEX_HOURS value: "1" - name: MINIO_ENDPOINT_URL value: "http://minio.minio.svc.cluster.local:9000" - name: MINIO_ACCESS_KEY value: "model-training-user" - name: MINIO_SECRET_KEY value: "modelTrainingUser123" - name: MINIO_REGION value: "us-east-1" - name: MINIO_USE_SSL value: "false" - name: MINIO_MAX_RETRY_ATTEMPTS value: "3" - name: MINIO_RETRY_MODE value: "adaptive" - name: MINIO_CONNECT_TIMEOUT value: "10" - name: MINIO_READ_TIMEOUT value: "60" - name: TIMEOUT_VALIDATE_PARAMS value: "30" - name: TIMEOUT_TRAIN_MODEL value: "2700" - name: TIMEOUT_DELETE_FILE value: "120" - name: TIMEOUT_UPDATE_DATABASE value: "30" - name: CLEANUP_RETENTION_HOURS value: "24" - name: CLEANUP_DRY_RUN value: "false" - name: TIMEOUT_CLEANUP_LOCAL value: "120" # Cleanup Schedule Configuration - name: CLEANUP_SCHEDULE_ID value: "cleanup-files-daily" - name: CLEANUP_CRON value: "0 0 * * *" # Midnight UTC - name: CLEANUP_TIMEZONE value: "UTC" - name: CLEANUP_EXECUTION_TIMEOUT_HOURS value: "1" - name: EXTRA_PIP_REQUIREMENTS value: "git+https://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@github.com/Aignosi/sientia-mlops-library.git" - name: POD_ID valueFrom: fieldRef: fieldPath: metadata.name ssh: enabled: false secretName: git-ssh-key-sientia-model-manager-worker sshPath: /mnt/.ssh knownHostsPath: /mnt/known_hosts # Configuração para dashboards do Grafana grafanaDashboard: # Habilita a criação de ConfigMaps para dashboards enabled: true # Namespace onde o Grafana está instalado (ajuste conforme seu ambiente) namespace: monitoring # Labels para que o sidecar do Grafana encontre os dashboards labels: grafana_dashboard: "1" # Lista de dashboards para importar dashboards: - name: sientia-dataops-model-manager title: "Sientia DataOps Model Manager" uid: "sientia-dataops-model-manager" folder: "Sientia" jsonFile: "dashboards/sientia-dataops-model-manager.json" overwrite: true # Sobrescreve dashboard se já existir version: "1.0.0" # Version inicial do dashboard # Configuração para datasources do Grafana grafanaDatasource: # Habilita a criação de ConfigMap para datasources enabled: false # Namespace onde o Grafana está instalado namespace: monitoring # Labels para que o sidecar do Grafana encontre os datasources labels: grafana_datasource: "1" # Lista de datasources para configurar datasources: [] # Exemplo de datasource: # - name: Prometheus # type: prometheus # url: http://prometheus-server.monitoring.svc.cluster.local # isDefault: true # jsonData: # timeInterval: "5s" # kubectl create secret docker-registry docker-hub-secret --namespace sientia --docker-server=http://aignosi.azurecr.io --docker-username=aignosi --docker-password=5I5zpQ6sRaHqX1hD3dr+2mo647yO3FRc359/wu6gsP+ACRDRz5mp # helm upgrade --install sientia-dataops-model-manager sientia/sientia-module -n sientia --create-namespace -f ./values.yaml --version 0.6.0 # kubectl create secret generic git-ssh-key-sientia-model-manager-worker \ # --namespace sientia \ # --from-file=ssh-privatekey=git_key \ # --type=kubernetes.io/ssh-auth