- Modified `.env.example` to set local defaults for PostgreSQL, MLflow, and MinIO configurations. - Added MongoDB configuration parameters to the environment setup. - Updated `README.md` to reflect changes in workflow input parameters and task queue naming conventions. - Removed the `ModelServing` class to streamline the codebase, as it was deemed unnecessary. - Adjusted `connectors_config.py` to align with new environment variable names and improve clarity. - Updated tests to reflect changes in configuration handling and removed tests related to the deleted `ModelServing` class.
387 lines
12 KiB
YAML
387 lines
12 KiB
YAML
#
|
|
# Default values for sientia-dataops-model-manager using the sientia-module chart (0.6.x).
|
|
# This is a YAML-formatted file.
|
|
# Declare variables to be passed into your templates.
|
|
#
|
|
|
|
projectName: &projectName "sientia-dataops-model-manager"
|
|
|
|
# -----------------------------------------------------------------------------
|
|
# Image configuration (chart-level)
|
|
# -----------------------------------------------------------------------------
|
|
# The sientia-module chart allows overriding the image used by all runtimes.
|
|
# Per requirement, we deploy using the sientia-module image v1.0.0.
|
|
image:
|
|
repository: aignosi.azurecr.io/sientia-module
|
|
pullPolicy: IfNotPresent
|
|
tag: "1.0.0"
|
|
|
|
# -----------------------------------------------------------------------------
|
|
# Global configuration shared by all runtimes
|
|
# -----------------------------------------------------------------------------
|
|
global:
|
|
# Namespace used by the chart.
|
|
namespace: sientia
|
|
|
|
# Common labels applied to pods (can be extended per project).
|
|
commonLabels: {}
|
|
|
|
# Resources inherited by all runtimes unless overridden.
|
|
resources: {}
|
|
# We usually recommend not to specify default resources and to leave this as a conscious
|
|
# choice for the user. This also increases chances charts run on environments with little
|
|
# resources, such as Minikube. If you do want to specify resources, uncomment the following
|
|
# lines, adjust them as necessary, and remove the curly braces after 'resources:'.
|
|
# limits:
|
|
# cpu: 100m
|
|
# memory: 128Mi
|
|
# requests:
|
|
# cpu: 100m
|
|
# memory: 128Mi
|
|
|
|
# Probes inherited by all runtimes unless overridden.
|
|
# More information:
|
|
# https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-startup-probes/
|
|
livenessProbe:
|
|
exec:
|
|
command:
|
|
- python3
|
|
- -c
|
|
- "import requests; requests.get('http://localhost:9090/metrics')"
|
|
initialDelaySeconds: 20
|
|
periodSeconds: 30
|
|
|
|
readinessProbe:
|
|
exec:
|
|
command:
|
|
- python3
|
|
- -c
|
|
- "import requests; requests.get('http://localhost:9090/metrics')"
|
|
initialDelaySeconds: 10
|
|
periodSeconds: 15
|
|
|
|
# Autoscaling configuration inherited by all runtimes unless overridden.
|
|
# More information:
|
|
# https://kubernetes.io/docs/concepts/workloads/autoscaling/
|
|
autoscaling:
|
|
enabled: false
|
|
minReplicas: 1
|
|
maxReplicas: 100
|
|
targetCPUUtilizationPercentage: 80
|
|
# targetMemoryUtilizationPercentage: 80
|
|
|
|
# Environment variables shared by all runtimes.
|
|
env:
|
|
- name: POSTGRES_HOST
|
|
value: "paradedb-rw.paradedb.svc.cluster.local"
|
|
- name: POSTGRES_PORT
|
|
value: "5432"
|
|
- name: POSTGRES_USER
|
|
value: "postgres"
|
|
- name: POSTGRES_PASSWORD
|
|
value: "nFqc81y6kwmr2zuAIx43DhiOosFCVPpeEfTtTWZflkNjB2j1KtEeIANkhFR9mAX3"
|
|
- name: POSTGRES_DBNAME
|
|
value: "sientia-core-mlops-bff"
|
|
- name: POSTGRES_MIN_CONNECTIONS
|
|
value: "10"
|
|
- name: POSTGRES_MAX_CONNECTIONS
|
|
value: "30"
|
|
|
|
- name: MLFLOW_URL
|
|
value: "http://sientia-tracker-mlflow-tracking.sientia-tracker.svc.cluster.local:80"
|
|
- name: MLFLOW_USERNAME
|
|
value: "aignosi"
|
|
- name: MLFLOW_PASSWORD
|
|
value: "1L0FP50j3ncp123"
|
|
|
|
- name: LOG_LEVEL
|
|
value: "DEBUG"
|
|
- name: HTTP_METRICS_PORT
|
|
value: "9090"
|
|
- name: HTTP_SDK_METRICS_PORT
|
|
value: "9091"
|
|
|
|
- name: TEMPORAL_HOST
|
|
value: "temporal-frontend.temporal.svc.cluster.local:7233"
|
|
- name: TEMPORAL_NAMESPACE
|
|
value: "model-manager"
|
|
- name: TRAIN_TASK_QUEUE
|
|
value: "train_model-queue"
|
|
- name: CLEANUP_TASK_QUEUE
|
|
value: "cleanup-queue"
|
|
- name: TEMPORAL_USE_TLS
|
|
value: "false"
|
|
|
|
- name: STORE_BASE_URL
|
|
value: "http://gitea-http.gitea.svc.cluster.local"
|
|
- name: STORE_OWNER
|
|
value: "aignosi"
|
|
- name: STORE_REPO
|
|
value: "suse-model-store"
|
|
- name: STORE_USERNAME
|
|
valueFrom:
|
|
secretKeyRef:
|
|
name: sientia-plugin-store-credentials
|
|
key: username
|
|
- name: STORE_PASSWORD
|
|
valueFrom:
|
|
secretKeyRef:
|
|
name: sientia-plugin-store-credentials
|
|
key: password
|
|
- name: STORE_CACHE_TTL_SECONDS
|
|
value: "3600"
|
|
|
|
- name: MONGODB_USERNAME
|
|
value: "root"
|
|
- name: MONGODB_PASSWORD
|
|
value: "wKZDbMNU1c"
|
|
- name: MONGODB_URL
|
|
value: "my-release-mongodb.mongodb.svc.cluster.local:27017"
|
|
- name: MONGODB_DATABASE
|
|
value: "sientia"
|
|
- name: MONGODB_TTL_INDEX_HOURS
|
|
value: "1"
|
|
|
|
- name: MINIO_ENDPOINT_URL
|
|
value: "http://minio.minio.svc.cluster.local:9000"
|
|
- name: MINIO_ACCESS_KEY
|
|
value: "model-training-user"
|
|
- name: MINIO_SECRET_KEY
|
|
value: "modelTrainingUser123"
|
|
- name: MINIO_REGION
|
|
value: "us-east-1"
|
|
- name: MINIO_SECURE
|
|
value: "false"
|
|
- name: MINIO_MAX_RETRY_ATTEMPTS
|
|
value: "3"
|
|
- name: MINIO_RETRY_MODE
|
|
value: "adaptive"
|
|
- name: MINIO_CONNECT_TIMEOUT
|
|
value: "10"
|
|
- name: MINIO_READ_TIMEOUT
|
|
value: "60"
|
|
- name: MINIO_DEFAULT_BUCKET
|
|
value: "model-training"
|
|
|
|
- name: TIMEOUT_VALIDATE_PARAMS
|
|
value: "30"
|
|
- name: TIMEOUT_TRAIN_MODEL
|
|
value: "2700"
|
|
- name: TIMEOUT_DELETE_FILE
|
|
value: "120"
|
|
- name: TIMEOUT_UPDATE_DATABASE
|
|
value: "30"
|
|
|
|
- name: CLEANUP_RETENTION_HOURS
|
|
value: "24"
|
|
- name: CLEANUP_DRY_RUN
|
|
value: "false"
|
|
- name: TIMEOUT_CLEANUP_MINIO
|
|
value: "300"
|
|
- name: TIMEOUT_CLEANUP_LOCAL
|
|
value: "120"
|
|
- name: MAX_KEYS_CLEANUP
|
|
value: "1000"
|
|
- name: DEFAULT_CLEANUP_BUCKET
|
|
value: "model-training"
|
|
|
|
# Cleanup Schedule Configuration
|
|
- name: CLEANUP_SCHEDULE_ID
|
|
value: "cleanup-files-daily"
|
|
- name: CLEANUP_CRON
|
|
value: "0 0 * * *" # Midnight UTC
|
|
- name: CLEANUP_TIMEZONE
|
|
value: "UTC"
|
|
- name: CLEANUP_EXECUTION_TIMEOUT_HOURS
|
|
value: "1"
|
|
|
|
- name: PYPI_SERVER
|
|
value: "http://library-distribution-server.library.svc.cluster.local:5000"
|
|
- name: PYPI_USERNAME
|
|
valueFrom:
|
|
secretKeyRef:
|
|
name: sientia-plugin-store-credentials
|
|
key: pypi_username
|
|
optional: true
|
|
- name: PYPI_PASSWORD
|
|
valueFrom:
|
|
secretKeyRef:
|
|
name: sientia-plugin-store-credentials
|
|
key: pypi_password
|
|
optional: true
|
|
|
|
# -----------------------------------------------------------------------------
|
|
# Runtimes configuration
|
|
# -----------------------------------------------------------------------------
|
|
# Each runtime inherits settings from `global` (resources, env, probes, autoscaling)
|
|
# unless overridden here.
|
|
runtimes:
|
|
- name: "model-manager-worker"
|
|
# Replicas for this runtime. Replaces the old replicaCount.
|
|
replicas: 1
|
|
|
|
# -----------------------------------------------------------------------------
|
|
# Chart-level configuration (applies to all runtimes)
|
|
# -----------------------------------------------------------------------------
|
|
|
|
# This is for the secrets for pulling an image from a private repository more information can be found here: https://kubernetes.io/docs/tasks/configure-pod-container/pull-image-private-registry/
|
|
imagePullSecrets:
|
|
- name: docker-hub-secret
|
|
|
|
# This is to override the chart name.
|
|
nameOverride: *projectName
|
|
fullnameOverride: *projectName
|
|
|
|
# This section builds out the service account more information can be found here: https://kubernetes.io/docs/concepts/security/service-accounts/
|
|
serviceAccount:
|
|
# Specifies whether a service account should be created
|
|
create: true
|
|
# Automatically mount a ServiceAccount's API credentials?
|
|
automount: true
|
|
# Annotations to add to the service account
|
|
annotations: {}
|
|
# The name of the service account to use.
|
|
# If not set and create is true, a name is generated using the fullname template
|
|
name: *projectName
|
|
|
|
# This is for setting Kubernetes Annotations to a Pod.
|
|
# For more information checkout: https://kubernetes.io/docs/concepts/overview/working-with-objects/annotations/
|
|
podAnnotations: {}
|
|
# This is for setting Kubernetes Labels to a Pod.
|
|
# For more information checkout: https://kubernetes.io/docs/concepts/overview/working-with-objects/labels/
|
|
podLabels: {}
|
|
|
|
podSecurityContext: {}
|
|
# fsGroup: 2000
|
|
|
|
securityContext: {}
|
|
# capabilities:
|
|
# drop:
|
|
# - ALL
|
|
# readOnlyRootFilesystem: true
|
|
# runAsNonRoot: true
|
|
# runAsUser: 1000
|
|
|
|
# Additional volumes on the output Deployment definition.
|
|
volumes:
|
|
- name: reports-volume
|
|
emptyDir:
|
|
sizeLimit: 1Gi
|
|
|
|
# Additional volumeMounts on the output Deployment definition.
|
|
volumeMounts:
|
|
- name: reports-volume
|
|
mountPath: "/app/model_manager/reports/temp"
|
|
|
|
# Deployment strategy configuration
|
|
# More information: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy
|
|
deploymentStrategy:
|
|
type: Recreate
|
|
# rollingUpdate:
|
|
# maxSurge: 0
|
|
# maxUnavailable: 1
|
|
|
|
# Number of old ReplicaSets to retain
|
|
revisionHistoryLimit: 2
|
|
|
|
nodeSelector: {}
|
|
|
|
tolerations: []
|
|
|
|
affinity: {}
|
|
|
|
services:
|
|
sdk-metrics:
|
|
enabled: true
|
|
type: ClusterIP
|
|
port: 9091
|
|
targetPort: 9091
|
|
name: sdk-metrics
|
|
metrics:
|
|
enabled: true
|
|
type: ClusterIP
|
|
port: 9090
|
|
targetPort: 9090
|
|
name: metrics
|
|
|
|
# Configuração do ServiceMonitor para o Prometheus Operator
|
|
# ref: https://github.com/prometheus-operator/prometheus-operator
|
|
serviceMonitor:
|
|
# Se true, um recurso ServiceMonitor será criado.
|
|
enabled: true
|
|
# O intervalo no qual as métricas devem ser coletadas (ex: 30s, 1m).
|
|
endpoints:
|
|
- port: metrics
|
|
path: /metrics
|
|
interval: 30s
|
|
relabelings: []
|
|
- port: sdk-metrics
|
|
path: /metrics
|
|
interval: 30s
|
|
relabelings: []
|
|
additionalLabels:
|
|
release: kube-prometheus-stack
|
|
|
|
ssh:
|
|
enabled: false
|
|
secretName: git-ssh-key-sientia-model-manager-worker
|
|
sshPath: /mnt/.ssh
|
|
knownHostsPath: /mnt/known_hosts
|
|
|
|
# Configuração para dashboards do Grafana
|
|
grafanaDashboard:
|
|
# Habilita a criação de ConfigMaps para dashboards
|
|
enabled: true
|
|
# Namespace onde o Grafana está instalado (ajuste conforme seu ambiente)
|
|
namespace: monitoring
|
|
# Labels para que o sidecar do Grafana encontre os dashboards
|
|
labels:
|
|
grafana_dashboard: "1"
|
|
# Lista de dashboards para importar
|
|
dashboards:
|
|
- name: sientia-dataops-model-manager
|
|
title: "Sientia DataOps Model Manager"
|
|
uid: "sientia-dataops-model-manager"
|
|
folder: "Sientia"
|
|
jsonFile: "dashboards/sientia-dataops-model-manager.json"
|
|
overwrite: true # Sobrescreve dashboard se já existir
|
|
version: "1.0.0" # Version inicial do dashboard
|
|
|
|
# Configuração para datasources do Grafana
|
|
grafanaDatasource:
|
|
# Habilita a criação de ConfigMap para datasources
|
|
enabled: false
|
|
# Namespace onde o Grafana está instalado
|
|
namespace: monitoring
|
|
# Labels para que o sidecar do Grafana encontre os datasources
|
|
labels:
|
|
grafana_datasource: "1"
|
|
# Lista de datasources para configurar
|
|
datasources: []
|
|
# Exemplo de datasource:
|
|
# - name: Prometheus
|
|
# type: prometheus
|
|
# url: http://prometheus-server.monitoring.svc.cluster.local
|
|
# isDefault: true
|
|
# jsonData:
|
|
# timeInterval: "5s"
|
|
|
|
# -----------------------------------------------------------------------------
|
|
# Helm usage examples
|
|
# -----------------------------------------------------------------------------
|
|
# kubectl create secret docker-registry docker-hub-secret --namespace sientia --docker-server=http://aignosi.azurecr.io --docker-username=aignosi --docker-password=<pwd>
|
|
#
|
|
# helm upgrade --install sientia-dataops-model-manager sientia/sientia-module -n sientia --create-namespace -f ./values.yaml --version 0.6.1
|
|
#
|
|
# Global/runtimes layout note:
|
|
# - Shared configuration lives under `global` (env, probes, autoscaling, namespace).
|
|
# - Individual runtimes are defined under `runtimes`, each with its own `name` and `replicas`.
|
|
# - Runtimes inherit `global` settings unless overridden at the runtime level.
|
|
#
|
|
# kubectl create secret generic git-ssh-key-sientia-model-manager-worker \
|
|
# --namespace sientia \
|
|
# --from-file=ssh-privatekey=git_key \
|
|
# --type=kubernetes.io/ssh-auth
|
|
|
|
|