Update values.yaml to adjust workflow and activity poller configurations for improved performance, increasing minimum, initial, and maximum values. Additionally, add new minimal retrain parameters to enhance retraining capabilities.
316 lines
10 KiB
YAML
316 lines
10 KiB
YAML
# Default values for sientia-module.
|
|
# This is a YAML-formatted file.
|
|
# Declare variables to be passed into your templates.
|
|
|
|
# This will set the replicaset count more information can be found here: https://kubernetes.io/docs/concepts/workloads/controllers/replicaset/
|
|
replicaCount: 1
|
|
|
|
# This sets the container image more information can be found here: https://kubernetes.io/docs/concepts/containers/images/
|
|
image:
|
|
repository: aignosi.azurecr.io/sientia-module
|
|
# This sets the pull policy for images.
|
|
pullPolicy: Always
|
|
# Overrides the image tag whose default is the chart appVersion.
|
|
tag: "1.1.2"
|
|
|
|
# This is for the secrets for pulling an image from a private repository more information can be found here: https://kubernetes.io/docs/tasks/configure-pod-container/pull-image-private-registry/
|
|
imagePullSecrets:
|
|
- name: docker-hub-secret
|
|
# This is to override the chart name.
|
|
nameOverride: "sientia-laborious-worker"
|
|
fullnameOverride: "sientia-laborious-worker"
|
|
namespace: sientia
|
|
|
|
# This section builds out the service account more information can be found here: https://kubernetes.io/docs/concepts/security/service-accounts/
|
|
serviceAccount:
|
|
# Specifies whether a service account should be created
|
|
create: true
|
|
# Automatically mount a ServiceAccount's API credentials?
|
|
automount: true
|
|
# Annotations to add to the service account
|
|
annotations: {}
|
|
# The name of the service account to use.
|
|
# If not set and create is true, a name is generated using the fullname template
|
|
name: "sientia-laborious-worker"
|
|
|
|
# This is for setting Kubernetes Annotations to a Pod.
|
|
# For more information checkout: https://kubernetes.io/docs/concepts/overview/working-with-objects/annotations/
|
|
podAnnotations: {}
|
|
# This is for setting Kubernetes Labels to a Pod.
|
|
# For more information checkout: https://kubernetes.io/docs/concepts/overview/working-with-objects/labels/
|
|
podLabels: {}
|
|
|
|
podSecurityContext: {}
|
|
# fsGroup: 2000
|
|
|
|
securityContext: {}
|
|
# capabilities:
|
|
# drop:
|
|
# - ALL
|
|
# readOnlyRootFilesystem: true
|
|
# runAsNonRoot: true
|
|
# runAsUser: 1000
|
|
|
|
|
|
resources:
|
|
# Resource limits and requests are important for ResourceBasedTuner to work correctly.
|
|
# The tuner monitors system CPU and memory usage, so proper resource limits must be set.
|
|
limits:
|
|
cpu: 2000m # 2 CPU cores
|
|
memory: 20Gi # 20 GB memory
|
|
requests:
|
|
cpu: 1000m # 1 CPU core
|
|
memory: 2Gi # 2 GB memory
|
|
|
|
# This is to setup the liveness and readiness probes more information can be found here: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-startup-probes/
|
|
# This is to setup the liveness and readiness probes more information can be found here: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-startup-probes/
|
|
livenessProbe:
|
|
exec:
|
|
command:
|
|
- sh
|
|
- -c
|
|
- |
|
|
curl -sf http://localhost:9090/metrics | grep -q '^app_up{.*} 1'
|
|
initialDelaySeconds: 1260
|
|
periodSeconds: 15
|
|
timeoutSeconds: 5
|
|
failureThreshold: 3
|
|
|
|
readinessProbe:
|
|
exec:
|
|
command:
|
|
- sh
|
|
- -c
|
|
- |
|
|
curl -sf http://localhost:9090/metrics | grep -q '^app_up{.*} 1'
|
|
initialDelaySeconds: 1200
|
|
periodSeconds: 10
|
|
timeoutSeconds: 3
|
|
failureThreshold: 2
|
|
|
|
|
|
|
|
# This section is for setting up autoscaling more information can be found here: https://kubernetes.io/docs/concepts/workloads/autoscaling/
|
|
autoscaling:
|
|
enabled: false
|
|
minReplicas: 1
|
|
maxReplicas: 100
|
|
targetCPUUtilizationPercentage: 80
|
|
# targetMemoryUtilizationPercentage: 80
|
|
|
|
# Additional volumes on the output Deployment definition.
|
|
volumes: []
|
|
# - name: foo
|
|
# secret:
|
|
# secretName: mysecret
|
|
# optional: false
|
|
|
|
# Additional volumeMounts on the output Deployment definition.
|
|
volumeMounts: []
|
|
# - name: foo
|
|
# mountPath: "/etc/foo"
|
|
# readOnly: true
|
|
|
|
nodeSelector: {}
|
|
|
|
tolerations: []
|
|
|
|
affinity: {}
|
|
|
|
services:
|
|
sdk-metrics:
|
|
enabled: true
|
|
type: ClusterIP
|
|
port: 9091
|
|
targetPort: 9091
|
|
name: sdk-metrics
|
|
|
|
metrics:
|
|
enabled: true
|
|
type: ClusterIP
|
|
port: 9090
|
|
targetPort: 9090
|
|
name: metrics
|
|
|
|
# Configuração do ServiceMonitor para o Prometheus Operator
|
|
# ref: https://github.com/prometheus-operator/prometheus-operator
|
|
serviceMonitor:
|
|
# Se true, um recurso ServiceMonitor será criado.
|
|
enabled: true
|
|
# O intervalo no qual as métricas devem ser coletadas (ex: 30s, 1m).
|
|
endpoints:
|
|
- port: metrics
|
|
path: /metrics
|
|
interval: 30s
|
|
relabelings: []
|
|
- port: sdk-metrics
|
|
path: /metrics
|
|
interval: 30s
|
|
relabelings: []
|
|
|
|
additionalLabels:
|
|
release: kube-prometheus-stack
|
|
|
|
|
|
env:
|
|
# Entrypoint variables
|
|
- name: GITHUB_REPO_URL
|
|
value: "git@github.com:Aignosi/sientia-dataops-laborious_temporal.git"
|
|
- name: GITHUB_BRANCH
|
|
value: "feature/SIENTIAPDE-1712"
|
|
- name: PYTHON_APP
|
|
value: "laborious.worker.worker"
|
|
|
|
# Application variables
|
|
- name: POSTGRES_HOST
|
|
value: "paradedb-rw.paradedb.svc.cluster.local"
|
|
- name: POSTGRES_PORT
|
|
value: "5432"
|
|
- name: POSTGRES_USER
|
|
value: "postgres"
|
|
- name: POSTGRES_PASSWORD
|
|
value: "nFqc81y6kwmr2zuAIx43DhiOosFCVPpeEfTtTWZflkNjB2j1KtEeIANkhFR9mAX3"
|
|
- name: POSTGRES_DBNAME
|
|
value: "sientia"
|
|
- name: POSTGRES_MIN_CONNECTIONS
|
|
value: "20"
|
|
# max_connections = number_of_workers * max_concurrent_activities * safety_factor
|
|
# Example: 4 workers * 50 activities * 0.5 = 100 connections
|
|
- name: POSTGRES_MAX_CONNECTIONS
|
|
value: "100"
|
|
|
|
- name: MLFLOW_HOST
|
|
value: "http://sientia-tracker-mlflow-tracking.sientia-tracker.svc.cluster.local"
|
|
- name: MLFLOW_PORT
|
|
value: "80"
|
|
- name: MLFLOW_USERNAME
|
|
value: "aignosi"
|
|
- name: MLFLOW_PASSWORD
|
|
value: "1L0FP50j3ncp123"
|
|
|
|
- name: OPC_ID
|
|
value: "1"
|
|
- name: OPC_SERVER_NAME
|
|
value: "default_server"
|
|
- name: OPC_URL
|
|
value: "opc.tcp://sientia-opc-simulator-opc.sientia.svc.cluster.local:4840"
|
|
|
|
|
|
- name: LOG_LEVEL
|
|
value: "DEBUG"
|
|
- name: HTTP_METRICS_PORT
|
|
value: "9090"
|
|
- name: HTTP_SDK_METRICS_PORT
|
|
value: "9091"
|
|
- name: PROJECT_NAME
|
|
value: "sientia-laborious"
|
|
|
|
- name: TEMPORAL_HOST
|
|
value: "temporal-frontend.temporal.svc.cluster.local:7233"
|
|
- name: TEMPORAL_NAMESPACE
|
|
value: "laborious"
|
|
|
|
- name: MONGODB_USERNAME
|
|
value: "root"
|
|
- name: MONGODB_PASSWORD
|
|
value: "wKZDbMNU1c"
|
|
- name: MONGODB_URL
|
|
value: "my-release-mongodb.mongodb.svc.cluster.local:27017"
|
|
- name: MONGODB_DATABASE
|
|
value: "sientia"
|
|
- name: MONGODB_TTL_INDEX_HOURS
|
|
value: "1"
|
|
|
|
- name: MINIO_ENDPOINT_URL
|
|
value: "minio.minio.svc.cluster.local:9000"
|
|
- name: MINIO_ACCESS_KEY
|
|
value: "admin"
|
|
- name: MINIO_SECRET_KEY
|
|
value: "LiArt4eNmJ"
|
|
- name: MINIO_DEFAULT_BUCKET
|
|
value: "sientia"
|
|
- name: MINIO_RETENTION_HOURS
|
|
value: "24"
|
|
- name: SIENTIA_MINIO_OFFLOAD_THRESHOLD_MEGABYTES
|
|
value: "0.5"
|
|
|
|
# Temporal worker tuning for PredictionsBatch.
|
|
# IMPORTANT: prefix must be PREDICTIONSBATCH_ (from class name PredictionsBatch).
|
|
# Keep workflow-task concurrency moderate to reduce task completion races under load.
|
|
- name: PREDICTIONSBATCH_MAX_CONCURRENT_WORKFLOW_TASKS
|
|
value: "20"
|
|
# Allow higher activity parallelism because most activities are I/O-bound, but keep headroom.
|
|
- name: PREDICTIONSBATCH_MAX_CONCURRENT_ACTIVITIES
|
|
value: "60"
|
|
# Keep local activities controlled so they do not monopolize the event loop.
|
|
- name: PREDICTIONSBATCH_MAX_CONCURRENT_LOCAL_ACTIVITIES
|
|
value: "20"
|
|
# Cache enough workflows for reuse without excessive memory growth.
|
|
- name: PREDICTIONSBATCH_MAX_CACHED_WORKFLOWS
|
|
value: "200"
|
|
# Start with one workflow poller to avoid burst contention at startup.
|
|
- name: PREDICTIONSBATCH_WORKFLOW_POLLER_BEHAVIOUR_MINIMUM
|
|
value: "3"
|
|
# Small initial poller count warms up gradually instead of spiking task fetches.
|
|
- name: PREDICTIONSBATCH_WORKFLOW_POLLER_BEHAVIOUR_INITIAL
|
|
value: "5"
|
|
# Cap workflow pollers to limit scheduling pressure and avoid over-polling.
|
|
- name: PREDICTIONSBATCH_WORKFLOW_POLLER_BEHAVIOUR_MAXIMUM
|
|
value: "15"
|
|
# Keep at least two activity pollers so activity queues do not starve during spikes.
|
|
- name: PREDICTIONSBATCH_ACTIVITY_POLLER_BEHAVIOUR_MINIMUM
|
|
value: "3"
|
|
# Moderate initial activity pollers for faster ramp-up with controlled pressure.
|
|
- name: PREDICTIONSBATCH_ACTIVITY_POLLER_BEHAVIOUR_INITIAL
|
|
value: "10"
|
|
# Limit max activity pollers to preserve CPU for workflow-task completion.
|
|
- name: PREDICTIONSBATCH_ACTIVITY_POLLER_BEHAVIOUR_MAXIMUM
|
|
value: "30"
|
|
|
|
- name: MINIMALRETRAIN_MAX_CONCURRENT_ACTIVITIES
|
|
value: "1"
|
|
- name: MINIMALRETRAIN_MAX_CONCURRENT_LOCAL_ACTIVITIES
|
|
value: "1"
|
|
- name: MINIMALRETRAIN_MAX_CACHED_WORKFLOWS
|
|
value: "1"
|
|
- name: MINIMALRETRAIN_WORKFLOW_POLLER_BEHAVIOUR_MINIMUM
|
|
value: "1"
|
|
- name: MINIMALRETRAIN_WORKFLOW_POLLER_BEHAVIOUR_INITIAL
|
|
value: "1"
|
|
- name: MINIMALRETRAIN_WORKFLOW_POLLER_BEHAVIOUR_MAXIMUM
|
|
value: "1"
|
|
- name: MINIMALRETRAIN_ACTIVITY_POLLER_BEHAVIOUR_MINIMUM
|
|
value: "1"
|
|
- name: MINIMALRETRAIN_ACTIVITY_POLLER_BEHAVIOUR_INITIAL
|
|
value: "1"
|
|
- name: MINIMALRETRAIN_ACTIVITY_POLLER_BEHAVIOUR_MAXIMUM
|
|
value: "1"
|
|
|
|
- name: PI_WEB_API_BASE_URL
|
|
value: "https://pivision.votorantimcimentos.com/piwebapi"
|
|
- name: PI_WEB_API_AUTH_TYPE
|
|
value: "basic"
|
|
- name: PI_WEB_API_AUTH_TOKEN
|
|
valueFrom:
|
|
secretKeyRef:
|
|
name: pi-web-api-auth-token
|
|
key: token
|
|
|
|
- name: PYPI_SERVER
|
|
value: "http://library-distribution-server.library.svc.cluster.local:5000"
|
|
|
|
ssh:
|
|
enabled: true
|
|
secretName: git-ssh-key-sientia-laborious-worker
|
|
sshPath: /mnt/.ssh
|
|
knownHostsPath: /mnt/known_hosts
|
|
|
|
# kubectl create secret docker-registry docker-hub-secret --namespace sientia --docker-server=http://aignosi.azurecr.io --docker-username=aignosi --docker-password=5I5zpQ6sRaHqX1hD3dr+2mo647yO3FRc359/wu6gsP+ACRDRz5mp
|
|
|
|
# helm upgrade --install sientia-laborious-worker sientia/sientia-module -n sientia --create-namespace -f ./values.yaml --version 0.6.0
|
|
|
|
# kubectl create secret generic git-ssh-key-sientia-laborious-worker \
|
|
# --namespace sientia \
|
|
# --from-file=ssh-privatekey=git_key \
|
|
# --type=kubernetes.io/ssh-auth
|