From 159607fc7213b578ac2a5d9a2da87b13815c5dff Mon Sep 17 00:00:00 2001 From: Bruno Domingues Date: Wed, 8 Jul 2026 20:22:32 -0300 Subject: [PATCH] SIENTIAPDE-1945: Adopt shared Docker image and Kubernetes deployment strategy. Removed the local Dockerfile, .dockerignore, and values.yaml, as image building and Kubernetes deployment are now handled by external sientia-module components. Updated README accordingly. --- .dockerignore | 93 --------------- Dockerfile | 78 ------------- README.md | 78 ++----------- values.yaml | 312 -------------------------------------------------- 4 files changed, 7 insertions(+), 554 deletions(-) delete mode 100644 .dockerignore delete mode 100644 Dockerfile delete mode 100644 values.yaml diff --git a/.dockerignore b/.dockerignore deleted file mode 100644 index 7f786a5..0000000 --- a/.dockerignore +++ /dev/null @@ -1,93 +0,0 @@ -# ============================================================================ -# WHITELIST APPROACH: Block everything by default, then allow only what's needed -# ============================================================================ - -# Block everything first -* - -# ============================================================================ -# ALLOW: Application source code (model_manager package) -# ============================================================================ - -# Allow the main package directory and all Python files -!model_manager/ -!model_manager/**/*.py -!model_manager/**/__init__.py - -# Allow subdirectories structure -!model_manager/activities/ -!model_manager/activities/** -!model_manager/schedules/ -!model_manager/schedules/** -!model_manager/sientia/ -!model_manager/sientia/** -!model_manager/utils/ -!model_manager/utils/** -!model_manager/utils/models/ -!model_manager/utils/models/** -!model_manager/utils/repository/ -!model_manager/utils/repository/** -!model_manager/worker/ -!model_manager/worker/** -!model_manager/workflows/ -!model_manager/workflows/** - -# Allow reports directory with header.html -!model_manager/reports/ -!model_manager/reports/header.html - -# Allow temp directory structure (but not its contents) -!model_manager/reports/temp/ - -# ============================================================================ -# ALLOW: Dependencies file (needed for pip install in Dockerfile) -# ============================================================================ -!requirements.txt - -# ============================================================================ -# BLOCK: Explicitly block unwanted files even if they match above patterns -# ============================================================================ - -# Python cache and compiled files -**/__pycache__/ -**/*.pyc -**/*.pyo -**/*.pyd -**/.Python -**/*.so -**/*.egg -**/*.egg-info/ - -# Tests (not needed in production) -model_manager/**/test_*.py -model_manager/**/*_test.py - -# IDE and editor files -**/.vscode/ -**/.idea/ -**/*.swp -**/*.swo -**/*~ - -# OS files -**/.DS_Store -**/Thumbs.db - -# Logs and temporary files -**/*.log -**/*.tmp -**/*.temp - -# Local configuration -**/.env -**/.env.local -**/*.local - -# Documentation inside code -**/*.md -**/README* - -# Backup files -**/*.bak -**/*.backup -**/*.old diff --git a/Dockerfile b/Dockerfile deleted file mode 100644 index 641b3b0..0000000 --- a/Dockerfile +++ /dev/null @@ -1,78 +0,0 @@ -# Multi-stage build for optimized Python application -FROM python:3.11-slim AS builder - -# Set build-time environment variables -ENV PYTHONDONTWRITEBYTECODE=1 \ - PYTHONUNBUFFERED=1 \ - PIP_NO_CACHE_DIR=1 \ - PIP_DISABLE_PIP_VERSION_CHECK=1 - -# Install build dependencies only -RUN apt-get update && apt-get install -y \ - build-essential \ - curl \ - git \ - && rm -rf /var/lib/apt/lists/* && \ - apt-get clean - -# Configure SSH to trust GitHub host key -RUN mkdir -p ~/.ssh && \ - ssh-keyscan -t rsa github.com >> ~/.ssh/known_hosts && \ - chmod 600 ~/.ssh/known_hosts - -# Create virtual environment -RUN python -m venv /opt/venv -ENV PATH="/opt/venv/bin:$PATH" - -# Upgrade pip and wheel for better caching -RUN pip install --upgrade pip setuptools wheel - -# Copy requirements files for better Docker layer caching -COPY requirements.txt ./ - -# Install only production dependencies with no cache -RUN --mount=type=ssh echo "=== Installing dependencies ===" && \ - pip install --no-cache-dir -r requirements.txt && \ - echo "=== Dependencies installed successfully ===" && \ - pip list | wc -l && \ - echo "=== Cleaning cache files ===" && \ - find /opt/venv -type d -name __pycache__ -exec rm -rf {} + 2>/dev/null || true && \ - find /opt/venv -name "*.pyc" -delete 2>/dev/null || true && \ - rm -rf /root/.cache/pip/* && \ - echo "=== Cleaning venv site-packages ===" && \ - find /opt/venv/lib/python3.11/site-packages/ -type f -name "*.md" -delete 2>/dev/null || true && \ - echo "=== Stripping .so files ===" && \ - find /opt/venv -name "*.so" -exec strip {} + 2>/dev/null || true - -# Production stage using python-slim for better functionality -FROM python:3.11-slim AS production - -# Set environment variables -ENV PYTHONDONTWRITEBYTECODE=1 \ - PYTHONUNBUFFERED=1 \ - PATH="/opt/venv/bin:$PATH" \ - POD_ID=unknown \ - HOME="/app" - -# Copy virtual environment from builder stage -COPY --from=builder /opt/venv /opt/venv - -# Set working directory -WORKDIR /app - -# Copy application code -COPY . . - -# Create necessary directories for runtime file creation -RUN mkdir -p /app/model_manager/reports /app/logs /app/temp /app/models /app/data && \ - chmod 755 /app/model_manager/reports /app/logs /app/temp /app/models /app/data - -# Create non-root user -RUN groupadd -r appuser && useradd -r -g appuser appuser && \ - chown -R appuser:appuser /app - -# Switch to non-root user -USER appuser - -# Set entrypoint for proper signal handling and PID 1 -ENTRYPOINT ["/opt/venv/bin/python", "-m", "model_manager.worker.worker"] diff --git a/README.md b/README.md index 60cd104..09ba470 100644 --- a/README.md +++ b/README.md @@ -73,7 +73,7 @@ An enterprise-grade ML model training orchestration platform built on Temporal. - [Usage](#usage) - [Command Reference](#command-reference) - [Docker](#docker) -- [Helm Chart](#helm-chart) +- [Kubernetes Deployment](#kubernetes-deployment) ## Features @@ -1239,8 +1239,6 @@ sientia-dataops-model-manager/ ├── .github/workflows/ # CI/CD workflows │ ├── quality-gate.yml # PR quality checks │ └── deploy.yml # Deployment workflow -├── Dockerfile # Container image definition -├── values.yaml # Helm chart values ├── pyproject.toml # Project configuration ├── requirements.txt # Production dependencies ├── requirements-dev.txt # Development dependencies @@ -1528,79 +1526,17 @@ act -n ## Docker -### Create image +Model Manager runs on the shared **`sientia-module`** image — the application code is cloned from the internal Gitea at runtime, not baked into an image. The image is defined in the [Aignosi/sientia-container-images](https://github.com/Aignosi/sientia-container-images) repository (`templates/sientia-module/`) and published to GCP Artifact Registry (`southamerica-east1-docker.pkg.dev/sientia-dev/sientia/sientia-module`). -```bash -$ docker build --ssh default --no-cache --progress=plain -t aignosi.azurecr.io/sientia-dataops-model-manager:0.0.0 . -``` +The legacy repo-specific Dockerfile was removed — no image is built from this repo. -### Create container +To run locally, use `./run_local.sh` (see [Local Execution](#local-execution)). -```bash -$ docker run --env-file .env --network="host" --name sientia-dataops-model-manager -d aignosi.azurecr.io/sientia-dataops-model-manager:0.0.0 +## Kubernetes Deployment -$ docker logs -f sientia-dataops-model-manager -``` +Kubernetes deployment is managed by the [Aignosi/sientia-helm-chart](https://github.com/Aignosi/sientia-helm-chart) umbrella chart (`src/charts-internal/model-manager/`, which wraps the shared `sientia-module` chart) — the local `values.yaml` and the vendored `sientia-module/` chart copy were removed. Worker configuration, Gitea repo/branch, task queues, and sizing are all defined in the umbrella values. -### Login using access token - -```bash -$ docker login -u -p aignosi.azurecr.io -``` - -### Push image to repository - -```bash -$ docker push aignosi.azurecr.io/sientia-dataops-model-manager:0.0.0 -``` - -## Helm Chart - -### Reference - -https://aignosi-wiki.atlassian.net/wiki/spaces/IT1/pages/274563074/Como+utilizar+o+Helm+Repo+Privado - -### Add Helm Chart repository - -```bash -$ helm repo add sientia \ - https://raw.githubusercontent.com/Aignosi/sientia-dataops-helm-repo/refs/heads/main/ \ - --username $GITHUB_USER \ - --password $GITHUB_PASS - -# Update repository -$ helm repo update - -# List repositories -$ helm repo list - -# List versions of a specific chart -$ helm search repo sientia --versions - -# List all charts available -$ helm search repo sientia - -# List chart details -$ helm show all sientia/sientia-module - -# Download chart to current directory -$ helm pull sientia/sientia-module --version 0.6.0 --untar - -# Remove chart directory -$ rm -rf sientia-module -``` - -### Helm Install - -```shell -$ helm upgrade --install sientia-dataops-model-manager sientia/sientia-module -n sientia --create-namespace -f ./values.yaml --version 0.6.0 -``` - -### Uninstall Helm Chart - -```shell -$ helm uninstall sientia-dataops-model-manager -n sientia -``` +The standalone `sientia-module` chart still lives in [sientia-dataops-helm-repo](https://github.com/Aignosi/sientia-dataops-helm-repo) for the legacy deploy flow (temporarily broken until the CI/CD redesign). --- diff --git a/values.yaml b/values.yaml deleted file mode 100644 index 016fa19..0000000 --- a/values.yaml +++ /dev/null @@ -1,312 +0,0 @@ -# Default values for sientia-module. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. - -# This will set the replicaset count more information can be found here: https://kubernetes.io/docs/concepts/workloads/controllers/replicaset/ -replicaCount: 1 - -# This sets the container image more information can be found here: https://kubernetes.io/docs/concepts/containers/images/ -image: - repository: aignosi.azurecr.io/sientia-dataops-model-manager - # This sets the pull policy for images. - pullPolicy: IfNotPresent - # Overrides the image tag whose default is the chart appVersion. - tag: "1.2.0" - -# This is for the secrets for pulling an image from a private repository more information can be found here: https://kubernetes.io/docs/tasks/configure-pod-container/pull-image-private-registry/ -imagePullSecrets: -- name: docker-hub-secret - -# This is to override the chart name. -nameOverride: "sientia-dataops-model-manager" -fullnameOverride: "sientia-dataops-model-manager" -namespace: sientia - -# This section builds out the service account more information can be found here: https://kubernetes.io/docs/concepts/security/service-accounts/ -serviceAccount: - # Specifies whether a service account should be created - create: true - # Automatically mount a ServiceAccount's API credentials? - automount: true - # Annotations to add to the service account - annotations: {} - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: "sientia-dataops-model-manager" - -# This is for setting Kubernetes Annotations to a Pod. -# For more information checkout: https://kubernetes.io/docs/concepts/overview/working-with-objects/annotations/ -podAnnotations: {} -# This is for setting Kubernetes Labels to a Pod. -# For more information checkout: https://kubernetes.io/docs/concepts/overview/working-with-objects/labels/ -podLabels: {} - -podSecurityContext: {} - # fsGroup: 2000 - -securityContext: {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - -resources: {} - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 128Mi - # requests: - # cpu: 100m - # memory: 128Mi - -# This is to setup the liveness and readiness probes more information can be found here: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-startup-probes/ -livenessProbe: - exec: - command: - - python3 - - -c - - "import requests; requests.get('http://localhost:9090/metrics')" - initialDelaySeconds: 20 - periodSeconds: 30 - -readinessProbe: - exec: - command: - - python3 - - -c - - "import requests; requests.get('http://localhost:9090/metrics')" - initialDelaySeconds: 10 - periodSeconds: 15 - -# This section is for setting up autoscaling more information can be found here: https://kubernetes.io/docs/concepts/workloads/autoscaling/ -autoscaling: - enabled: false - minReplicas: 1 - maxReplicas: 100 - targetCPUUtilizationPercentage: 80 - # targetMemoryUtilizationPercentage: 80 - -# Additional volumes on the output Deployment definition. -volumes: - - name: reports-volume - emptyDir: - sizeLimit: 1Gi - -# Additional volumeMounts on the output Deployment definition. -volumeMounts: - - name: reports-volume - mountPath: "/app/model_manager/reports/temp" - -# Deployment strategy configuration -# More information: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy -deploymentStrategy: - type: Recreate - # rollingUpdate: - # maxSurge: 0 - # maxUnavailable: 1 - -# Number of old ReplicaSets to retain -revisionHistoryLimit: 2 - -nodeSelector: {} - -tolerations: [] - -affinity: {} - -services: - sdk-metrics: - enabled: true - type: ClusterIP - port: 9091 - targetPort: 9091 - name: sdk-metrics - metrics: - enabled: true - type: ClusterIP - port: 9090 - targetPort: 9090 - name: metrics - -# Configuração do ServiceMonitor para o Prometheus Operator -# ref: https://github.com/prometheus-operator/prometheus-operator -serviceMonitor: - # Se true, um recurso ServiceMonitor será criado. - enabled: true - # O intervalo no qual as métricas devem ser coletadas (ex: 30s, 1m). - endpoints: - - port: metrics - path: /metrics - interval: 30s - relabelings: [] - - port: sdk-metrics - path: /metrics - interval: 30s - relabelings: [] - additionalLabels: - release: kube-prometheus-stack - -env: - - name: POSTGRES_HOST - value: "paradedb-rw.paradedb.svc.cluster.local" - - name: POSTGRES_PORT - value: "5432" - - name: POSTGRES_USER - value: "postgres" - - name: POSTGRES_PASSWORD - value: "nFqc81y6kwmr2zuAIx43DhiOosFCVPpeEfTtTWZflkNjB2j1KtEeIANkhFR9mAX3" - - name: POSTGRES_DBNAME - value: "sientia-core-mlops-bff" - - name: POSTGRES_MIN_CONNECTIONS - value: "10" - - name: POSTGRES_MAX_CONNECTIONS - value: "30" - - - name: MLFLOW_URL - value: "http://sientia-tracker-mlflow-tracking.sientia-tracker.svc.cluster.local:80" - - name: MLFLOW_USERNAME - value: "aignosi" - - name: MLFLOW_PASSWORD - value: "1L0FP50j3ncp123" - - - name: LOG_LEVEL - value: "DEBUG" - - name: HTTP_METRICS_PORT - value: "9090" - - name: HTTP_SDK_METRICS_PORT - value: "9091" - - name: PROJECT_NAME - value: "sientia-model-manager" - - - name: TEMPORAL_HOST - value: "temporal-frontend.temporal.svc.cluster.local:7233" - - name: TEMPORAL_NAMESPACE - value: "model-manager" - - name: TRAIN_TASK_QUEUE - value: "train_model-queue" - - name: CLEANUP_TASK_QUEUE - value: "cleanup-queue" - - name: TEMPORAL_USE_TLS - value: "false" - - - name: MONGODB_USERNAME - value: "root" - - name: MONGODB_PASSWORD - value: "wKZDbMNU1c" - - name: MONGODB_URL - value: "my-release-mongodb.mongodb.svc.cluster.local:27017" - - name: MONGODB_DATABASE - value: "sientia" - - name: MONGODB_TTL_INDEX_HOURS - value: "1" - - - name: MINIO_ENDPOINT_URL - value: "http://minio.minio.svc.cluster.local:9000" - - name: MINIO_ACCESS_KEY - value: "model-training-user" - - name: MINIO_SECRET_KEY - value: "modelTrainingUser123" - - name: MINIO_REGION - value: "us-east-1" - - name: MINIO_USE_SSL - value: "false" - - name: MINIO_MAX_RETRY_ATTEMPTS - value: "3" - - name: MINIO_RETRY_MODE - value: "adaptive" - - name: MINIO_CONNECT_TIMEOUT - value: "10" - - name: MINIO_READ_TIMEOUT - value: "60" - - - name: TIMEOUT_VALIDATE_PARAMS - value: "30" - - name: TIMEOUT_TRAIN_MODEL - value: "2700" - - name: TIMEOUT_DELETE_FILE - value: "120" - - name: TIMEOUT_UPDATE_DATABASE - value: "30" - - - name: CLEANUP_RETENTION_HOURS - value: "24" - - name: CLEANUP_DRY_RUN - value: "false" - - name: TIMEOUT_CLEANUP_LOCAL - value: "120" - - # Cleanup Schedule Configuration - - name: CLEANUP_SCHEDULE_ID - value: "cleanup-files-daily" - - name: CLEANUP_CRON - value: "0 0 * * *" # Midnight UTC - - name: CLEANUP_TIMEZONE - value: "UTC" - - name: CLEANUP_EXECUTION_TIMEOUT_HOURS - value: "1" - - - name: EXTRA_PIP_REQUIREMENTS - value: "git+https://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@github.com/Aignosi/sientia-mlops-library.git" - - - name: POD_ID - valueFrom: - fieldRef: - fieldPath: metadata.name - -ssh: - enabled: false - secretName: git-ssh-key-sientia-model-manager-worker - sshPath: /mnt/.ssh - knownHostsPath: /mnt/known_hosts - -# Configuração para dashboards do Grafana -grafanaDashboard: - # Habilita a criação de ConfigMaps para dashboards - enabled: true - # Namespace onde o Grafana está instalado (ajuste conforme seu ambiente) - namespace: monitoring - # Labels para que o sidecar do Grafana encontre os dashboards - labels: - grafana_dashboard: "1" - # Lista de dashboards para importar - dashboards: - - name: sientia-dataops-model-manager - title: "Sientia DataOps Model Manager" - uid: "sientia-dataops-model-manager" - folder: "Sientia" - jsonFile: "dashboards/sientia-dataops-model-manager.json" - overwrite: true # Sobrescreve dashboard se já existir - version: "1.0.0" # Version inicial do dashboard - -# Configuração para datasources do Grafana -grafanaDatasource: - # Habilita a criação de ConfigMap para datasources - enabled: false - # Namespace onde o Grafana está instalado - namespace: monitoring - # Labels para que o sidecar do Grafana encontre os datasources - labels: - grafana_datasource: "1" - # Lista de datasources para configurar - datasources: [] - # Exemplo de datasource: - # - name: Prometheus - # type: prometheus - # url: http://prometheus-server.monitoring.svc.cluster.local - # isDefault: true - # jsonData: - # timeInterval: "5s" - -# kubectl create secret docker-registry docker-hub-secret --namespace sientia --docker-server=http://aignosi.azurecr.io --docker-username=aignosi --docker-password=5I5zpQ6sRaHqX1hD3dr+2mo647yO3FRc359/wu6gsP+ACRDRz5mp - -# helm upgrade --install sientia-dataops-model-manager sientia/sientia-module -n sientia --create-namespace -f ./values.yaml --version 0.6.0 - -# kubectl create secret generic git-ssh-key-sientia-model-manager-worker \ -# --namespace sientia \ -# --from-file=ssh-privatekey=git_key \ -# --type=kubernetes.io/ssh-auth