# svcforge chart values. # # The digests below are the deployment. CI builds each image once, pushes it, reads the # digest back with `docker buildx imagetools inspect`, and `yq -i`s it into this file as # its last act. ArgoCD notices the commit and syncs. Nothing else deploys svcforge. # # Tags are banned. A tag is a mutable pointer, which means "what is running" and "what # this file says" can silently diverge. A digest cannot. nameOverride: "" fullnameOverride: "" image: registry: gitea.oci-oci.duckdns.org pullPolicy: IfNotPresent pullSecrets: [] # One repo + one digest per service: CI's matrix builds three images, so there are three # digests. The zeros are placeholders — a fresh clone must be bumped by CI before it can # deploy, which is the intended failure mode. Never hand-edit these. api: repo: gitea.oci-oci.duckdns.org/gitea_admin/svcforge-api digest: sha256:cba5ba8ed88cbb84f96dcd25d0ff44ea16208cf9315056af73bfdd60b4128f2e worker: repo: gitea.oci-oci.duckdns.org/gitea_admin/svcforge-worker digest: sha256:14f829a20365a21f3057a95622ed262f139a82e96d02986136188361f66618e9 reconciler: repo: gitea.oci-oci.duckdns.org/gitea_admin/svcforge-reconciler digest: sha256:6c73743931d6a343475b9c5933055bfd6710e1fe3e4d336b5f22ee4893773dd3 api: replicas: 2 # One process per pod. Module 7 took the "scale with replicas" fix over # PROMETHEUS_MULTIPROC_DIR, so `uvicorn --workers N` here would corrupt the metrics. resources: requests: {cpu: 50m, memory: 128Mi} limits: {memory: 256Mi} service: type: ClusterIP port: 80 targetPort: 8000 ingress: enabled: true className: nginx annotations: cert-manager.io/cluster-issuer: letsencrypt-prod host: svcforge.oci-oci.duckdns.org tls: enabled: true secretName: svcforge-tls worker: # Plain replicas. No HPA: the day `replicas: 2` stops keeping up, not before. replicas: 2 concurrency: 4 resources: requests: {cpu: 100m, memory: 192Mi} limits: {memory: 512Mi} reconciler: # A singleton, and not by convention — the four checks are not safe to run twice # concurrently. replicas is deliberately not a value: there is nothing to tune. intervalSeconds: 60 resources: requests: {cpu: 50m, memory: 128Mi} limits: {memory: 256Mi} # Postgres pool sizing. replicas × maxSize is spent against the Supabase pooler budget: # api(2 × 5) + worker(2 × 5) + reconciler(1 × 2) = 22 connections. Raise with care. pool: minSize: 1 maxSize: 5 migrate: # backoffLimit: 0 — a failed migration must fail the release, not retry into a # half-applied schema. Migrations run here and only here; never on app startup. enabled: true resources: requests: {cpu: 50m, memory: 128Mi} limits: {memory: 256Mi} auth: jwksUrl: https://auth.oci-oci.duckdns.org/realms/svcforge/protocol/openid-connect/certs issuer: https://auth.oci-oci.duckdns.org/realms/svcforge audience: svcforge otel: enabled: true endpoint: http://alloy.observability.svc.cluster.local:4317 log: level: info # The DSNs are pulled from Vault by external-secrets into a Secret the pods envFrom. # No DSN is ever a chart value, a ConfigMap key, or a CI variable. That invariant holds # either way here — what changes below is only who creates the Secret. # # DISABLED ON THIS CLUSTER, AND THIS IS A DEVIATION, NOT THE DESIGN. # # The block below describes a ClusterSecretStore named `vault` with HashiCorp-style # key/property refs. This cluster has `oci-vault` instead: OCI Vault via InstancePrincipal, # whose provider addresses a secret by NAME and takes a JSON property, so these remoteRefs # do not translate as written. There are also no ExternalSecrets anywhere on the cluster # yet, so nothing has ever exercised this path. # # With this false, `svcforge.secretName` still resolves through targetName, so the # deployments and the migrate hook read a Secret called `svcforge-secrets` that was created # out of band: # # kubectl -n svcforge create secret generic svcforge-secrets \ # --from-env-file=~/.config/svcforge/secrets.env # # ArgoCD does not manage that Secret, so prune and selfHeal cannot touch it — which is also # why it is invisible in git, and the one part of this deployment you cannot read from the # repo. Restoring the intended design means adding oci_vault_secret resources to # oci-k8s/infra/vault.tf and repointing secretStoreRef at oci-vault. externalSecret: enabled: false secretStoreRef: name: vault kind: ClusterSecretStore refreshInterval: 1h # target Secret name; keys land as SVCFORGE_PG_DSN / SVCFORGE_PG_DSN_SESSION / SVCFORGE_REDIS_DSN targetName: svcforge-secrets remoteRefs: - secretKey: SVCFORGE_PG_DSN key: svcforge/postgres property: dsn_pooler - secretKey: SVCFORGE_PG_DSN_SESSION key: svcforge/postgres property: dsn_session - secretKey: SVCFORGE_REDIS_DSN key: svcforge/redis property: dsn rbac: # The worker helm-installs tenant releases into namespaces it creates. `namespaces` is a # cluster-scoped resource, so `create namespaces` cannot be granted by a namespaced Role # — this has to be a ClusterRole. It is still least-privilege: named resources, named # verbs, no `*`, no cluster-admin, and no rbac.authorization.k8s.io group at all, so the # worker cannot grant itself anything further. create: true serviceAccount: create: true annotations: {} # Chart-native only. A hand-authored ServiceMonitor/PrometheusRule CR is banned — the # chart owns these, gated by these flags. serviceMonitor: enabled: true interval: 30s prometheusRule: enabled: true rules: - alert: SvcforgeQueueDepthRising expr: deriv(svcforge_queue_depth[10m]) > 0 for: 10m labels: severity: warning annotations: summary: svcforge queue depth is rising and not draining runbook_url: https://gitea.oci-oci.duckdns.org/gitea_admin/svcforge/src/branch/master/RUNBOOK.md#queue-stuck - alert: SvcforgeProvisionSlow expr: histogram_quantile(0.95, sum by (le) (rate(svcforge_provision_duration_seconds_bucket[30m]))) > 300 for: 15m labels: severity: warning annotations: summary: svcforge p95 provision time is over 5 minutes runbook_url: https://gitea.oci-oci.duckdns.org/gitea_admin/svcforge/src/branch/master/RUNBOOK.md#provision-failing # `increase(...[15m])`, not the raw counter. `sum(counter) > 0` on a monotonic counter # latches: one dead-lettered task at any point keeps this firing until the pod restarts, # and a restart silently clears it — so it can never distinguish "failing now" from # "failed last Tuesday". The dead-letter counter is also the right one: the attempts # counter increments on ordinary transient retries that later succeed. - alert: SvcforgeTaskDeadLettered expr: sum(increase(svcforge_tasks_dead_lettered_total[15m])) > 0 for: 5m labels: severity: warning annotations: summary: a svcforge task exhausted its retries runbook_url: https://gitea.oci-oci.duckdns.org/gitea_admin/svcforge/src/branch/master/RUNBOOK.md#provision-failing # Scoped to the reconciler job, and aggregated with max(). # # RECONCILER_LAST_TICK is a module-level Gauge in obs.py, so EVERY service that imports # svcforge_core.obs registers and exports it — api and worker included, permanently at # 0. Unscoped, `time() - 0` is ~1.7e9, so this critical alert fires from the moment the # chart is installed, from pods that have no reconciler in them. Scope by job, then # max() so a rolling restart of the single reconciler does not flap it. - alert: SvcforgeReconcilerStale expr: time() - max(svcforge_reconciler_last_tick_timestamp_seconds{job=~".*reconciler.*"}) > 300 for: 5m labels: severity: critical annotations: summary: the svcforge reconciler has not ticked in 5 minutes runbook_url: https://gitea.oci-oci.duckdns.org/gitea_admin/svcforge/src/branch/master/RUNBOOK.md#orphaned-release # Not in the spec. Off by default: on a three-node k3s a PDB that cannot be satisfied # blocks drains, which is worse than the disruption it prevents. podDisruptionBudget: enabled: false minAvailable: 1 nodeSelector: {} tolerations: [] affinity: {}