Files
svcforge/deploy/chart/values.yaml
T
svcforge-ci 173acc8612 ci: bump image digests to 60ee0f1cbf
Built and scanned by 60ee0f1cbf. ArgoCD syncs from this commit.

[skip ci]
2026-07-22 03:15:59 +00:00

194 lines
8.2 KiB
YAML
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
# svcforge chart values.
#
# The digests below are the deployment. CI builds each image once, pushes it, reads the
# digest back with `docker buildx imagetools inspect`, and `yq -i`s it into this file as
# its last act. ArgoCD notices the commit and syncs. Nothing else deploys svcforge.
#
# Tags are banned. A tag is a mutable pointer, which means "what is running" and "what
# this file says" can silently diverge. A digest cannot.
nameOverride: ""
fullnameOverride: ""
image:
registry: gitea.oci-oci.duckdns.org
pullPolicy: IfNotPresent
pullSecrets: []
# One repo + one digest per service: CI's matrix builds three images, so there are three
# digests. The zeros are placeholders — a fresh clone must be bumped by CI before it can
# deploy, which is the intended failure mode. Never hand-edit these.
api:
repo: gitea.oci-oci.duckdns.org/gitea_admin/svcforge-api
digest: sha256:8c75fd5ba67a6966fccc6403ece5c608221f3dddba156fc2386c05d99478c074
worker:
repo: gitea.oci-oci.duckdns.org/gitea_admin/svcforge-worker
digest: sha256:5d242e5c8ac31837b42ce77d12c1c7d1e86f2bd41686f9ff69fa544a60173d2c
reconciler:
repo: gitea.oci-oci.duckdns.org/gitea_admin/svcforge-reconciler
digest: sha256:de738deea35c9c5aadd09abc6567ea7a1de09df0f2803117525114097ee81edb
api:
replicas: 2
# One process per pod. Module 7 took the "scale with replicas" fix over
# PROMETHEUS_MULTIPROC_DIR, so `uvicorn --workers N` here would corrupt the metrics.
resources:
requests: {cpu: 50m, memory: 128Mi}
limits: {memory: 256Mi}
service:
type: ClusterIP
port: 80
targetPort: 8000
ingress:
enabled: true
className: nginx
annotations:
cert-manager.io/cluster-issuer: letsencrypt-prod
host: svcforge.oci-oci.duckdns.org
tls:
enabled: true
secretName: svcforge-tls
worker:
# Plain replicas. No HPA: the day `replicas: 2` stops keeping up, not before.
replicas: 2
concurrency: 4
resources:
requests: {cpu: 100m, memory: 192Mi}
limits: {memory: 512Mi}
reconciler:
# A singleton, and not by convention — the four checks are not safe to run twice
# concurrently. replicas is deliberately not a value: there is nothing to tune.
intervalSeconds: 60
resources:
requests: {cpu: 50m, memory: 128Mi}
limits: {memory: 256Mi}
# Postgres pool sizing. replicas × maxSize is spent against the Supabase pooler budget:
# api(2 × 5) + worker(2 × 5) + reconciler(1 × 2) = 22 connections. Raise with care.
pool:
minSize: 1
maxSize: 5
migrate:
# backoffLimit: 0 — a failed migration must fail the release, not retry into a
# half-applied schema. Migrations run here and only here; never on app startup.
enabled: true
resources:
requests: {cpu: 50m, memory: 128Mi}
limits: {memory: 256Mi}
auth:
jwksUrl: https://auth.oci-oci.duckdns.org/realms/svcforge/protocol/openid-connect/certs
issuer: https://auth.oci-oci.duckdns.org/realms/svcforge
audience: svcforge
otel:
enabled: true
endpoint: http://alloy.observability.svc.cluster.local:4317
log:
level: info
# The DSNs are pulled from Vault by external-secrets into a Secret the pods envFrom.
# No DSN is ever a chart value, a ConfigMap key, or a CI variable. That invariant holds
# either way here — what changes below is only who creates the Secret.
#
# DISABLED ON THIS CLUSTER, AND THIS IS A DEVIATION, NOT THE DESIGN.
#
# The block below describes a ClusterSecretStore named `vault` with HashiCorp-style
# key/property refs. This cluster has `oci-vault` instead: OCI Vault via InstancePrincipal,
# whose provider addresses a secret by NAME and takes a JSON property, so these remoteRefs
# do not translate as written. There are also no ExternalSecrets anywhere on the cluster
# yet, so nothing has ever exercised this path.
#
# With this false, `svcforge.secretName` still resolves through targetName, so the
# deployments and the migrate hook read a Secret called `svcforge-secrets` that was created
# out of band:
#
# kubectl -n svcforge create secret generic svcforge-secrets \
# --from-env-file=~/.config/svcforge/secrets.env
#
# ArgoCD does not manage that Secret, so prune and selfHeal cannot touch it — which is also
# why it is invisible in git, and the one part of this deployment you cannot read from the
# repo. Restoring the intended design means adding oci_vault_secret resources to
# oci-k8s/infra/vault.tf and repointing secretStoreRef at oci-vault.
externalSecret:
enabled: false
secretStoreRef:
name: vault
kind: ClusterSecretStore
refreshInterval: 1h
# target Secret name; keys land as SVCFORGE_PG_DSN / SVCFORGE_PG_DSN_SESSION / SVCFORGE_REDIS_DSN
targetName: svcforge-secrets
remoteRefs:
- secretKey: SVCFORGE_PG_DSN
key: svcforge/postgres
property: dsn_pooler
- secretKey: SVCFORGE_PG_DSN_SESSION
key: svcforge/postgres
property: dsn_session
- secretKey: SVCFORGE_REDIS_DSN
key: svcforge/redis
property: dsn
rbac:
# The worker helm-installs tenant releases into namespaces it creates. `namespaces` is a
# cluster-scoped resource, so `create namespaces` cannot be granted by a namespaced Role
# — this has to be a ClusterRole. It is still least-privilege: named resources, named
# verbs, no `*`, no cluster-admin, and no rbac.authorization.k8s.io group at all, so the
# worker cannot grant itself anything further.
create: true
serviceAccount:
create: true
annotations: {}
# Chart-native only. A hand-authored ServiceMonitor/PrometheusRule CR is banned — the
# chart owns these, gated by these flags.
serviceMonitor:
enabled: true
interval: 30s
prometheusRule:
enabled: true
rules:
- alert: SvcforgeQueueDepthRising
expr: deriv(svcforge_queue_depth[10m]) > 0
for: 10m
labels:
severity: warning
annotations:
summary: svcforge queue depth is rising and not draining
runbook_url: https://gitea.oci-oci.duckdns.org/gitea_admin/svcforge/src/branch/master/RUNBOOK.md#queue-stuck
- alert: SvcforgeProvisionSlow
expr: histogram_quantile(0.95, sum by (le) (rate(svcforge_provision_duration_seconds_bucket[30m]))) > 300
for: 15m
labels:
severity: warning
annotations:
summary: svcforge p95 provision time is over 5 minutes
runbook_url: https://gitea.oci-oci.duckdns.org/gitea_admin/svcforge/src/branch/master/RUNBOOK.md#provision-failing
# `increase(...[15m])`, not the raw counter. `sum(counter) > 0` on a monotonic counter
# latches: one dead-lettered task at any point keeps this firing until the pod restarts,
# and a restart silently clears it — so it can never distinguish "failing now" from
# "failed last Tuesday". The dead-letter counter is also the right one: the attempts
# counter increments on ordinary transient retries that later succeed.
- alert: SvcforgeTaskDeadLettered
expr: sum(increase(svcforge_tasks_dead_lettered_total[15m])) > 0
for: 5m
labels:
severity: warning
annotations:
summary: a svcforge task exhausted its retries
runbook_url: https://gitea.oci-oci.duckdns.org/gitea_admin/svcforge/src/branch/master/RUNBOOK.md#provision-failing
# Scoped to the reconciler job, and aggregated with max().
#
# RECONCILER_LAST_TICK is a module-level Gauge in obs.py, so EVERY service that imports
# svcforge_core.obs registers and exports it — api and worker included, permanently at
# 0. Unscoped, `time() - 0` is ~1.7e9, so this critical alert fires from the moment the
# chart is installed, from pods that have no reconciler in them. Scope by job, then
# max() so a rolling restart of the single reconciler does not flap it.
- alert: SvcforgeReconcilerStale
expr: time() - max(svcforge_reconciler_last_tick_timestamp_seconds{job=~".*reconciler.*"}) > 300
for: 5m
labels:
severity: critical
annotations:
summary: the svcforge reconciler has not ticked in 5 minutes
runbook_url: https://gitea.oci-oci.duckdns.org/gitea_admin/svcforge/src/branch/master/RUNBOOK.md#orphaned-release
# Not in the spec. Off by default: on a three-node k3s a PDB that cannot be satisfied
# blocks drains, which is worse than the disruption it prevents.
podDisruptionBudget:
enabled: false
minAvailable: 1
nodeSelector: {}
tolerations: []
affinity: {}