fix: Fixes to running on kubernetes
Chart Release / chart (push) Failing after 13s
Server Deploy / deploy (push) Successful in 6m35s

This commit is contained in:
2026-07-31 10:34:10 +01:00
parent de78688093
commit 165114471f
31 changed files with 1880 additions and 278 deletions
+27
View File
@@ -1,12 +1,39 @@
Vantage has been deployed as release "{{ .Release.Name }}" in namespace "{{ .Release.Namespace }}".
Services created:
{{- if .Values.redis.enabled }}
- {{ .Release.Name }}-redis (ClusterIP {{ .Values.redis.port }})
{{- else }}
- Redis: not deployed, using external {{ .Values.redis.addr }}
{{- end }}
{{- if .Values.mongo.enabled }}
- {{ .Release.Name }}-mongo (ClusterIP {{ .Values.mongo.port }})
{{- else }}
- MongoDB: not deployed, using the external server.env.mongoUri
{{- end }}
- {{ .Release.Name }}-guacd ({{ .Values.guacd.service.type }} {{ .Values.guacd.service.port }})
- {{ .Release.Name }}-server ({{ .Values.server.service.type }} http:{{ .Values.server.service.httpPort }} grpc:{{ .Values.server.service.grpcPort }})
- {{ .Release.Name }}-web ({{ .Values.web.service.type }} {{ .Values.web.service.port }})
Scaling (server.replicaCount / web.replicaCount):
- Both scale. Pin the image tags first — replicas on different builds serve
mismatched web asset hashes, and mixed server versions share one bus.
- server replicas route agent commands, step results and console relays to
each other over Redis, so every replica must use the SAME Redis. Workflow
logs are in MongoDB, not on a volume.
- Background work (monitor scheduler, Free reaper, log and audit retention,
the offline sweep) runs on one replica at a time under a Redis leader lock.
- server.persistence must be off to scale past one replica on a ReadWriteOnce
volume. Nothing writes to it any more.
{{- if gt (int .Values.server.replicaCount) 1 }}
- Console relays are reached by pod IP; guacd must be able to dial pod IPs
directly (it can, inside the cluster network).
{{- end }}
{{- if .Values.server.migrationJob.enabled }}
- Migrations run in the {{ .Release.Name }}-migrate Job before each upgrade;
the pods skip them. Its logs are kept: kubectl logs job/{{ .Release.Name }}-migrate
{{- end }}
By default the server/web/guacd services are ClusterIP only (no host port publishing,
unlike the original docker-compose file). To expose them externally, set
server.service.type / web.service.type / guacd.service.type to NodePort or LoadBalancer,
@@ -9,3 +9,76 @@ Common name helpers
app.kubernetes.io/instance: {{ .Release.Name }}
app.kubernetes.io/managed-by: {{ .Release.Service }}
{{- end -}}
{{/*
vantage.server.env renders the server container's environment.
It lives here because two workloads need it identically: the Deployment and the
pre-upgrade migration Job. A Job that connected to a different database than the
pods it migrates for would be worse than no Job at all, so there is one copy and
both read it.
*/}}
{{- define "vantage.server.env" -}}
- name: MONGO_URI
{{- $mongoUri := tpl .Values.server.env.mongoUri . }}
{{- if and (not .Values.mongo.enabled) (contains (printf "%s-mongo" .Release.Name) $mongoUri) }}
{{- fail "mongo.enabled is false, so server.env.mongoUri must point at an external MongoDB rather than the in-chart one" }}
{{- end }}
value: {{ $mongoUri | quote }}
- name: REDIS_ADDR
{{- if .Values.redis.enabled }}
value: "{{ .Release.Name }}-redis:{{ .Values.redis.port }}"
{{- else }}
{{- if not .Values.redis.addr }}
{{- fail "redis.enabled is false, so redis.addr must be set to an external Redis host:port" }}
{{- end }}
value: {{ .Values.redis.addr | quote }}
{{- end }}
{{- if .Values.redis.auth.existingSecret }}
- name: REDIS_USERNAME
valueFrom:
secretKeyRef:
name: {{ .Values.redis.auth.existingSecret }}
key: {{ .Values.redis.auth.usernameKey }}
optional: true
- name: REDIS_PASSWORD
valueFrom:
secretKeyRef:
name: {{ .Values.redis.auth.existingSecret }}
key: {{ .Values.redis.auth.passwordKey }}
{{- else }}
{{- if .Values.redis.auth.username }}
- name: REDIS_USERNAME
value: {{ .Values.redis.auth.username | quote }}
{{- end }}
{{- if .Values.redis.auth.password }}
- name: REDIS_PASSWORD
value: {{ .Values.redis.auth.password | quote }}
{{- end }}
{{- end }}
- name: GRPC_HOST
value: {{ .Values.server.env.grpcHost | quote }}
- name: GRPC_PORT
value: {{ .Values.server.service.grpcPort | quote }}
- name: HTTP_PORT
value: {{ .Values.server.service.httpPort | quote }}
- name: KEY_ENCRYPTION_KEY
value: {{ .Values.server.env.keyEncryptionKey | quote }}
- name: GUACD_ADDR
value: "{{ .Release.Name }}-guacd:{{ .Values.guacd.service.port }}"
- name: APP_ROOT_LABEL
value: {{ .Values.server.env.appRootLabel | quote }}
- name: PROXY_ADVERTISE_HOST
value: {{ .Values.server.env.proxyAdvertiseHost | quote }}
- name: PROXY_LISTEN_HOST
value: {{ .Values.server.env.proxyListenHost | quote }}
# The address guacd dials to reach a console relay. It must name one pod, not
# the Service: the relay listener is bound by whichever pod holds that agent's
# command stream, and a Service would send guacd to a different one. POD_IP
# takes precedence over PROXY_ADVERTISE_HOST in the server for exactly this
# reason, so the setting above stays meaningful only outside Kubernetes.
- name: POD_IP
valueFrom:
fieldRef:
fieldPath: status.podIP
{{- end -}}
@@ -0,0 +1,68 @@
{{- if .Values.server.migrationJob.enabled }}
{{/*
Schema setup, lifted out of the serving pods.
Every server process used to run migrations, index builders and default-step
seeding at boot. With one replica that is fine. With two it is not: 0004 renames
the orgs collection to instances, and a sibling reading it mid-rename is a
corruption, not a retry.
A Helm hook Job runs it once, before any pod of the new version starts. The
Deployment then sets VANTAGE_SKIP_MIGRATIONS, which is what makes the Job's
existence load-bearing rather than decorative — if you disable the Job, the
pods go back to migrating themselves and you must go back to one replica.
hook-weight orders this after the dependency waits; before-hook-creation deletes
the previous Job so a repeat upgrade is not blocked by an immutable object. The
Job is deliberately NOT deleted on success: its logs are the record of what the
upgrade did to the database.
*/}}
apiVersion: batch/v1
kind: Job
metadata:
name: {{ .Release.Name }}-migrate
labels:
{{- include "vantage.labels" . | nindent 4 }}
app.kubernetes.io/component: migrate
annotations:
"helm.sh/hook": pre-install,pre-upgrade
"helm.sh/hook-weight": "0"
"helm.sh/hook-delete-policy": before-hook-creation
spec:
backoffLimit: {{ .Values.server.migrationJob.backoffLimit }}
# A migration that has not finished in this long is stuck, and a stuck
# migration should fail the upgrade rather than hold it open forever.
activeDeadlineSeconds: {{ .Values.server.migrationJob.activeDeadlineSeconds }}
template:
metadata:
labels:
app.kubernetes.io/instance: {{ .Release.Name }}
app.kubernetes.io/component: migrate
spec:
restartPolicy: Never
{{- if .Values.imagePullSecrets }}
imagePullSecrets:
{{- toYaml .Values.imagePullSecrets | nindent 8 }}
{{- end }}
{{- if .Values.mongo.enabled }}
# Only Mongo. The Job never opens Redis, and waiting on a Redis this
# chart may not even deploy would block an upgrade for no reason.
initContainers:
- name: wait-for-mongo
image: busybox:1.36
command:
- sh
- -c
- |
until nc -z {{ .Release.Name }}-mongo {{ .Values.mongo.port }}; do
echo "waiting for mongo..."; sleep 2;
done
{{- end }}
containers:
- name: migrate
image: "{{ .Values.server.image.repository }}:{{ .Values.server.image.tag }}"
env:
{{- include "vantage.server.env" . | nindent 12 }}
- name: VANTAGE_MIGRATE_ONLY
value: "true"
{{- end }}
@@ -1,3 +1,4 @@
{{- if .Values.mongo.enabled }}
{{- if .Values.mongo.persistence.enabled }}
apiVersion: v1
kind: PersistentVolumeClaim
@@ -88,3 +89,4 @@ spec:
ports:
- port: {{ .Values.mongo.port }}
targetPort: {{ .Values.mongo.port }}
{{- end }}
@@ -1,3 +1,4 @@
{{- if .Values.redis.enabled }}
{{- if .Values.redis.persistence.enabled }}
apiVersion: v1
kind: PersistentVolumeClaim
@@ -88,3 +89,4 @@ spec:
ports:
- port: {{ .Values.redis.port }}
targetPort: {{ .Values.redis.port }}
{{- end }}
+79 -28
View File
@@ -25,7 +25,19 @@ metadata:
{{- include "vantage.labels" . | nindent 4 }}
app.kubernetes.io/component: server
spec:
replicas: 1
{{- $replicas := int .Values.server.replicaCount }}
replicas: {{ $replicas }}
{{- if and .Values.server.persistence.enabled (eq .Values.server.persistence.accessMode "ReadWriteOnce") }}
# A ReadWriteOnce volume cannot be mounted by a second pod at all, and cannot
# be handed to a new pod while the old one still holds it. Persistence is off
# by default now that nothing writes to it; if it is on, replicas are capped
# at one and updates go through Recreate.
{{- if gt $replicas 1 }}
{{- fail "server.persistence.enabled with a ReadWriteOnce volume cannot be combined with server.replicaCount > 1. Nothing in the server writes to that volume any more (workflow logs live in MongoDB); set server.persistence.enabled=false, or use a ReadWriteMany accessMode if you are keeping it for another reason." }}
{{- end }}
strategy:
type: Recreate
{{- end }}
selector:
matchLabels:
app.kubernetes.io/instance: {{ .Release.Name }}
@@ -40,9 +52,13 @@ spec:
imagePullSecrets:
{{- toYaml .Values.imagePullSecrets | nindent 8 }}
{{- end }}
# Wait for redis & mongo to be reachable, approximating compose's
# `depends_on: condition: service_healthy`
{{- if or .Values.redis.enabled .Values.mongo.enabled }}
# Wait for the dependencies this chart deploys to be reachable,
# approximating compose's `depends_on: condition: service_healthy`. An
# external Redis or Mongo is assumed to be up already — waiting on one
# would only turn someone else's outage into a stuck pod.
initContainers:
{{- if .Values.redis.enabled }}
- name: wait-for-redis
image: busybox:1.36
command:
@@ -52,6 +68,8 @@ spec:
until nc -z {{ .Release.Name }}-redis {{ .Values.redis.port }}; do
echo "waiting for redis..."; sleep 2;
done
{{- end }}
{{- if .Values.mongo.enabled }}
- name: wait-for-mongo
image: busybox:1.36
command:
@@ -61,6 +79,8 @@ spec:
until nc -z {{ .Release.Name }}-mongo {{ .Values.mongo.port }}; do
echo "waiting for mongo..."; sleep 2;
done
{{- end }}
{{- end }}
containers:
- name: server
image: "{{ .Values.server.image.repository }}:{{ .Values.server.image.tag }}"
@@ -68,40 +88,71 @@ spec:
- containerPort: {{ .Values.server.service.httpPort }}
- containerPort: {{ .Values.server.service.grpcPort }}
env:
- name: MONGO_URI
value: {{ tpl .Values.server.env.mongoUri . | quote }}
- name: REDIS_ADDR
value: "{{ .Release.Name }}-redis:{{ .Values.redis.port }}"
- name: GRPC_HOST
value: {{ .Values.server.env.grpcHost | quote }}
- name: GRPC_PORT
value: {{ .Values.server.service.grpcPort | quote }}
- name: HTTP_PORT
value: {{ .Values.server.service.httpPort | quote }}
- name: KEY_ENCRYPTION_KEY
value: {{ .Values.server.env.keyEncryptionKey | quote }}
- name: VANTAGE_WORKFLOW_LOG_DIR
value: {{ .Values.server.env.vantageWorkflowLogDir | quote }}
- name: GUACD_ADDR
value: "{{ .Release.Name }}-guacd:{{ .Values.guacd.service.port }}"
- name: APP_ROOT_LABEL
value: {{ .Values.server.env.appRootLabel | quote }}
- name: PROXY_ADVERTISE_HOST
value: {{ .Values.server.env.proxyAdvertiseHost | quote }}
- name: PROXY_LISTEN_HOST
value: {{ .Values.server.env.proxyListenHost | quote }}
{{- include "vantage.server.env" . | nindent 12 }}
{{- if .Values.server.migrationJob.enabled }}
# Schema setup ran in the pre-upgrade Job. Pods that repeated it
# would race each other, and the rename migration is not a race
# that tolerates a loser.
- name: VANTAGE_SKIP_MIGRATIONS
value: "true"
{{- end }}
# Liveness never touches Mongo or Redis: restarting every pod cannot
# fix a database outage, and each restart drops every agent command
# stream and console session it was carrying. Readiness does check
# both, so a pod that cannot serve leaves the Service and stays up.
startupProbe:
httpGet:
path: /healthz
port: {{ .Values.server.service.httpPort }}
periodSeconds: 5
# Generous: without the migration Job this pod runs every migration
# before it listens, and the rename has a ten-minute budget.
failureThreshold: 150
livenessProbe:
httpGet:
path: /healthz
port: {{ .Values.server.service.httpPort }}
periodSeconds: 20
failureThreshold: 3
readinessProbe:
httpGet:
path: /readyz
port: {{ .Values.server.service.httpPort }}
periodSeconds: 10
failureThreshold: 3
{{- if .Values.server.persistence.enabled }}
# Nothing in the server writes here any more — workflow logs moved to
# MongoDB so that every replica can read and write them. The mount
# remains only so an operator upgrading from a file-log release can
# still reach the old files before turning persistence off.
volumeMounts:
- name: server-data
mountPath: /data
volumes:
- name: server-data
{{- if .Values.server.persistence.enabled }}
persistentVolumeClaim:
claimName: {{ .Release.Name }}-server-data
{{- else }}
emptyDir: {}
{{- end }}
---
{{- if gt (int .Values.server.replicaCount) 1 }}
apiVersion: policy/v1
kind: PodDisruptionBudget
metadata:
name: {{ .Release.Name }}-server
labels:
{{- include "vantage.labels" . | nindent 4 }}
app.kubernetes.io/component: server
spec:
# Agents reconnect on their own, but a drain that took every replica at once
# would disconnect every agent in the fleet simultaneously and stall every
# workflow run in flight.
minAvailable: 1
selector:
matchLabels:
app.kubernetes.io/instance: {{ .Release.Name }}
app.kubernetes.io/component: server
---
{{- end }}
apiVersion: v1
kind: Service
metadata:
+24 -1
View File
@@ -6,7 +6,9 @@ metadata:
{{- include "vantage.labels" . | nindent 4 }}
app.kubernetes.io/component: web
spec:
replicas: 1
# web holds no per-process state: sessions live in Redis and every request is
# proxied to the server. It is the one component here that scales freely.
replicas: {{ .Values.web.replicaCount }}
selector:
matchLabels:
app.kubernetes.io/instance: {{ .Release.Name }}
@@ -39,6 +41,27 @@ spec:
env:
- name: API_URL
value: {{ tpl .Values.web.env.apiUrl . | quote }}
# /healthz is served by this Next process; /api is rewritten to the
# server, so a probe there would report the backend's health and keep
# passing while this pod was wedged.
startupProbe:
httpGet:
path: /healthz
port: {{ .Values.web.service.port }}
periodSeconds: 3
failureThreshold: 20
livenessProbe:
httpGet:
path: /healthz
port: {{ .Values.web.service.port }}
periodSeconds: 20
failureThreshold: 3
readinessProbe:
httpGet:
path: /healthz
port: {{ .Values.web.service.port }}
periodSeconds: 10
failureThreshold: 3
---
apiVersion: v1
kind: Service
+47 -2
View File
@@ -1,6 +1,10 @@
# Default values for the vantage chart.
redis:
# false deploys no Redis and points the server at `redis.addr` instead.
enabled: true
# Only read when enabled is false. host:port of an external Redis.
addr: ""
image:
repository: redis
tag: "8"
@@ -10,8 +14,22 @@ redis:
storageClass: ""
accessMode: ReadWriteOnce
port: 6379
# Both empty for an unauthenticated Redis. Redis 6+ ACL auth takes both; a
# legacy `requirepass` instance takes the password alone and must leave the
# username empty. Set existingSecret to keep the password out of values.
auth:
username: ""
password: ""
# Secret holding the credentials. When set, username/password above are
# ignored and these keys are read from the secret instead.
existingSecret: ""
usernameKey: username
passwordKey: password
mongo:
# false deploys no MongoDB. server.env.mongoUri must then point at an
# external one — the chart cannot guess it, and refuses to render without it.
enabled: true
image:
repository: mongo
tag: "7"
@@ -31,6 +49,23 @@ guacd:
port: 4822
server:
# Safe to raise. Agent commands, step results and console relays are routed
# between replicas over Redis, workflow logs live in MongoDB, and the
# background jobs (monitor scheduler, reaper, retention sweeps) run under a
# Redis leader lock so exactly one replica performs them.
#
# Two requirements come with raising it: server.persistence.enabled must be
# false (or the volume ReadWriteMany), and Redis must be shared by every
# replica — the bus is not optional and a per-pod Redis would partition it.
replicaCount: 1
# Runs migrations, index builders and default-step seeding once, as a Helm
# pre-install/pre-upgrade hook, instead of in every starting pod. Leave it
# on for Kubernetes. Turning it off puts schema setup back in the pods.
migrationJob:
enabled: true
backoffLimit: 0
# 15 minutes: the instance rename alone carries a 10-minute budget.
activeDeadlineSeconds: 900
image:
repository: gitea.hostxtra.co.uk/mrhid6/vantage/server
tag: latest
@@ -42,18 +77,28 @@ server:
mongoUri: "mongodb://{{ .Release.Name }}-mongo:27017/vantage"
grpcHost: "{{ .Release.Name }}-server:9090"
keyEncryptionKey: ""
vantageWorkflowLogDir: ""
appRootLabel: vantage
# Ignored under Kubernetes: the chart sets POD_IP from the downward API
# and the server prefers it, because a console relay listener belongs to
# one pod and a Service address cannot name one.
proxyAdvertiseHost: "{{ .Release.Name }}-server"
proxyListenHost: "0.0.0.0"
# Off by default: nothing in the server writes to disk any more. Workflow
# logs, the only thing that ever did, are in MongoDB so that every replica
# can read and write them. Turn this on only to reach files left behind by
# a release that predates that move — and note a ReadWriteOnce volume caps
# replicaCount at 1 while it is on.
persistence:
enabled: true
enabled: false
size: 1Gi
storageClass: ""
accessMode: ReadWriteOnce
hostPath: /data
web:
# Stateless — safe to raise. Pin web.image.tag when you do: replicas on
# different builds serve mismatched chunk hashes and the UI 404s mid-session.
replicaCount: 1
image:
repository: gitea.hostxtra.co.uk/mrhid6/vantage/web
tag: latest
-1
View File
@@ -18,4 +18,3 @@ KEY_ENCRYPTION_KEY=
MONGO_URI=mongodb://mongo:27017/vantage
# Where workflow run logs are written inside the server container.
# VANTAGE_WORKFLOW_LOG_DIR=/data/workflow-logs
-1
View File
@@ -45,7 +45,6 @@ services:
GRPC_PORT: "9090"
HTTP_PORT: "8080"
KEY_ENCRYPTION_KEY: ${KEY_ENCRYPTION_KEY:-}
VANTAGE_WORKFLOW_LOG_DIR: ${VANTAGE_WORKFLOW_LOG_DIR:-}
GUACD_ADDR: guacd:4822
PROXY_ADVERTISE_HOST: server
depends_on: