From 293e9191bc2baa903d635bb8f4d3a5bb311e908c Mon Sep 17 00:00:00 2001 From: DJ Mountney Date: Wed, 9 Sep 2026 11:21:13 -0700 Subject: [PATCH] fix: stop the default probes and pm2 log storage from restarting pods Three defaults made a busy install restart itself, all found while debugging a self-hosted customer whose pods entered CrashLoopBackOff under CI load. No probe set timeoutSeconds, so every one ran at the Kubernetes default of 1 second. A service that is merely busy answers more slowly than that, and the writer, scheduler and webhooks probes fork a Node CLI that cannot finish in a second at all while the container is under CPU pressure -- the timed-out probe processes then pile up and make the pressure worse. There was also no startup probe anywhere, so the liveness delay was the only thing covering a slow start and a slow start became a restart loop. Give every probe an explicit timeout, period and failure threshold, and add a startup probe so liveness is held off until the container is up. PM2_HOME was a memory-backed emptyDir with no size limit. pm2 writes its log files there for services started from an ecosystem file, so on the writer those logs were charged to the pod's memory limit and the container was OOMKilled under load. It is now disk-backed and capped at 256Mi, which pm2's sockets and logs should never approach. PM2_INSTANCES was unset, so each writer pod ran a single Node process and could not use a second core no matter how the pod was sized. Default it to 2, which is what the hosted service runs. The bundled Redis had no resource requests, making it BestEffort and the first thing the kubelet evicts under node memory pressure -- which drops every service's queue connection at once. Give it requests but no limit, so a queue backlog cannot turn into an OOMKill instead. Also bumps the chart to 0.7.5 and the image to 2026-07-26-004, and corrects the Server section's closing marker in values.yaml. Co-Authored-By: Claude Opus 5 (1M context) --- charts/currents/Chart.yaml | 4 +- .../templates/changestreams/deployment.yaml | 4 + .../templates/director/deployment.yaml | 4 + .../templates/scheduler/deployment.yaml | 6 +- .../currents/templates/server/deployment.yaml | 6 +- .../templates/webhooks/deployment.yaml | 6 +- .../currents/templates/writer/deployment.yaml | 8 +- charts/currents/values.yaml | 143 +++++++++++++++++- docs/configuration.md | 35 +++-- 9 files changed, 191 insertions(+), 25 deletions(-) diff --git a/charts/currents/Chart.yaml b/charts/currents/Chart.yaml index d71f4b3..6a362bf 100644 --- a/charts/currents/Chart.yaml +++ b/charts/currents/Chart.yaml @@ -5,10 +5,10 @@ home: https://currents.dev type: application # The chart version. # Versions are expected to follow Semantic Versioning (https://semver.org/) -version: 0.7.4 +version: 0.7.5 # Version number of the application being deployed. # Versions are not expected to follow Semantic Versioning. They should reflect the version the application is using. -appVersion: "2026-07-26-003" +appVersion: "2026-07-26-004" maintainers: - name: Currents-dev url: https://currents.dev diff --git a/charts/currents/templates/changestreams/deployment.yaml b/charts/currents/templates/changestreams/deployment.yaml index b066dee..dea91dc 100644 --- a/charts/currents/templates/changestreams/deployment.yaml +++ b/charts/currents/templates/changestreams/deployment.yaml @@ -51,6 +51,10 @@ spec: {{- with (concat .Values.global.env .Values.changestreams.env) }} {{- toYaml . | nindent 12 }} {{- end }} + {{- with .Values.changestreams.startupProbe }} + startupProbe: + {{- toYaml . | nindent 12 }} + {{- end }} {{- with .Values.changestreams.livenessProbe }} livenessProbe: {{- toYaml . | nindent 12 }} diff --git a/charts/currents/templates/director/deployment.yaml b/charts/currents/templates/director/deployment.yaml index 34313e0..a3834c0 100644 --- a/charts/currents/templates/director/deployment.yaml +++ b/charts/currents/templates/director/deployment.yaml @@ -72,6 +72,10 @@ spec: - name: http containerPort: {{ .Values.director.service.port }} protocol: TCP + {{- with .Values.director.startupProbe }} + startupProbe: + {{- toYaml . | nindent 12 }} + {{- end }} {{- with .Values.director.livenessProbe }} livenessProbe: {{- toYaml . | nindent 12 }} diff --git a/charts/currents/templates/scheduler/deployment.yaml b/charts/currents/templates/scheduler/deployment.yaml index e82906f..4d19ffb 100644 --- a/charts/currents/templates/scheduler/deployment.yaml +++ b/charts/currents/templates/scheduler/deployment.yaml @@ -56,6 +56,10 @@ spec: mountPath: /home/node/.pm2 - name: startup mountPath: /app/packages/scheduler/dist/.startup + {{- with .Values.scheduler.startupProbe }} + startupProbe: + {{- toYaml . | nindent 12 }} + {{- end }} {{- with .Values.scheduler.livenessProbe }} livenessProbe: {{- toYaml . | nindent 12 }} @@ -77,7 +81,7 @@ spec: volumes: - name: pm2-data emptyDir: - medium: "Memory" + sizeLimit: {{ .Values.scheduler.pm2HomeSizeLimit }} - name: startup persistentVolumeClaim: claimName: {{ include "currents.scheduler.fullname" . }}-startup-tmp diff --git a/charts/currents/templates/server/deployment.yaml b/charts/currents/templates/server/deployment.yaml index aea701b..94bd8b3 100644 --- a/charts/currents/templates/server/deployment.yaml +++ b/charts/currents/templates/server/deployment.yaml @@ -100,6 +100,10 @@ spec: mountPath: /etc/currents/sso readOnly: true {{- end }} + {{- with .Values.server.startupProbe }} + startupProbe: + {{- toYaml . | nindent 12 }} + {{- end }} {{- with .Values.server.livenessProbe }} livenessProbe: {{- toYaml . | nindent 12 }} @@ -121,7 +125,7 @@ spec: volumes: - name: pm2-data emptyDir: - medium: "Memory" + sizeLimit: {{ .Values.server.pm2HomeSizeLimit }} {{- if .Values.currents.sso.saml.enabled }} - name: sso-saml secret: diff --git a/charts/currents/templates/webhooks/deployment.yaml b/charts/currents/templates/webhooks/deployment.yaml index 222775d..e720315 100644 --- a/charts/currents/templates/webhooks/deployment.yaml +++ b/charts/currents/templates/webhooks/deployment.yaml @@ -53,6 +53,10 @@ spec: volumeMounts: - name: pm2-data mountPath: /home/node/.pm2 + {{- with .Values.webhooks.startupProbe }} + startupProbe: + {{- toYaml . | nindent 12 }} + {{- end }} {{- with .Values.webhooks.livenessProbe }} livenessProbe: {{- toYaml . | nindent 12 }} @@ -74,7 +78,7 @@ spec: volumes: - name: pm2-data emptyDir: - medium: "Memory" + sizeLimit: {{ .Values.webhooks.pm2HomeSizeLimit }} {{- with .Values.global.volumes }} {{- toYaml . | nindent 8 }} {{- end }} diff --git a/charts/currents/templates/writer/deployment.yaml b/charts/currents/templates/writer/deployment.yaml index 45822c3..e7104ca 100644 --- a/charts/currents/templates/writer/deployment.yaml +++ b/charts/currents/templates/writer/deployment.yaml @@ -45,6 +45,8 @@ spec: env: - name: CURRENTS_ENV value: "onprem" + - name: PM2_INSTANCES + value: {{ .Values.writer.pm2Instances | quote }} {{- include "currents.connectionConfigEnv" . | nindent 12 }} {{- include "currents.URLConfigEnv" . | nindent 12 }} {{- include "currents.emailEnv" . | nindent 12 }} @@ -54,6 +56,10 @@ spec: volumeMounts: - name: pm2-data mountPath: /home/node/.pm2 + {{- with .Values.writer.startupProbe }} + startupProbe: + {{- toYaml . | nindent 12 }} + {{- end }} {{- with .Values.writer.livenessProbe }} livenessProbe: {{- toYaml . | nindent 12 }} @@ -75,7 +81,7 @@ spec: volumes: - name: pm2-data emptyDir: - medium: "Memory" + sizeLimit: {{ .Values.writer.pm2HomeSizeLimit }} {{- with .Values.global.volumes }} {{- toYaml . | nindent 8 }} {{- end }} diff --git a/charts/currents/values.yaml b/charts/currents/values.yaml index 9d0dcc2..6038175 100644 --- a/charts/currents/values.yaml +++ b/charts/currents/values.yaml @@ -27,7 +27,7 @@ currents: # @section -- Frequently Used key: password # -- The image tag to use for the Currents images - imageTag: 2026-07-26-003 + imageTag: 2026-07-26-004 email: # -- Which transport to send outgoing email through: `smtp` or `ses`. # With `ses` the SMTP settings are ignored and no SMTP credentials are @@ -290,16 +290,34 @@ director: volumes: [] # -- Additional volumeMounts on the output Deployment definition. volumeMounts: [] - # -- Liveness probe to check if the container is alive + # -- Startup probe. While it is running the liveness and readiness probes are + # held off, so a slow start is not mistaken for an unhealthy container. + startupProbe: + httpGet: + path: / + port: http + periodSeconds: 10 + timeoutSeconds: 5 + failureThreshold: 30 + # -- Liveness probe to check if the container is alive. `timeoutSeconds` is + # set explicitly: Kubernetes defaults it to 1 second, and a service that is + # merely busy answers more slowly than that under load, which turns a slow + # pod into a restarting one. livenessProbe: httpGet: path: / port: http + periodSeconds: 20 + timeoutSeconds: 10 + failureThreshold: 6 # -- Readiness probe to check if the container is ready readinessProbe: httpGet: path: / port: http + periodSeconds: 10 + timeoutSeconds: 5 + failureThreshold: 3 # -- Resources to provide # [Resource Management for Pods and Containers](https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/) # @section -- Frequently Used @@ -384,16 +402,39 @@ server: volumes: [] # -- Additional volumeMounts on the output Deployment definition volumeMounts: [] - # -- Liveness probe to check if the container is alive + # -- Size of the emptyDir backing PM2_HOME (`/home/node/.pm2`). It holds pm2's + # sockets and, for services started from an ecosystem file, its log files. + # Disk-backed on purpose: on a memory-backed volume those log files count + # against the pod's memory limit and the container is OOMKilled under load. + pm2HomeSizeLimit: 256Mi + # -- Startup probe. While it is running the liveness and readiness probes are + # held off, so a slow start is not mistaken for an unhealthy container. + startupProbe: + httpGet: + path: / + port: http + periodSeconds: 10 + timeoutSeconds: 5 + failureThreshold: 30 + # -- Liveness probe to check if the container is alive. `timeoutSeconds` is + # set explicitly: Kubernetes defaults it to 1 second, and a service that is + # merely busy answers more slowly than that under load, which turns a slow + # pod into a restarting one. livenessProbe: httpGet: path: / port: http + periodSeconds: 20 + timeoutSeconds: 10 + failureThreshold: 6 # -- Readiness probe to check if the container is ready readinessProbe: httpGet: path: / port: http + periodSeconds: 10 + timeoutSeconds: 5 + failureThreshold: 3 # -- Resources to provide # [Resource Management for Pods and Containers](https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/) # @section -- Frequently Used @@ -445,7 +486,7 @@ server: # - secretName: chart-example-tls # hosts: # - chart-example.local -# Server Configuration +# END Server Configuration # Writer Configuration writer: @@ -479,13 +520,39 @@ writer: volumes: [] # -- Additional volumeMounts on the output Deployment definition volumeMounts: [] - # -- Liveness probe to check if the container is alive + # -- Node processes pm2 runs per writer pod. Each is single threaded, so a + # pod with more than one core does no more work until this is raised. Matches + # the value the hosted service runs. + pm2Instances: 2 + # -- Size of the emptyDir backing PM2_HOME (`/home/node/.pm2`). It holds pm2's + # sockets and, for services started from an ecosystem file, its log files. + # Disk-backed on purpose: on a memory-backed volume those log files count + # against the pod's memory limit and the container is OOMKilled under load. + pm2HomeSizeLimit: 256Mi + # -- Startup probe. While it is running the liveness and readiness probes are + # held off, so a slow start is not mistaken for an unhealthy container. + startupProbe: + exec: + command: + - ./node_modules/.bin/pm2 + - show + - writer-service + periodSeconds: 10 + timeoutSeconds: 15 + failureThreshold: 30 + # -- Liveness probe to check if the container is alive. The command forks a + # Node CLI, which cannot finish within the 1 second Kubernetes defaults + # `timeoutSeconds` to while the container is under CPU pressure. Timed-out + # probe processes then accumulate and make the pressure worse. livenessProbe: exec: command: - ./node_modules/.bin/pm2 - show - writer-service + periodSeconds: 30 + timeoutSeconds: 15 + failureThreshold: 5 # -- Readiness probe to check if the container is ready readinessProbe: exec: @@ -493,6 +560,9 @@ writer: - ./node_modules/.bin/pm2 - show - writer-service + periodSeconds: 30 + timeoutSeconds: 15 + failureThreshold: 5 # -- Resources to provide # [Resource Management for Pods and Containers](https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/) # @section -- Frequently Used @@ -547,13 +617,35 @@ scheduler: volumes: [] # -- Additional volumeMounts on the output Deployment definition volumeMounts: [] - # -- Liveness probe to check if the container is alive + # -- Size of the emptyDir backing PM2_HOME (`/home/node/.pm2`). It holds pm2's + # sockets and, for services started from an ecosystem file, its log files. + # Disk-backed on purpose: on a memory-backed volume those log files count + # against the pod's memory limit and the container is OOMKilled under load. + pm2HomeSizeLimit: 256Mi + # -- Startup probe. While it is running the liveness and readiness probes are + # held off, so a slow start is not mistaken for an unhealthy container. + startupProbe: + exec: + command: + - ./node_modules/.bin/pm2 + - show + - dist + periodSeconds: 10 + timeoutSeconds: 15 + failureThreshold: 30 + # -- Liveness probe to check if the container is alive. The command forks a + # Node CLI, which cannot finish within the 1 second Kubernetes defaults + # `timeoutSeconds` to while the container is under CPU pressure. Timed-out + # probe processes then accumulate and make the pressure worse. livenessProbe: exec: command: - ./node_modules/.bin/pm2 - show - dist + periodSeconds: 30 + timeoutSeconds: 15 + failureThreshold: 5 # -- Readiness probe to check if the container is ready readinessProbe: exec: @@ -561,6 +653,9 @@ scheduler: - ./node_modules/.bin/pm2 - show - dist + periodSeconds: 30 + timeoutSeconds: 15 + failureThreshold: 5 # -- Resources to provide # [Resource Management for Pods and Containers](https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/) # @section -- Frequently Used @@ -648,13 +743,35 @@ webhooks: volumes: [] # -- Additional volumeMounts on the output Deployment definition volumeMounts: [] - # -- Liveness probe to check if the container is alive + # -- Size of the emptyDir backing PM2_HOME (`/home/node/.pm2`). It holds pm2's + # sockets and, for services started from an ecosystem file, its log files. + # Disk-backed on purpose: on a memory-backed volume those log files count + # against the pod's memory limit and the container is OOMKilled under load. + pm2HomeSizeLimit: 256Mi + # -- Startup probe. While it is running the liveness and readiness probes are + # held off, so a slow start is not mistaken for an unhealthy container. + startupProbe: + exec: + command: + - ./node_modules/.bin/pm2 + - show + - dist + periodSeconds: 10 + timeoutSeconds: 15 + failureThreshold: 30 + # -- Liveness probe to check if the container is alive. The command forks a + # Node CLI, which cannot finish within the 1 second Kubernetes defaults + # `timeoutSeconds` to while the container is under CPU pressure. Timed-out + # probe processes then accumulate and make the pressure worse. livenessProbe: exec: command: - ./node_modules/.bin/pm2 - show - dist + periodSeconds: 30 + timeoutSeconds: 15 + failureThreshold: 5 # -- Readiness probe to check if the container is ready readinessProbe: exec: @@ -662,6 +779,9 @@ webhooks: - ./node_modules/.bin/pm2 - show - dist + periodSeconds: 30 + timeoutSeconds: 15 + failureThreshold: 5 # -- Liveness probe to check if the container is alive # -- Resources to provide # [Resource Management for Pods and Containers](https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/) @@ -755,6 +875,15 @@ redis: enabled: false master: resourcesPreset: "none" + # -- Requests without limits, deliberately. With neither, the Redis pod is + # BestEffort and is the first thing the kubelet evicts under node memory + # pressure, which drops every service's queue connection at once. No limit + # is set so a queue backlog cannot turn into an OOMKill instead. Size this + # to your queue depth. + resources: + requests: + cpu: 500m + memory: 1Gi replica: resourcesPreset: "none" sentinel: diff --git a/docs/configuration.md b/docs/configuration.md index 1d425be..3e33f82 100644 --- a/docs/configuration.md +++ b/docs/configuration.md @@ -1,6 +1,6 @@ # Configuration Reference -![Version: 0.7.4](https://img.shields.io/badge/Version-0.7.4-informational?style=flat-square) ![Type: application](https://img.shields.io/badge/Type-application-informational?style=flat-square) ![AppVersion: 2026-07-26-003](https://img.shields.io/badge/AppVersion-2026--07--26--003-informational?style=flat-square) +![Version: 0.7.5](https://img.shields.io/badge/Version-0.7.5-informational?style=flat-square) ![Type: application](https://img.shields.io/badge/Type-application-informational?style=flat-square) ![AppVersion: 2026-07-26-004](https://img.shields.io/badge/AppVersion-2026--07--26--004-informational?style=flat-square) ## Requirements @@ -88,7 +88,7 @@ The following table lists the configurable parameters of the `currents` chart an | Key | Type | Default | Description | |-----|------|---------|-------------| | currents.rootUser.email | string | `"admin@{{ .Values.currents.domains.appHost }}"` | The email address of the root user | -| currents.imageTag | string | `"2026-07-26-003"` | The image tag to use for the Currents images | +| currents.imageTag | string | `"2026-07-26-004"` | The image tag to use for the Currents images | | currents.email.transporter | string | `"smtp"` | Which transport to send outgoing email through: `smtp` or `ses`. With `ses` the SMTP settings are ignored and no SMTP credentials are needed — the AWS SDK resolves credentials from the pod itself, so grant the Currents service account permission to send. See [Using IAM Roles for Sending Email with SES](./eks/iam.md#using-iam-roles-for-sending-email-with-ses). | | currents.email.from | tpl/string | `""` | The email address to send from. Defaults to `currents.email.smtp.from` when unset, which is retained for compatibility. | | currents.email.ses.region | string | `""` | The AWS region to send through. Required when `transporter` is `ses`, and the `from` address must be a verified identity in that region. | @@ -133,8 +133,9 @@ The following table lists the configurable parameters of the `currents` chart an | director.env | list | `[]` | Env variables to pass to the container | | director.volumes | list | `[]` | Additional volumes on the output Deployment definition. | | director.volumeMounts | list | `[]` | Additional volumeMounts on the output Deployment definition. | -| director.livenessProbe | object | `{"httpGet":{"path":"/","port":"http"}}` | Liveness probe to check if the container is alive | -| director.readinessProbe | object | `{"httpGet":{"path":"/","port":"http"}}` | Readiness probe to check if the container is ready | +| director.startupProbe | object | `{"failureThreshold":30,"httpGet":{"path":"/","port":"http"},"periodSeconds":10,"timeoutSeconds":5}` | Startup probe. While it is running the liveness and readiness probes are held off, so a slow start is not mistaken for an unhealthy container. | +| director.livenessProbe | object | `{"failureThreshold":6,"httpGet":{"path":"/","port":"http"},"periodSeconds":20,"timeoutSeconds":10}` | Liveness probe to check if the container is alive. `timeoutSeconds` is set explicitly: Kubernetes defaults it to 1 second, and a service that is merely busy answers more slowly than that under load, which turns a slow pod into a restarting one. | +| director.readinessProbe | object | `{"failureThreshold":3,"httpGet":{"path":"/","port":"http"},"periodSeconds":10,"timeoutSeconds":5}` | Readiness probe to check if the container is ready | | director.nodeSelector | object | `{}` (defaults to global.nodeSelector) | [Node selector] | | director.tolerations | list | `[]` (defaults to global.tolerations) | [Tolerations] for use with node taints | | director.affinity | object | `{}` (defaults to the global.affinity preset) | Assign custom [affinity] rules to the deployment | @@ -150,8 +151,10 @@ The following table lists the configurable parameters of the `currents` chart an | server.env | list | `[]` | Env variables to pass to the container | | server.volumes | list | `[]` | Additional volumes on the output Deployment definition | | server.volumeMounts | list | `[]` | Additional volumeMounts on the output Deployment definition | -| server.livenessProbe | object | `{"httpGet":{"path":"/","port":"http"}}` | Liveness probe to check if the container is alive | -| server.readinessProbe | object | `{"httpGet":{"path":"/","port":"http"}}` | Readiness probe to check if the container is ready | +| server.pm2HomeSizeLimit | string | `"256Mi"` | Size of the emptyDir backing PM2_HOME (`/home/node/.pm2`). It holds pm2's sockets and, for services started from an ecosystem file, its log files. Disk-backed on purpose: on a memory-backed volume those log files count against the pod's memory limit and the container is OOMKilled under load. | +| server.startupProbe | object | `{"failureThreshold":30,"httpGet":{"path":"/","port":"http"},"periodSeconds":10,"timeoutSeconds":5}` | Startup probe. While it is running the liveness and readiness probes are held off, so a slow start is not mistaken for an unhealthy container. | +| server.livenessProbe | object | `{"failureThreshold":6,"httpGet":{"path":"/","port":"http"},"periodSeconds":20,"timeoutSeconds":10}` | Liveness probe to check if the container is alive. `timeoutSeconds` is set explicitly: Kubernetes defaults it to 1 second, and a service that is merely busy answers more slowly than that under load, which turns a slow pod into a restarting one. | +| server.readinessProbe | object | `{"failureThreshold":3,"httpGet":{"path":"/","port":"http"},"periodSeconds":10,"timeoutSeconds":5}` | Readiness probe to check if the container is ready | | server.nodeSelector | object | `{}` (defaults to global.nodeSelector) | [Node selector] | | server.tolerations | list | `[]` (defaults to global.tolerations) | [Tolerations] for use with node taints | | server.affinity | object | `{}` (defaults to the global.affinity preset) | Assign custom [affinity] rules to the deployment | @@ -167,8 +170,11 @@ The following table lists the configurable parameters of the `currents` chart an | writer.env | list | `[]` | Env variables to pass to the container | | writer.volumes | list | `[]` | Additional volumes on the output Deployment definition | | writer.volumeMounts | list | `[]` | Additional volumeMounts on the output Deployment definition | -| writer.livenessProbe | object | `{"exec":{"command":["./node_modules/.bin/pm2","show","writer-service"]}}` | Liveness probe to check if the container is alive | -| writer.readinessProbe | object | `{"exec":{"command":["./node_modules/.bin/pm2","show","writer-service"]}}` | Readiness probe to check if the container is ready | +| writer.pm2Instances | int | `2` | Node processes pm2 runs per writer pod. Each is single threaded, so a pod with more than one core does no more work until this is raised. Matches the value the hosted service runs. | +| writer.pm2HomeSizeLimit | string | `"256Mi"` | Size of the emptyDir backing PM2_HOME (`/home/node/.pm2`). It holds pm2's sockets and, for services started from an ecosystem file, its log files. Disk-backed on purpose: on a memory-backed volume those log files count against the pod's memory limit and the container is OOMKilled under load. | +| writer.startupProbe | object | `{"exec":{"command":["./node_modules/.bin/pm2","show","writer-service"]},"failureThreshold":30,"periodSeconds":10,"timeoutSeconds":15}` | Startup probe. While it is running the liveness and readiness probes are held off, so a slow start is not mistaken for an unhealthy container. | +| writer.livenessProbe | object | `{"exec":{"command":["./node_modules/.bin/pm2","show","writer-service"]},"failureThreshold":5,"periodSeconds":30,"timeoutSeconds":15}` | Liveness probe to check if the container is alive. The command forks a Node CLI, which cannot finish within the 1 second Kubernetes defaults `timeoutSeconds` to while the container is under CPU pressure. Timed-out probe processes then accumulate and make the pressure worse. | +| writer.readinessProbe | object | `{"exec":{"command":["./node_modules/.bin/pm2","show","writer-service"]},"failureThreshold":5,"periodSeconds":30,"timeoutSeconds":15}` | Readiness probe to check if the container is ready | | writer.nodeSelector | object | `{}` (defaults to global.nodeSelector) | [Node selector] | | writer.tolerations | list | `[]` (defaults to global.tolerations) | [Tolerations] for use with node taints | | writer.affinity | object | `{}` (defaults to the global.affinity preset) | Assign custom [affinity] rules to the deployment | @@ -183,8 +189,10 @@ The following table lists the configurable parameters of the `currents` chart an | scheduler.env | list | `[]` | Env variables to pass to the container | | scheduler.volumes | list | `[]` | Additional volumes on the output Deployment definition | | scheduler.volumeMounts | list | `[]` | Additional volumeMounts on the output Deployment definition | -| scheduler.livenessProbe | object | `{"exec":{"command":["./node_modules/.bin/pm2","show","dist"]}}` | Liveness probe to check if the container is alive | -| scheduler.readinessProbe | object | `{"exec":{"command":["./node_modules/.bin/pm2","show","dist"]}}` | Readiness probe to check if the container is ready | +| scheduler.pm2HomeSizeLimit | string | `"256Mi"` | Size of the emptyDir backing PM2_HOME (`/home/node/.pm2`). It holds pm2's sockets and, for services started from an ecosystem file, its log files. Disk-backed on purpose: on a memory-backed volume those log files count against the pod's memory limit and the container is OOMKilled under load. | +| scheduler.startupProbe | object | `{"exec":{"command":["./node_modules/.bin/pm2","show","dist"]},"failureThreshold":30,"periodSeconds":10,"timeoutSeconds":15}` | Startup probe. While it is running the liveness and readiness probes are held off, so a slow start is not mistaken for an unhealthy container. | +| scheduler.livenessProbe | object | `{"exec":{"command":["./node_modules/.bin/pm2","show","dist"]},"failureThreshold":5,"periodSeconds":30,"timeoutSeconds":15}` | Liveness probe to check if the container is alive. The command forks a Node CLI, which cannot finish within the 1 second Kubernetes defaults `timeoutSeconds` to while the container is under CPU pressure. Timed-out probe processes then accumulate and make the pressure worse. | +| scheduler.readinessProbe | object | `{"exec":{"command":["./node_modules/.bin/pm2","show","dist"]},"failureThreshold":5,"periodSeconds":30,"timeoutSeconds":15}` | Readiness probe to check if the container is ready | | scheduler.nodeSelector | object | `{}` (defaults to global.nodeSelector) | [Node selector] | | scheduler.tolerations | list | `[]` (defaults to global.tolerations) | [Tolerations] for use with node taints | | scheduler.affinity | object | `{}` (defaults to the global.affinity preset) | Assign custom [affinity] rules to the deployment | @@ -211,8 +219,10 @@ The following table lists the configurable parameters of the `currents` chart an | webhooks.env | list | `[]` | Env variables to pass to the container | | webhooks.volumes | list | `[]` | Additional volumes on the output Deployment definition | | webhooks.volumeMounts | list | `[]` | Additional volumeMounts on the output Deployment definition | -| webhooks.livenessProbe | object | `{"exec":{"command":["./node_modules/.bin/pm2","show","dist"]}}` | Liveness probe to check if the container is alive | -| webhooks.readinessProbe | object | `{"exec":{"command":["./node_modules/.bin/pm2","show","dist"]}}` | Readiness probe to check if the container is ready | +| webhooks.pm2HomeSizeLimit | string | `"256Mi"` | Size of the emptyDir backing PM2_HOME (`/home/node/.pm2`). It holds pm2's sockets and, for services started from an ecosystem file, its log files. Disk-backed on purpose: on a memory-backed volume those log files count against the pod's memory limit and the container is OOMKilled under load. | +| webhooks.startupProbe | object | `{"exec":{"command":["./node_modules/.bin/pm2","show","dist"]},"failureThreshold":30,"periodSeconds":10,"timeoutSeconds":15}` | Startup probe. While it is running the liveness and readiness probes are held off, so a slow start is not mistaken for an unhealthy container. | +| webhooks.livenessProbe | object | `{"exec":{"command":["./node_modules/.bin/pm2","show","dist"]},"failureThreshold":5,"periodSeconds":30,"timeoutSeconds":15}` | Liveness probe to check if the container is alive. The command forks a Node CLI, which cannot finish within the 1 second Kubernetes defaults `timeoutSeconds` to while the container is under CPU pressure. Timed-out probe processes then accumulate and make the pressure worse. | +| webhooks.readinessProbe | object | `{"exec":{"command":["./node_modules/.bin/pm2","show","dist"]},"failureThreshold":5,"periodSeconds":30,"timeoutSeconds":15}` | Readiness probe to check if the container is ready | | webhooks.nodeSelector | object | `{}` (defaults to global.nodeSelector) | [Node selector] | | webhooks.tolerations | list | `[]` (defaults to global.tolerations) | [Tolerations] for use with node taints | | webhooks.affinity | object | `{}` (defaults to the global.affinity preset) | Assign custom [affinity] rules to the deployment | @@ -236,6 +246,7 @@ The following table lists the configurable parameters of the `currents` chart an | redis.architecture | string | `"standalone"` | | | redis.auth.enabled | bool | `false` | | | redis.master.resourcesPreset | string | `"none"` | | +| redis.master.resources | object | `{"requests":{"cpu":"500m","memory":"1Gi"}}` | Requests without limits, deliberately. With neither, the Redis pod is BestEffort and is the first thing the kubelet evicts under node memory pressure, which drops every service's queue connection at once. No limit is set so a queue backlog cannot turn into an OOMKill instead. Size this to your queue depth. | | redis.replica.resourcesPreset | string | `"none"` | | | redis.sentinel.resourcesPreset | string | `"none"` | | | redis.metrics.resourcesPreset | string | `"none"` | |