From 826cb1b73f6935bf69caccad078e11346f0a41f2 Mon Sep 17 00:00:00 2001 From: Cody Maffucci <46459665+Maffooch@users.noreply.github.com> Date: Fri, 25 Sep 2026 17:02:35 -0600 Subject: [PATCH] fix(helm): raise uwsgi memory limit so the first login is not OOM-killed With the default 4 uwsgi processes, warm workers use about 500Mi of anonymous memory. Django's Argon2 hasher allocates about 100Mi per password check, so the first login or token-auth request pushes the container over its 512Mi limit. On cgroup v2 the kubelet OOM-kills the whole container, which is what the k8s CI job has been hitting: the first POST /api/v2/api-token-auth/ gets a 502 in most runs, and the job fails whenever the restarted container is killed again before the retry loop gives up. Raise the default limit to 1Gi. The CI workflow now also fails when a django container restarts during the test instead of letting the retry loop hide it, and on failure prints the previous container's logs, termination reason, pod description and events. Co-Authored-By: Claude Opus 5.5 --- .github/workflows/k8s-tests.yml | 30 ++++++++++++++++++++++++++++++ helm/defectdojo/Chart.yaml | 2 +- helm/defectdojo/README.md | 2 +- helm/defectdojo/values.yaml | 6 +++++- 4 files changed, 37 insertions(+), 3 deletions(-) diff --git a/.github/workflows/k8s-tests.yml b/.github/workflows/k8s-tests.yml index 1c7235fbe6f..a71913c6013 100644 --- a/.github/workflows/k8s-tests.yml +++ b/.github/workflows/k8s-tests.yml @@ -193,6 +193,23 @@ jobs: fi done + # The retry loops above tolerate a 502 and try again, so a uwsgi container that is + # killed and restarted after the readiness wait (for example by the OOM killer on the + # first password check) used to pass as long as the restarted container answered. + # Fail on any restart instead, so the "Failed Logs" step shows why it happened. + - name: Check django pod did not restart + timeout-minutes: 5 + run: |- + kubectl get pods --selector=defectdojo.org/component=django + restarts=$(kubectl get pods --selector=defectdojo.org/component=django \ + -o jsonpath='{range .items[*]}{range .status.containerStatuses[*]}{.restartCount}{"\n"}{end}{end}' \ + | awk '{s+=$1} END {print s+0}') + if [[ "$restarts" -ne 0 ]]; then + echo "ERROR: django pod containers restarted $restarts time(s) during the test" + exit 1 + fi + echo "No django container restarts" + - name: Check of logs timeout-minutes: 10 run: |- @@ -216,3 +233,16 @@ jobs: kubectl logs deployment/defectdojo-django --all-pods=true --all-containers=true --tail=100 echo "And all pod status one more time" kubectl get pods + echo "Container restarts and why the previous instance ended (OOMKilled means the memory limit was hit):" + kubectl get pods -o jsonpath='{range .items[*]}{.metadata.name}{"\n"}{range .status.containerStatuses[*]}{" "}{.name}{": restarts="}{.restartCount}{" lastState="}{.lastState.terminated.reason}{" exitCode="}{.lastState.terminated.exitCode}{" finishedAt="}{.lastState.terminated.finishedAt}{"\n"}{end}{end}' + echo "Describe of the django pods (probe failures, OOM kills and restarts are listed under Events):" + kubectl describe pods --selector=defectdojo.org/component=django + echo "Logs of the previous (terminated) django containers, if any:" + for pod in $(kubectl get pods --selector=defectdojo.org/component=django -o name); do + for container in uwsgi nginx; do + echo "--- $pod/$container (previous)" + kubectl logs "$pod" -c "$container" --previous --tail=100 || true + done + done + echo "Cluster events:" + kubectl get events --sort-by=.lastTimestamp diff --git a/helm/defectdojo/Chart.yaml b/helm/defectdojo/Chart.yaml index 4d35324688b..1a787fd588b 100644 --- a/helm/defectdojo/Chart.yaml +++ b/helm/defectdojo/Chart.yaml @@ -34,4 +34,4 @@ dependencies: # description: Critical bug annotations: artifacthub.io/prerelease: "true" - artifacthub.io/changes: "" + artifacthub.io/changes: "- kind: fixed\n description: Raise the default uwsgi memory limit to 1Gi so the first login no longer OOM-kills the container\n" diff --git a/helm/defectdojo/README.md b/helm/defectdojo/README.md index 44880e07e03..a0276abbf03 100644 --- a/helm/defectdojo/README.md +++ b/helm/defectdojo/README.md @@ -674,7 +674,7 @@ A Helm chart for Kubernetes to install DefectDojo | django.uwsgi.readinessProbe.successThreshold | int | `1` | | | django.uwsgi.readinessProbe.timeoutSeconds | int | `5` | | | django.uwsgi.resources.limits.cpu | string | `"2000m"` | | -| django.uwsgi.resources.limits.memory | string | `"512Mi"` | | +| django.uwsgi.resources.limits.memory | string | `"1Gi"` | | | django.uwsgi.resources.requests.cpu | string | `"100m"` | | | django.uwsgi.resources.requests.memory | string | `"256Mi"` | | | django.uwsgi.startupProbe.enabled | bool | `true` | Enable startup checks on uwsgi container. | diff --git a/helm/defectdojo/values.yaml b/helm/defectdojo/values.yaml index 29f8f900db5..ea4676629a9 100644 --- a/helm/defectdojo/values.yaml +++ b/helm/defectdojo/values.yaml @@ -465,7 +465,11 @@ django: memory: 256Mi limits: cpu: 2000m - memory: 512Mi + # With the default 4 processes, warm uwsgi workers use roughly 500Mi, and every + # password check (login, token auth, basic auth) allocates about 100Mi more for + # Argon2 while it runs. A 512Mi limit OOM-kills the whole uwsgi container on the + # first login. Raise this further if you raise processes/threads. + memory: 1Gi appSettings: processes: 4 threads: 4