diff --git a/.github/workflows/k8s-tests.yml b/.github/workflows/k8s-tests.yml index 1c7235fbe6f..a71913c6013 100644 --- a/.github/workflows/k8s-tests.yml +++ b/.github/workflows/k8s-tests.yml @@ -193,6 +193,23 @@ jobs: fi done + # The retry loops above tolerate a 502 and try again, so a uwsgi container that is + # killed and restarted after the readiness wait (for example by the OOM killer on the + # first password check) used to pass as long as the restarted container answered. + # Fail on any restart instead, so the "Failed Logs" step shows why it happened. + - name: Check django pod did not restart + timeout-minutes: 5 + run: |- + kubectl get pods --selector=defectdojo.org/component=django + restarts=$(kubectl get pods --selector=defectdojo.org/component=django \ + -o jsonpath='{range .items[*]}{range .status.containerStatuses[*]}{.restartCount}{"\n"}{end}{end}' \ + | awk '{s+=$1} END {print s+0}') + if [[ "$restarts" -ne 0 ]]; then + echo "ERROR: django pod containers restarted $restarts time(s) during the test" + exit 1 + fi + echo "No django container restarts" + - name: Check of logs timeout-minutes: 10 run: |- @@ -216,3 +233,16 @@ jobs: kubectl logs deployment/defectdojo-django --all-pods=true --all-containers=true --tail=100 echo "And all pod status one more time" kubectl get pods + echo "Container restarts and why the previous instance ended (OOMKilled means the memory limit was hit):" + kubectl get pods -o jsonpath='{range .items[*]}{.metadata.name}{"\n"}{range .status.containerStatuses[*]}{" "}{.name}{": restarts="}{.restartCount}{" lastState="}{.lastState.terminated.reason}{" exitCode="}{.lastState.terminated.exitCode}{" finishedAt="}{.lastState.terminated.finishedAt}{"\n"}{end}{end}' + echo "Describe of the django pods (probe failures, OOM kills and restarts are listed under Events):" + kubectl describe pods --selector=defectdojo.org/component=django + echo "Logs of the previous (terminated) django containers, if any:" + for pod in $(kubectl get pods --selector=defectdojo.org/component=django -o name); do + for container in uwsgi nginx; do + echo "--- $pod/$container (previous)" + kubectl logs "$pod" -c "$container" --previous --tail=100 || true + done + done + echo "Cluster events:" + kubectl get events --sort-by=.lastTimestamp diff --git a/helm/defectdojo/Chart.yaml b/helm/defectdojo/Chart.yaml index 4d35324688b..1a787fd588b 100644 --- a/helm/defectdojo/Chart.yaml +++ b/helm/defectdojo/Chart.yaml @@ -34,4 +34,4 @@ dependencies: # description: Critical bug annotations: artifacthub.io/prerelease: "true" - artifacthub.io/changes: "" + artifacthub.io/changes: "- kind: fixed\n description: Raise the default uwsgi memory limit to 1Gi so the first login no longer OOM-kills the container\n" diff --git a/helm/defectdojo/README.md b/helm/defectdojo/README.md index 44880e07e03..a0276abbf03 100644 --- a/helm/defectdojo/README.md +++ b/helm/defectdojo/README.md @@ -674,7 +674,7 @@ A Helm chart for Kubernetes to install DefectDojo | django.uwsgi.readinessProbe.successThreshold | int | `1` | | | django.uwsgi.readinessProbe.timeoutSeconds | int | `5` | | | django.uwsgi.resources.limits.cpu | string | `"2000m"` | | -| django.uwsgi.resources.limits.memory | string | `"512Mi"` | | +| django.uwsgi.resources.limits.memory | string | `"1Gi"` | | | django.uwsgi.resources.requests.cpu | string | `"100m"` | | | django.uwsgi.resources.requests.memory | string | `"256Mi"` | | | django.uwsgi.startupProbe.enabled | bool | `true` | Enable startup checks on uwsgi container. | diff --git a/helm/defectdojo/values.yaml b/helm/defectdojo/values.yaml index 29f8f900db5..ea4676629a9 100644 --- a/helm/defectdojo/values.yaml +++ b/helm/defectdojo/values.yaml @@ -465,7 +465,11 @@ django: memory: 256Mi limits: cpu: 2000m - memory: 512Mi + # With the default 4 processes, warm uwsgi workers use roughly 500Mi, and every + # password check (login, token auth, basic auth) allocates about 100Mi more for + # Argon2 while it runs. A 512Mi limit OOM-kills the whole uwsgi container on the + # first login. Raise this further if you raise processes/threads. + memory: 1Gi appSettings: processes: 4 threads: 4