diff --git a/deploy/Containerfile.tei-base b/deploy/Containerfile.tei-base index 58e3887..dc7dbad 100644 --- a/deploy/Containerfile.tei-base +++ b/deploy/Containerfile.tei-base @@ -12,7 +12,7 @@ # Used as FROM base by: # Containerfile.tei-runtime -ARG TEI_VERSION=cpu-1.9.3 +ARG TEI_VERSION=cpu-1.9.4 # ── Stage 1: patchelf binary ────────────────────────────────────────────────── FROM debian:bookworm-slim AS tools diff --git a/deploy/helm/templates/embedding/pdb.yaml b/deploy/helm/templates/embedding/pdb.yaml new file mode 100644 index 0000000..6f67cd1 --- /dev/null +++ b/deploy/helm/templates/embedding/pdb.yaml @@ -0,0 +1,16 @@ +{{- if and .Values.embedding.enabled (gt (int .Values.embedding.replicaCount) 1) }} +apiVersion: policy/v1 +kind: PodDisruptionBudget +metadata: + name: {{ include "inference-stack.fullname" . }}-embedding + labels: + {{- include "inference-stack.labels" . | nindent 4 }} + app.kubernetes.io/component: embedding +spec: + maxUnavailable: 1 + selector: + matchLabels: + app.kubernetes.io/name: {{ include "inference-stack.name" . }} + app.kubernetes.io/instance: {{ .Release.Name }} + app.kubernetes.io/component: embedding +{{- end }} diff --git a/deploy/helm/values.yaml b/deploy/helm/values.yaml index 4d8b302..ca39176 100644 --- a/deploy/helm/values.yaml +++ b/deploy/helm/values.yaml @@ -154,10 +154,12 @@ embedding: # failureThreshold gives a legitimately-busy-but-alive pod enough runway # to drain its backlog before Kubernetes pulls it out of Service # Endpoints (which would make things worse, not better, since it reduces - # available capacity right when it's needed most). livenessProbe is left - # tighter/unchanged deliberately: restarting a busy-but-alive pod destroys - # its in-progress work and makes the backlog worse, so only a genuinely - # wedged process (not just a busy one) should trigger it. + # available capacity right when it's needed most). livenessProbe uses the + # same /health endpoint, so it queues behind the same backlog: with a + # tight threshold a busy-but-alive pod gets SIGTERMed (clean exit 0, not + # OOM), which destroys in-progress work and makes the backlog worse. The + # liveness window is therefore long (~5 min); only a genuinely wedged + # process should be restarted. readinessProbe: periodSeconds: 10 timeoutSeconds: 5 @@ -165,7 +167,7 @@ embedding: livenessProbe: periodSeconds: 30 timeoutSeconds: 10 - failureThreshold: 3 + failureThreshold: 10 # ~5 min of consecutive failures # Pod-level security context podSecurityContext: