Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 1 addition & 1 deletion deploy/Containerfile.tei-base
Original file line number Diff line number Diff line change
Expand Up @@ -12,7 +12,7 @@
# Used as FROM base by:
# Containerfile.tei-runtime

ARG TEI_VERSION=cpu-1.9.3
ARG TEI_VERSION=cpu-1.9.4

# ── Stage 1: patchelf binary ──────────────────────────────────────────────────
FROM debian:bookworm-slim AS tools
Expand Down
16 changes: 16 additions & 0 deletions deploy/helm/templates/embedding/pdb.yaml
Original file line number Diff line number Diff line change
@@ -0,0 +1,16 @@
{{- if and .Values.embedding.enabled (gt (int .Values.embedding.replicaCount) 1) }}
apiVersion: policy/v1
kind: PodDisruptionBudget
metadata:
name: {{ include "inference-stack.fullname" . }}-embedding
labels:
{{- include "inference-stack.labels" . | nindent 4 }}
app.kubernetes.io/component: embedding
spec:
maxUnavailable: 1
selector:
matchLabels:
app.kubernetes.io/name: {{ include "inference-stack.name" . }}
app.kubernetes.io/instance: {{ .Release.Name }}
app.kubernetes.io/component: embedding
{{- end }}
12 changes: 7 additions & 5 deletions deploy/helm/values.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -154,18 +154,20 @@ embedding:
# failureThreshold gives a legitimately-busy-but-alive pod enough runway
# to drain its backlog before Kubernetes pulls it out of Service
# Endpoints (which would make things worse, not better, since it reduces
# available capacity right when it's needed most). livenessProbe is left
# tighter/unchanged deliberately: restarting a busy-but-alive pod destroys
# its in-progress work and makes the backlog worse, so only a genuinely
# wedged process (not just a busy one) should trigger it.
# available capacity right when it's needed most). livenessProbe uses the
# same /health endpoint, so it queues behind the same backlog: with a
# tight threshold a busy-but-alive pod gets SIGTERMed (clean exit 0, not
# OOM), which destroys in-progress work and makes the backlog worse. The
# liveness window is therefore long (~5 min); only a genuinely wedged
# process should be restarted.
readinessProbe:
periodSeconds: 10
timeoutSeconds: 5
failureThreshold: 9 # ~90s tolerance for transient queue backlog
livenessProbe:
periodSeconds: 30
timeoutSeconds: 10
failureThreshold: 3
failureThreshold: 10 # ~5 min of consecutive failures

# Pod-level security context
podSecurityContext:
Expand Down
Loading