126 lines
4 KiB
YAML
126 lines
4 KiB
YAML
apiVersion: apps/v1
|
|
kind: Deployment
|
|
metadata:
|
|
name: prop
|
|
namespace: prop
|
|
spec:
|
|
# replicas is owned by the HPA in hpa.yaml (min 2 / max 5).
|
|
# Leaving it out here so Flux's periodic reconcile doesn't overwrite
|
|
# whatever replica count the autoscaler has settled on.
|
|
minReadySeconds: 5
|
|
strategy:
|
|
type: RollingUpdate
|
|
rollingUpdate:
|
|
maxSurge: 1
|
|
maxUnavailable: 1
|
|
selector:
|
|
matchLabels:
|
|
app: prop
|
|
template:
|
|
metadata:
|
|
labels:
|
|
# `tier: hot` distinguishes these pods from prop-backfill (same
|
|
# app: prop label) so the Service selector can exclude backfill
|
|
# from its endpoint pool — backfill runs PHX_SERVER=false and
|
|
# doesn't listen on :5000.
|
|
app: prop
|
|
tier: hot
|
|
spec:
|
|
tolerations:
|
|
- key: node-role.kubernetes.io/control-plane
|
|
operator: Exists
|
|
effect: NoSchedule
|
|
serviceAccountName: prop
|
|
imagePullSecrets:
|
|
- name: forgejo-registry
|
|
securityContext:
|
|
runAsUser: 65534
|
|
runAsNonRoot: true
|
|
fsGroup: 65534
|
|
seccompProfile:
|
|
type: RuntimeDefault
|
|
containers:
|
|
- name: prop
|
|
image: git.mcintire.me/graham/prop:main-1777154970-cbc20d0 # {"$imagepolicy": "flux-system:prop"}
|
|
imagePullPolicy: IfNotPresent
|
|
env:
|
|
- name: POD_IP
|
|
valueFrom:
|
|
fieldRef:
|
|
fieldPath: status.podIP
|
|
- name: PHX_SERVER
|
|
value: "true"
|
|
- name: PHX_HOST
|
|
value: "prop.w5isp.com"
|
|
- name: PORT
|
|
value: "5000"
|
|
- name: HRRR_BASE_URL
|
|
value: "http://skippy.w5isp.com:8080"
|
|
envFrom:
|
|
- secretRef:
|
|
name: prop-secrets
|
|
ports:
|
|
- name: http
|
|
containerPort: 5000
|
|
protocol: TCP
|
|
startupProbe:
|
|
httpGet:
|
|
path: /health
|
|
port: 5000
|
|
initialDelaySeconds: 5
|
|
periodSeconds: 3
|
|
failureThreshold: 10
|
|
timeoutSeconds: 3
|
|
readinessProbe:
|
|
httpGet:
|
|
path: /health
|
|
port: 5000
|
|
initialDelaySeconds: 5
|
|
periodSeconds: 5
|
|
failureThreshold: 2
|
|
timeoutSeconds: 2
|
|
# /live skips the DB — a saturated Ecto pool (e.g. a PromEx
|
|
# poller holding a connection >15s) would otherwise cause
|
|
# /health's `SELECT 1` to time out and the kubelet to SIGKILL
|
|
# every replica at once. Readiness still points at /health so
|
|
# DB outages drain the pod from Service endpoints gracefully.
|
|
# timeoutSeconds raised to 5s: under an IEM 429 retry storm
|
|
# Bandit's acceptor loop can stall briefly as the scheduler
|
|
# chews through retrying HTTP sleeps, and 3s was tripping
|
|
# failureThreshold across all 3 replicas at once.
|
|
livenessProbe:
|
|
httpGet:
|
|
path: /live
|
|
port: 5000
|
|
periodSeconds: 10
|
|
failureThreshold: 3
|
|
timeoutSeconds: 5
|
|
resources:
|
|
requests:
|
|
cpu: 100m
|
|
memory: 512Mi
|
|
limits:
|
|
# Bumped 2 → 3 on 2026-04-23: BEAM was getting 2 schedulers
|
|
# online on a 2-core limit, and the :05 propagation chain
|
|
# plus ScoreCache warming spiked the second run queue to
|
|
# ~10, starving /live and /health probe handlers and
|
|
# tripping liveness/readiness timeouts.
|
|
cpu: "3"
|
|
memory: 6Gi
|
|
securityContext:
|
|
allowPrivilegeEscalation: false
|
|
runAsNonRoot: true
|
|
runAsUser: 65534
|
|
capabilities:
|
|
drop:
|
|
- ALL
|
|
seccompProfile:
|
|
type: RuntimeDefault
|
|
volumeMounts:
|
|
- name: data
|
|
mountPath: /data
|
|
volumes:
|
|
- name: data
|
|
nfs:
|
|
server: 10.0.15.103
|
|
path: /data
|