apiVersion: apps/v1 kind: Deployment metadata: name: jarvis-api namespace: jarvis spec: replicas: 2 selector: matchLabels: app.kubernetes.io/name: jarvis app.kubernetes.io/component: api strategy: type: RollingUpdate rollingUpdate: # Never fewer pods than are being served now. A run holds its conversation through a unique # index, so a surge pod cannot double-answer anything — the old constraint that made an # overlap dangerous is gone. maxUnavailable: 0 maxSurge: 1 template: metadata: labels: app.kubernetes.io/name: jarvis app.kubernetes.io/component: api spec: # 90 seconds, and every part of it is accounted for: # 10s preStop, so this pod leaves the Service's endpoints BEFORE it starts draining # 25s RUN_SHUTDOWN_GRACE_SEC — in-flight assistant runs stop and close their transcripts # ~15s agent sockets closed and their rows written, httpServer.close(), Prisma, Redis # the rest is margin # The default of 30 would SIGKILL the pod in the middle of the run drain, which is the one # thing the drain exists to prevent. terminationGracePeriodSeconds: 90 # Required by the `restricted` Pod Security Standard that 00-namespace.yaml enforces. At pod # level rather than per container, because the agent-releases initContainer below needs it # too — and a missing seccompProfile is not a warning, it is the pod being refused outright. securityContext: seccompProfile: type: RuntimeDefault # SPREAD THE REPLICAS, because nothing else will. # # The scheduler has no reason of its own to put two pods of the same Deployment on different # nodes, and quite often has a reason not to. Two API replicas on one node is a deployment # that reads as highly available in `kubectl get pods` and loses every agent websocket, every # live run and every open shell the moment that one node reboots. The PodDisruptionBudget in # 07-disruption.yaml does not help here: it constrains voluntary evictions, and a node dying # is not one. # # `ScheduleAnyway` rather than `DoNotSchedule`: on a single-node cluster the strict form # leaves the second replica Pending forever, which is a worse first experience than an # unspread pair. On a real cluster it spreads. If yours spans zones, add a second constraint # on `topology.kubernetes.io/zone` — a rack is a failure domain too. topologySpreadConstraints: - maxSkew: 1 topologyKey: kubernetes.io/hostname whenUnsatisfiable: ScheduleAnyway labelSelector: matchLabels: app.kubernetes.io/name: jarvis app.kubernetes.io/component: api containers: - name: api # BY DIGEST. `:stable` is a moving name, so replicas that restart at different times land # on different builds — and a rollback becomes "hope the tag still points where it did". image: git.luxit.be/luxit/jarvis-api@sha256:REPLACE_ME ports: - name: http containerPort: 4000 lifecycle: preStop: exec: # Endpoint removal and SIGTERM race each other, and losing that race is visible: # a prompt that arrives after Nest sets `draining` is persisted and broadcast as # queued, and this pod will never answer it. Ten seconds is comfortably more than # kube-proxy needs to stop sending new connections here. command: ["/bin/sh", "-c", "sleep 10"] env: - name: JARVIS_SKIP_MIGRATIONS # The schema is the migration Job's business, once, before this rolls. N pods each # running `prisma db push` is N writers racing the same DDL while the previous # generation is still selecting the columns being altered. value: "true" envFrom: - configMapRef: name: jarvis-api-config # The half that is random bytes, minted once by `01-secret-job.yaml`. Kept apart from # the half an operator writes because one of them can be regenerated and the other # cannot: re-running the minter is a no-op by design, which is only safe while nothing # anybody typed lives in the same object. # # `optional: true` so a deployment that mints by hand — or through SealedSecrets, # ExternalSecrets, Vault — does not need this object to exist at all. - secretRef: name: jarvis-generated optional: true # LAST, and the order is the safety property rather than a formatting choice. `envFrom` # resolves in sequence and the last occurrence of a name wins, so anything written by # hand here overrides the generated value — never the other way round. Reversed, an # operator who had already minted a vault key and then applied the Job by reflex would # have every stored credential silently become unreadable. - secretRef: name: jarvis-secrets volumeMounts: - name: agent-releases mountPath: /srv/agent-releases readOnly: true # Liveness NEVER touches the database. A Postgres failover with a database check here # restarts every pod at once, turning a thirty-second blip into a cold start of the whole # deployment at the moment the database can least afford a reconnection stampede. livenessProbe: httpGet: path: /api/health/live port: http periodSeconds: 10 failureThreshold: 3 # Readiness DOES, and it checks Redis too. A pod cut off from Redis keeps answering HTTP # perfectly while its broadcasts reach nobody — the clients queue rather than fail — so it # looks healthier than a pod that is down and is more dangerous. readinessProbe: httpGet: path: /api/health/ready port: http periodSeconds: 5 timeoutSeconds: 3 failureThreshold: 3 # Boot does a run-recovery sweep and reads settings before it listens. The startup probe # is what stops liveness killing it halfway through on a slow database. startupProbe: httpGet: path: /api/health/live port: http periodSeconds: 5 failureThreshold: 30 resources: requests: cpu: 500m # Not generous — honest. A terminal recording holds up to 8 MiB of Buffers in heap # for the life of the session, and argon2 takes 64 MiB per share verification. memory: 1Gi limits: memory: 2Gi securityContext: runAsNonRoot: true runAsUser: 1000 allowPrivilegeEscalation: false capabilities: drop: ["ALL"] initContainers: # The compiled agent binaries, into this pod's own emptyDir. # # PER POD ON PURPOSE. The content is immutable and small, so a copy per pod costs nothing; # a shared RWO PVC would either pin every pod to one node or leave the second in # Multi-Attach error. Leaving this out entirely is supported: enrollment still works and the # installer answers 503 saying no build is published. - name: agent-releases image: git.luxit.be/luxit/jarvis-agent-dist@sha256:REPLACE_ME command: ["/bin/sh", "-c", "cp -a /dist/. /srv/agent-releases/"] # AN initContainer IS A CONTAINER, and the `restricted` Pod Security Standard judges it # exactly like the one below. This block was missing, and the result was not a warning # about the initContainer — it was the WHOLE POD refused, so the Deployment sat at zero # replicas reporting FailedCreate while every other line in this file was correct. securityContext: runAsNonRoot: true runAsUser: 1000 allowPrivilegeEscalation: false readOnlyRootFilesystem: true capabilities: drop: ["ALL"] volumeMounts: - name: agent-releases mountPath: /srv/agent-releases volumes: - name: agent-releases emptyDir: {} --- apiVersion: v1 kind: Service metadata: name: jarvis-api namespace: jarvis spec: type: ClusterIP # STATED, not left to the default, because the reflex is to reach for it the moment websockets # are involved. It would not help: the hard case is co-locating an operator's browser with an # agent's websocket, which arrives separately and whose id is unknown at the browser's handshake. # That is handled in the application, over Redis. See the README. sessionAffinity: None selector: app.kubernetes.io/name: jarvis app.kubernetes.io/component: api ports: - name: http port: 4000 targetPort: http