From c9c76f250b2929c6849577f14b992a765f53be65 Mon Sep 17 00:00:00 2001 From: Ben Vincent Date: Mon, 10 Aug 2026 23:26:52 +1000 Subject: [PATCH] =?UTF-8?q?Rework=20jellyfin-ha=20into=20a=E7=9C=9F-HA=20S?= =?UTF-8?q?tatefulSet=20deployment?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Turn the single-replica jellyfin-ha app into a proper high-availability deployment so the fork's Redis-coordinated distributed transcoding and PostgreSQL main database can actually be exercised. - Replace the Deployment with a 2-replica StatefulSet for stable pod identity; set JELLYFIN_INSTANCE_ID from metadata.name (the fork's Redis transcode-lease owner id), add soft podAntiAffinity and a PDB minAvailable 1. - Move the main Jellyfin DB to PostgreSQL via a CloudNativePG trio (3-instance Cluster, PgBouncer Pooler, Ceph RGW barman backups) mirroring the litellm pattern; an init container writes database.xml selecting the fork's Jellyfin-PostgreSQL provider and the DSN is composed from the CNPG-generated app secret pointed at the pooler. - Share /config on an RWX cephfs PVC across replicas; keep /cache per-pod via a volumeClaimTemplate. - Fix the transcode mount to the fork's real path /config/transcodes on the RWX PVC (raid5) so a surviving pod can resume the segments of the pod it takes over. - Add Intel iGPU hardware transcoding via the gpu.intel.com/i915 device plugin resource plus render/video supplemental groups. - Switch the Service to sessionAffinity ClientIP to reduce transcode churn. - Disable UDP auto-discovery. Library scans still run on every replica; a single-scanner leader election is a planned follow-up. --- apps/base/jellyfin/cnpg_backup.yaml | 45 ++++++ apps/base/jellyfin/cnpg_cluster.yaml | 119 +++++++++++++++ apps/base/jellyfin/cnpg_pooler.yaml | 36 +++++ apps/base/jellyfin/deployment.yaml | 80 ----------- apps/base/jellyfin/kustomization.yaml | 6 +- apps/base/jellyfin/pdb.yaml | 13 ++ apps/base/jellyfin/pvc-config.yaml | 8 +- apps/base/jellyfin/pvc-transcode.yaml | 5 +- apps/base/jellyfin/service.yaml | 3 +- apps/base/jellyfin/statefulset.yaml | 199 ++++++++++++++++++++++++++ 10 files changed, 427 insertions(+), 87 deletions(-) create mode 100644 apps/base/jellyfin/cnpg_backup.yaml create mode 100644 apps/base/jellyfin/cnpg_cluster.yaml create mode 100644 apps/base/jellyfin/cnpg_pooler.yaml delete mode 100644 apps/base/jellyfin/deployment.yaml create mode 100644 apps/base/jellyfin/pdb.yaml create mode 100644 apps/base/jellyfin/statefulset.yaml diff --git a/apps/base/jellyfin/cnpg_backup.yaml b/apps/base/jellyfin/cnpg_backup.yaml new file mode 100644 index 0000000..723c82c --- /dev/null +++ b/apps/base/jellyfin/cnpg_backup.yaml @@ -0,0 +1,45 @@ +--- +# Ceph RGW (S3) backup target for the jellyfin CNPG cluster, provisioned by the +# in-estate cephrgw-operator: one dedicated bucket + owner user. CNPG reads the +# S3 credential Secret from its own namespace. +apiVersion: ceph.unkin.net/v1alpha1 +kind: ObjectStoreUser +metadata: + name: cnpg-jellyfin-backup + namespace: jellyfin +spec: + displayName: "CNPG backup owner (jellyfin)" + uid: cnpg-jellyfin-backup + maxBuckets: 5 + secretName: cnpg-jellyfin-backup-s3 + retainOnDelete: true +--- +apiVersion: ceph.unkin.net/v1alpha1 +kind: Bucket +metadata: + name: cnpg-jellyfin + namespace: jellyfin +spec: + placementTarget: ec + bucketName: cnpg-jellyfin + ownerRef: cnpg-jellyfin-backup + versioning: false + tags: + app: jellyfin + purpose: cnpg-backup + retainOnDelete: true +--- +# Nightly base backup on top of always-on WAL archiving. Scheduled off-peak and +# staggered from the other CNPG clusters (6-field cron, seconds first). +apiVersion: postgresql.cnpg.io/v1 +kind: ScheduledBackup +metadata: + name: cnpg-jellyfin-nightly + namespace: jellyfin +spec: + schedule: "0 35 3 * * *" + immediate: false + backupOwnerReference: self + method: barmanObjectStore + cluster: + name: jellyfin-postgres diff --git a/apps/base/jellyfin/cnpg_cluster.yaml b/apps/base/jellyfin/cnpg_cluster.yaml new file mode 100644 index 0000000..9acd6c4 --- /dev/null +++ b/apps/base/jellyfin/cnpg_cluster.yaml @@ -0,0 +1,119 @@ +--- +# Main Jellyfin database. The jellyfin-ha fork's experimental EF Core provider +# moves the entire Jellyfin DB (incl. library items) off SQLite into PostgreSQL, +# which is what makes a shared-nothing multi-replica deployment possible. No +# bootstrap secret is given, so CNPG generates the jellyfin-postgres-app secret +# (username/password/dbname) that the StatefulSet composes its DSN from. +apiVersion: postgresql.cnpg.io/v1 +kind: Cluster +metadata: + name: jellyfin-postgres + namespace: jellyfin +spec: + affinity: + podAntiAffinityType: preferred + backup: + retentionPolicy: 30d + barmanObjectStore: + # Dedicated per-cluster Ceph RGW bucket (cephrgw-operator provisions it). + destinationPath: s3://cnpg-jellyfin + endpointURL: https://s3.ceph.unkin.net + endpointCA: + name: vault-ca-cert + key: ca.crt + s3Credentials: + accessKeyId: + name: cnpg-jellyfin-backup-s3 + key: AWS_ACCESS_KEY_ID + secretAccessKey: + name: cnpg-jellyfin-backup-s3 + key: AWS_SECRET_ACCESS_KEY + serverName: jellyfin + data: + compression: bzip2 + jobs: 2 + wal: + compression: zstd + maxParallel: 2 + bootstrap: + initdb: + database: jellyfin + encoding: UTF8 + localeCType: C + localeCollate: C + owner: jellyfin + enablePDB: true + enableSuperuserAccess: false + failoverDelay: 0 + # PG 17 — accepted by the fork's Npgsql/EF Core provider (needs PG14+); the + # provider generates its own migrations on first start. + imageName: ghcr.io/cloudnative-pg/postgresql:17-system-trixie + instances: 3 + logLevel: info + maxSyncReplicas: 0 + minSyncReplicas: 0 + monitoring: + customQueriesConfigMap: + - key: queries + name: cnpg-default-monitoring + disableDefaultQueries: false + enablePodMonitor: false + postgresql: + parameters: + archive_mode: "on" + archive_timeout: 5min + dynamic_shared_memory_type: posix + effective_cache_size: 256MB + full_page_writes: "on" + log_destination: csvlog + log_directory: /controller/log + log_filename: postgres + log_rotation_age: "0" + log_rotation_size: "0" + log_truncate_on_rotation: "false" + logging_collector: "on" + max_connections: "200" + max_parallel_workers: "16" + max_replication_slots: "16" + max_worker_processes: "16" + shared_buffers: 128MB + shared_memory_type: mmap + ssl_max_protocol_version: TLSv1.3 + ssl_min_protocol_version: TLSv1.3 + wal_keep_size: 256MB + wal_level: logical + wal_log_hints: "on" + wal_receiver_timeout: 5s + wal_sender_timeout: 5s + syncReplicaElectionConstraint: + enabled: false + primaryUpdateMethod: restart + primaryUpdateStrategy: unsupervised + probes: + liveness: + isolationCheck: + connectionTimeout: 1000 + enabled: true + requestTimeout: 1000 + replicationSlots: + highAvailability: + enabled: true + slotPrefix: _cnpg_ + synchronizeReplicas: + enabled: true + updateInterval: 30 + resources: + limits: + cpu: "1" + memory: 1Gi + requests: + cpu: 50m + memory: 512Mi + smartShutdownTimeout: 180 + startDelay: 3600 + stopDelay: 1800 + storage: + resizeInUseVolumes: true + size: 10Gi + storageClass: cephrbd-fast-delete + switchoverDelay: 3600 diff --git a/apps/base/jellyfin/cnpg_pooler.yaml b/apps/base/jellyfin/cnpg_pooler.yaml new file mode 100644 index 0000000..50f9263 --- /dev/null +++ b/apps/base/jellyfin/cnpg_pooler.yaml @@ -0,0 +1,36 @@ +--- +# PgBouncer pooler in front of the jellyfin-postgres cluster. Jellyfin connects +# here (jellyfin-postgres-pooler:5432) rather than the -rw service so EF Core's +# connection churn is absorbed by the pool. +apiVersion: postgresql.cnpg.io/v1 +kind: Pooler +metadata: + name: jellyfin-postgres-pooler + namespace: jellyfin +spec: + cluster: + name: jellyfin-postgres + instances: 2 + pgbouncer: + parameters: + default_pool_size: "50" + max_client_conn: "200" + paused: false + poolMode: session + template: + metadata: + labels: + app: jellyfin-pooler + spec: + affinity: + podAntiAffinity: + requiredDuringSchedulingIgnoredDuringExecution: + - labelSelector: + matchExpressions: + - key: app + operator: In + values: + - jellyfin-pooler + topologyKey: kubernetes.io/hostname + containers: [] + type: rw diff --git a/apps/base/jellyfin/deployment.yaml b/apps/base/jellyfin/deployment.yaml deleted file mode 100644 index 7aaed32..0000000 --- a/apps/base/jellyfin/deployment.yaml +++ /dev/null @@ -1,80 +0,0 @@ ---- -apiVersion: apps/v1 -kind: Deployment -metadata: - name: jellyfin - namespace: jellyfin -spec: - # Start single-replica. Scaling to >1 (true HA) is a follow-up once the Redis - # transcode store and RWX transcode scratch are validated end-to-end. - replicas: 1 - strategy: - # Config PVC is RWO; a Recreate rollout avoids two pods contending for it. - type: Recreate - selector: - matchLabels: - app: jellyfin - template: - metadata: - labels: - app: jellyfin - spec: - containers: - - name: jellyfin - image: git.unkin.net/unkin/jellyfin-ha:v0.1.0 - imagePullPolicy: IfNotPresent - ports: - - name: http - containerPort: 8096 - protocol: TCP - env: - # Distributed transcode session store (jellyfin-ha additions). - - name: Jellyfin__TranscodeStore__RedisConnectionString - value: "redis:6379,abortConnect=false" - - name: Jellyfin__TranscodeStore__LeaseDurationSeconds - value: "30" - livenessProbe: - httpGet: - path: /health - port: http - initialDelaySeconds: 30 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 3 - readinessProbe: - httpGet: - path: /health - port: http - initialDelaySeconds: 10 - periodSeconds: 10 - timeoutSeconds: 5 - failureThreshold: 3 - resources: - requests: - cpu: "1" - memory: 1Gi - limits: - cpu: "4" - memory: 6Gi - volumeMounts: - - name: config - mountPath: /config - - name: cache - mountPath: /cache - - name: transcode - mountPath: /transcode - - name: media - mountPath: /media - readOnly: true - volumes: - - name: config - persistentVolumeClaim: - claimName: jellyfin-config - - name: cache - emptyDir: {} - - name: transcode - persistentVolumeClaim: - claimName: jellyfin-transcode - - name: media - persistentVolumeClaim: - claimName: jellyfin-media diff --git a/apps/base/jellyfin/kustomization.yaml b/apps/base/jellyfin/kustomization.yaml index b0ada69..2fdb41c 100644 --- a/apps/base/jellyfin/kustomization.yaml +++ b/apps/base/jellyfin/kustomization.yaml @@ -4,10 +4,14 @@ kind: Kustomization resources: - namespace.yaml + - cnpg_cluster.yaml + - cnpg_pooler.yaml + - cnpg_backup.yaml - pvc-config.yaml - pvc-transcode.yaml - pvc-media.yaml - - deployment.yaml + - statefulset.yaml + - pdb.yaml - service.yaml - redis-deployment.yaml - redis-pvc.yaml diff --git a/apps/base/jellyfin/pdb.yaml b/apps/base/jellyfin/pdb.yaml new file mode 100644 index 0000000..4b21b5b --- /dev/null +++ b/apps/base/jellyfin/pdb.yaml @@ -0,0 +1,13 @@ +--- +# Keep at least one Jellyfin replica serving through voluntary disruptions +# (node drains, rollouts) so active streams can fail over rather than drop. +apiVersion: policy/v1 +kind: PodDisruptionBudget +metadata: + name: jellyfin + namespace: jellyfin +spec: + minAvailable: 1 + selector: + matchLabels: + app: jellyfin diff --git a/apps/base/jellyfin/pvc-config.yaml b/apps/base/jellyfin/pvc-config.yaml index 3bef28c..53e6a4f 100644 --- a/apps/base/jellyfin/pvc-config.yaml +++ b/apps/base/jellyfin/pvc-config.yaml @@ -1,5 +1,7 @@ --- -# Jellyfin config + SQLite library database. Single-writer, block storage. +# Jellyfin config: metadata images, plugins, subtitles and config XML. Shared +# ReadWriteMany across replicas (all pods read/write the same library metadata); +# the main library DB now lives in PostgreSQL, not here. Retain — this is state. apiVersion: v1 kind: PersistentVolumeClaim metadata: @@ -7,9 +9,9 @@ metadata: namespace: jellyfin spec: accessModes: - - ReadWriteOnce + - ReadWriteMany resources: requests: storage: 20Gi - storageClassName: cephrbd-fast-retain + storageClassName: cephfs-raid5-retain volumeMode: Filesystem diff --git a/apps/base/jellyfin/pvc-transcode.yaml b/apps/base/jellyfin/pvc-transcode.yaml index b33c821..d23721a 100644 --- a/apps/base/jellyfin/pvc-transcode.yaml +++ b/apps/base/jellyfin/pvc-transcode.yaml @@ -1,7 +1,8 @@ --- # Shared transcode scratch. ReadWriteMany is the hard requirement for the HA # fork: a taking-over pod must read the in-flight HLS segments written by the -# pod it replaces. Scratch data, so delete reclaim policy. +# pod it replaces. Scratch data (delete reclaim); raid5 avoids the raid6 +# double-parity write penalty on the many small HLS segment writes. apiVersion: v1 kind: PersistentVolumeClaim metadata: @@ -13,5 +14,5 @@ spec: resources: requests: storage: 100Gi - storageClassName: cephfs-raid6-delete + storageClassName: cephfs-raid5-delete volumeMode: Filesystem diff --git a/apps/base/jellyfin/service.yaml b/apps/base/jellyfin/service.yaml index 38b5c1d..a8c810e 100644 --- a/apps/base/jellyfin/service.yaml +++ b/apps/base/jellyfin/service.yaml @@ -13,5 +13,6 @@ spec: targetPort: http selector: app: jellyfin - sessionAffinity: None + # Pin each client to one replica to reduce transcode-session churn/takeover. + sessionAffinity: ClientIP type: ClusterIP diff --git a/apps/base/jellyfin/statefulset.yaml b/apps/base/jellyfin/statefulset.yaml new file mode 100644 index 0000000..8bed69d --- /dev/null +++ b/apps/base/jellyfin/statefulset.yaml @@ -0,0 +1,199 @@ +--- +apiVersion: apps/v1 +kind: StatefulSet +metadata: + name: jellyfin + namespace: jellyfin +spec: + # HA: two replicas coordinate transcode session ownership through Redis and + # resume each other's HLS segments off the shared RWX transcode PVC. Stable + # pod names (jellyfin-0/1) are the lease owner identity, hence StatefulSet. + replicas: 2 + serviceName: jellyfin + podManagementPolicy: Parallel + updateStrategy: + type: RollingUpdate + selector: + matchLabels: + app: jellyfin + template: + metadata: + labels: + app: jellyfin + spec: + securityContext: + # Group-write the shared RWX volumes and grant the render/video groups so + # the runAsUser 1000 process can open the Intel DRI render node injected + # by the device plugin. + fsGroup: 1000 + supplementalGroups: + - 44 + - 105 + - 109 + seccompProfile: + type: RuntimeDefault + affinity: + # Spread the two replicas across nodes for node-level HA. Soft so a + # single-GPU-node cluster still schedules both (i915 has 4 shared slots). + podAntiAffinity: + preferredDuringSchedulingIgnoredDuringExecution: + - weight: 100 + podAffinityTerm: + labelSelector: + matchLabels: + app: jellyfin + topologyKey: kubernetes.io/hostname + initContainers: + # Select the fork's experimental PostgreSQL provider by writing + # database.xml before Jellyfin starts. Runs as root to chown into the + # shared config volume; mirrors the fork Helm chart's inject-db-config. + - name: inject-db-config + image: busybox:1.37.0 + command: + - sh + - -c + - | + mkdir -p /config/config + chown 1000:1000 /config/config + chmod 775 /config/config + cat > /config/config/database.xml << 'DBEOF' + + + Jellyfin-PostgreSQL + NoLock + + DBEOF + chown 1000:1000 /config/config/database.xml + chmod 664 /config/config/database.xml + resources: + requests: + cpu: 10m + memory: 32Mi + limits: + cpu: 100m + memory: 64Mi + volumeMounts: + - name: config + mountPath: /config + containers: + - name: jellyfin + image: git.unkin.net/unkin/jellyfin-ha:v0.1.0 + imagePullPolicy: IfNotPresent + ports: + - name: http + containerPort: 8096 + protocol: TCP + env: + # Pod identity for the Redis transcode lease owner. The fork reads + # JELLYFIN_INSTANCE_ID (falling back to MachineName); the stable + # StatefulSet pod name gives each replica a unique lease identity so + # takeover can target a dead replica. JELLYFIN_HA_POD_NAME is set for + # parity with the fork Helm chart (nothing currently reads it). + - name: JELLYFIN_INSTANCE_ID + valueFrom: + fieldRef: + fieldPath: metadata.name + - name: JELLYFIN_HA_POD_NAME + valueFrom: + fieldRef: + fieldPath: metadata.name + # Multiple replicas must not each answer UDP auto-discovery. + - name: JELLYFIN_Network__AutoDiscovery + value: "false" + # Config dir must differ from the data root (Jellyfin sanity check). + - name: JELLYFIN_CONFIG_DIR + value: /config/config + # Distributed transcode session store (jellyfin-ha additions). + - name: Jellyfin__TranscodeStore__RedisConnectionString + value: "redis:6379,abortConnect=false" + - name: Jellyfin__TranscodeStore__LeaseDurationSeconds + value: "30" + # PostgreSQL main DB via the CNPG-generated app secret, routed through + # the PgBouncer pooler. Composed with $(VAR) expansion so the password + # is never rendered into the manifest; CNPG passwords are URL-safe. + - name: PGUSER + valueFrom: + secretKeyRef: + name: jellyfin-postgres-app + key: username + - name: PGPASSWORD + valueFrom: + secretKeyRef: + name: jellyfin-postgres-app + key: password + - name: PGDB + valueFrom: + secretKeyRef: + name: jellyfin-postgres-app + key: dbname + - name: POSTGRES_CONNECTION_STRING + value: "postgresql://$(PGUSER):$(PGPASSWORD)@jellyfin-postgres-pooler:5432/$(PGDB)" + - name: DATABASE_URL + value: "postgresql://$(PGUSER):$(PGPASSWORD)@jellyfin-postgres-pooler:5432/$(PGDB)" + livenessProbe: + httpGet: + path: /health + port: http + initialDelaySeconds: 30 + periodSeconds: 30 + timeoutSeconds: 5 + failureThreshold: 3 + readinessProbe: + httpGet: + path: /health + port: http + initialDelaySeconds: 10 + periodSeconds: 10 + timeoutSeconds: 5 + failureThreshold: 3 + resources: + requests: + cpu: "1" + memory: 1Gi + gpu.intel.com/i915: "1" + limits: + cpu: "4" + memory: 6Gi + # Intel iGPU (QSV/VA-API) slot. Requesting it pins the pod to a + # GPU-labelled node and injects /dev/dri/renderD* automatically, so + # no /dev/dri hostPath or privileged container is needed. Enable + # QSV/VA-API once in the Jellyfin admin UI; it persists to /config. + gpu.intel.com/i915: "1" + securityContext: + runAsUser: 1000 + runAsGroup: 1000 + volumeMounts: + - name: config + mountPath: /config + - name: transcode + # Fork's real transcode temp path. RWX so a surviving pod reads the + # in-flight .ts/.m3u8 segments of the pod it takes over. A per-pod + # volume here silently breaks HA takeover. + mountPath: /config/transcodes + - name: cache + mountPath: /cache + - name: media + mountPath: /media + readOnly: true + volumes: + - name: config + persistentVolumeClaim: + claimName: jellyfin-config + - name: transcode + persistentVolumeClaim: + claimName: jellyfin-transcode + - name: media + persistentVolumeClaim: + claimName: jellyfin-media + volumeClaimTemplates: + # Per-pod scratch cache — RWO, disposable, one PVC per replica. + - metadata: + name: cache + spec: + accessModes: + - ReadWriteOnce + resources: + requests: + storage: 30Gi + storageClassName: cephrbd-fast-delete + volumeMode: Filesystem