Deploy jellyfin-ha as a true-HA StatefulSet under a new media project #237

Merged
benvin merged 10 commits from benvin/jellyfin-ha into main 2026-08-15 13:39:00 +10:00
28 changed files with 1047 additions and 0 deletions
@@ -0,0 +1,32 @@
---
# Second Ceph RGW (S3) bucket owned by the existing jellyfin backup user
# (cnpg-jellyfin-backup, defined in cnpg_backup.yaml) — one user, two buckets:
# the CNPG barman bucket plus this one, which k8up uses to hold restic backups
# of the jellyfin-config PVC. The BucketAccess emits read-write S3 creds into a
# Secret the k8up Schedule consumes.
apiVersion: ceph.unkin.net/v1alpha1
kind: Bucket
metadata:
name: jellyfin-config-backup
namespace: jellyfin
spec:
placementTarget: ec
bucketName: jellyfin-config-backup
ownerRef: cnpg-jellyfin-backup
versioning: false
tags:
app: jellyfin
purpose: config-backup
retainOnDelete: true
---
apiVersion: ceph.unkin.net/v1alpha1
kind: BucketAccess
metadata:
name: jellyfin-config-backup
namespace: jellyfin
spec:
bucketRef: jellyfin-config-backup
level: read-write
# Operator writes AWS_ACCESS_KEY_ID / AWS_SECRET_ACCESS_KEY (+ S3_ENDPOINT,
# BUCKET_NAME) into this Secret; the k8up Schedule reads the access keys.
secretName: jellyfin-config-backup-s3
+45
View File
@@ -0,0 +1,45 @@
---
# Ceph RGW (S3) backup target for the jellyfin CNPG cluster, provisioned by the
# in-estate cephrgw-operator: one dedicated bucket + owner user. CNPG reads the
# S3 credential Secret from its own namespace.
apiVersion: ceph.unkin.net/v1alpha1
kind: ObjectStoreUser
metadata:
name: cnpg-jellyfin-backup
namespace: jellyfin
spec:
displayName: "CNPG backup owner (jellyfin)"
uid: cnpg-jellyfin-backup
maxBuckets: 5
secretName: cnpg-jellyfin-backup-s3
retainOnDelete: true
---
apiVersion: ceph.unkin.net/v1alpha1
kind: Bucket
metadata:
name: cnpg-jellyfin
namespace: jellyfin
spec:
placementTarget: ec
bucketName: cnpg-jellyfin
ownerRef: cnpg-jellyfin-backup
versioning: false
tags:
app: jellyfin
purpose: cnpg-backup
retainOnDelete: true
---
# Nightly base backup on top of always-on WAL archiving. Scheduled off-peak and
# staggered from the other CNPG clusters (6-field cron, seconds first).
apiVersion: postgresql.cnpg.io/v1
kind: ScheduledBackup
metadata:
name: cnpg-jellyfin-nightly
namespace: jellyfin
spec:
schedule: "0 35 3 * * *"
immediate: false
backupOwnerReference: self
method: barmanObjectStore
cluster:
name: jellyfin-postgres
+126
View File
@@ -0,0 +1,126 @@
---
# Main Jellyfin database. The jellyfin-ha fork's experimental EF Core provider
# moves the entire Jellyfin DB (incl. library items) off SQLite into PostgreSQL,
# which is what makes a shared-nothing multi-replica deployment possible. No
# bootstrap secret is given, so CNPG generates the jellyfin-postgres-app secret
# (username/password/dbname) that the StatefulSet composes its DSN from.
apiVersion: postgresql.cnpg.io/v1
kind: Cluster
metadata:
name: jellyfin-postgres
namespace: jellyfin
spec:
# Exclude the operator-managed data PVCs (jellyfin-postgres-N) from the
# jellyfin-config k8up Schedule (skipWithoutAnnotation is false cluster-wide,
# so unannotated PVCs are swept in). Postgres has its own barmanObjectStore
# backup below; restic must not touch the raw RWO data volumes.
inheritedMetadata:
annotations:
k8up.io/backup: "false"
affinity:
podAntiAffinityType: preferred
backup:
retentionPolicy: 30d
barmanObjectStore:
# Dedicated per-cluster Ceph RGW bucket (cephrgw-operator provisions it).
destinationPath: s3://cnpg-jellyfin
endpointURL: https://s3.ceph.unkin.net
endpointCA:
name: vault-ca-cert
key: ca.crt
s3Credentials:
accessKeyId:
name: cnpg-jellyfin-backup-s3
key: AWS_ACCESS_KEY_ID
secretAccessKey:
name: cnpg-jellyfin-backup-s3
key: AWS_SECRET_ACCESS_KEY
serverName: jellyfin
data:
compression: bzip2
jobs: 2
wal:
compression: zstd
maxParallel: 2
bootstrap:
initdb:
database: jellyfin
encoding: UTF8
localeCType: C
localeCollate: C
owner: jellyfin
enablePDB: true
enableSuperuserAccess: false
failoverDelay: 0
# PG 17 — accepted by the fork's Npgsql/EF Core provider (needs PG14+); the
# provider generates its own migrations on first start.
imageName: ghcr.io/cloudnative-pg/postgresql:17-system-trixie
instances: 3
logLevel: info
maxSyncReplicas: 0
minSyncReplicas: 0
monitoring:
customQueriesConfigMap:
- key: queries
name: cnpg-default-monitoring
disableDefaultQueries: false
enablePodMonitor: false
postgresql:
parameters:
archive_mode: "on"
archive_timeout: 5min
dynamic_shared_memory_type: posix
effective_cache_size: 256MB
full_page_writes: "on"
log_destination: csvlog
log_directory: /controller/log
log_filename: postgres
log_rotation_age: "0"
log_rotation_size: "0"
log_truncate_on_rotation: "false"
logging_collector: "on"
max_connections: "200"
max_parallel_workers: "16"
max_replication_slots: "16"
max_worker_processes: "16"
shared_buffers: 128MB
shared_memory_type: mmap
ssl_max_protocol_version: TLSv1.3
ssl_min_protocol_version: TLSv1.3
wal_keep_size: 256MB
wal_level: logical
wal_log_hints: "on"
wal_receiver_timeout: 5s
wal_sender_timeout: 5s
syncReplicaElectionConstraint:
enabled: false
primaryUpdateMethod: restart
primaryUpdateStrategy: unsupervised
probes:
liveness:
isolationCheck:
connectionTimeout: 1000
enabled: true
requestTimeout: 1000
replicationSlots:
highAvailability:
enabled: true
slotPrefix: _cnpg_
synchronizeReplicas:
enabled: true
updateInterval: 30
resources:
limits:
cpu: "1"
memory: 1Gi
requests:
cpu: 50m
memory: 512Mi
smartShutdownTimeout: 180
startDelay: 3600
stopDelay: 1800
storage:
resizeInUseVolumes: true
size: 10Gi
storageClass: cephrbd-fast-delete
switchoverDelay: 3600
+36
View File
@@ -0,0 +1,36 @@
---
# PgBouncer pooler in front of the jellyfin-postgres cluster. Jellyfin connects
# here (jellyfin-postgres-pooler:5432) rather than the -rw service so EF Core's
# connection churn is absorbed by the pool.
apiVersion: postgresql.cnpg.io/v1
kind: Pooler
metadata:
name: jellyfin-postgres-pooler
namespace: jellyfin
spec:
cluster:
name: jellyfin-postgres
instances: 2
pgbouncer:
parameters:
default_pool_size: "50"
max_client_conn: "200"
paused: false
poolMode: session
template:
metadata:
labels:
app: jellyfin-pooler
spec:
affinity:
podAntiAffinity:
requiredDuringSchedulingIgnoredDuringExecution:
- labelSelector:
matchExpressions:
- key: app
operator: In
values:
- jellyfin-pooler
topologyKey: kubernetes.io/hostname
containers: []
type: rw
+37
View File
@@ -0,0 +1,37 @@
---
apiVersion: gateway.networking.k8s.io/v1
kind: Gateway
metadata:
labels:
traefik.io/instance: internal
annotations:
cert-manager.io/cluster-issuer: vault-issuer
cert-manager.io/common-name: jellyfin.k8s.syd1.au.unkin.net
cert-manager.io/private-key-size: "4096"
external-dns.alpha.kubernetes.io/hostname: jellyfin.k8s.syd1.au.unkin.net
external-dns.alpha.kubernetes.io/target: 198.18.200.4
name: jellyfin
namespace: jellyfin
spec:
gatewayClassName: traefik-internal
listeners:
- allowedRoutes:
namespaces:
from: Same
hostname: jellyfin.k8s.syd1.au.unkin.net
name: http
port: 80
protocol: HTTP
- allowedRoutes:
namespaces:
from: Same
hostname: jellyfin.k8s.syd1.au.unkin.net
name: https
port: 443
protocol: HTTPS
tls:
certificateRefs:
- group: ""
kind: Secret
name: jellyfin-tls
mode: Terminate
+49
View File
@@ -0,0 +1,49 @@
---
apiVersion: gateway.networking.k8s.io/v1
kind: HTTPRoute
metadata:
name: http-redirect
namespace: jellyfin
spec:
hostnames:
- jellyfin.k8s.syd1.au.unkin.net
parentRefs:
- group: gateway.networking.k8s.io
kind: Gateway
name: jellyfin
sectionName: http
rules:
- filters:
- type: RequestRedirect
requestRedirect:
scheme: https
statusCode: 301
matches:
- path:
type: PathPrefix
value: /
---
apiVersion: gateway.networking.k8s.io/v1
kind: HTTPRoute
metadata:
name: jellyfin-route
namespace: jellyfin
spec:
hostnames:
- jellyfin.k8s.syd1.au.unkin.net
parentRefs:
- group: gateway.networking.k8s.io
kind: Gateway
name: jellyfin
sectionName: https
rules:
- backendRefs:
- group: ""
kind: Service
name: jellyfin
port: 8096
weight: 1
matches:
- path:
type: PathPrefix
value: /
+27
View File
@@ -0,0 +1,27 @@
---
apiVersion: kustomize.config.k8s.io/v1beta1
kind: Kustomization
resources:
- namespace.yaml
- cnpg_cluster.yaml
- cnpg_pooler.yaml
- cnpg_backup.yaml
- cephrgw-config-backup.yaml
- vaultauth.yaml
- vaultstaticsecret.yaml
- schedule.yaml
- pvc-config.yaml
- pvc-transcode.yaml
- pv-media-tv.yaml
- pv-media-movies.yaml
- pvc-media-tv.yaml
- pvc-media-movies.yaml
- statefulset.yaml
- pdb.yaml
- service.yaml
- redis-deployment.yaml
- redis-pvc.yaml
- redis-service.yaml
- gateway.yaml
- httproute.yaml
+5
View File
@@ -0,0 +1,5 @@
---
apiVersion: v1
kind: Namespace
metadata:
name: jellyfin
+13
View File
@@ -0,0 +1,13 @@
---
# Keep at least one Jellyfin replica serving through voluntary disruptions
# (node drains, rollouts) so active streams can fail over rather than drop.
apiVersion: policy/v1
kind: PodDisruptionBudget
metadata:
name: jellyfin
namespace: jellyfin
spec:
minAvailable: 1
selector:
matchLabels:
app: jellyfin
+30
View File
@@ -0,0 +1,30 @@
---
# Static PV for the shared MOVIES CephFS subvolume. Same rootPath as arrstack's
# movies PV so radarr writes and jellyfin reads the identical library tree; each
# namespace gets its own PV (unique name + volumeHandle) pinned by claimRef.
apiVersion: v1
kind: PersistentVolume
metadata:
name: jellyfin-media-movies
spec:
capacity:
storage: 1Ti
accessModes:
- ReadWriteMany
persistentVolumeReclaimPolicy: Retain
storageClassName: ""
volumeMode: Filesystem
claimRef:
namespace: jellyfin
name: jellyfin-media-movies
csi:
driver: cephfs.csi.ceph.com
volumeHandle: jellyfin-media-movies-static
nodeStageSecretRef:
name: csi-cephfs-secret
namespace: csi-cephfs
volumeAttributes:
staticVolume: "true"
clusterID: cephfs_csi_ssd_ec_4_1
fsName: cephfs
rootPath: /volumes/csi_ssd_ec_4_1/media-movies/e95d8ace-c736-465a-acc3-0c3e46dcede9
+30
View File
@@ -0,0 +1,30 @@
---
# Static PV for the shared TV CephFS subvolume. Same rootPath as arrstack's
# TV PV so sonarr writes and jellyfin reads the identical library tree; each
# namespace gets its own PV (unique name + volumeHandle) pinned by claimRef.
apiVersion: v1
kind: PersistentVolume
metadata:
name: jellyfin-media-tv
spec:
capacity:
storage: 1Ti
accessModes:
- ReadWriteMany
persistentVolumeReclaimPolicy: Retain
storageClassName: ""
volumeMode: Filesystem
claimRef:
namespace: jellyfin
name: jellyfin-media-tv
csi:
driver: cephfs.csi.ceph.com
volumeHandle: jellyfin-media-tv-static
nodeStageSecretRef:
name: csi-cephfs-secret
namespace: csi-cephfs
volumeAttributes:
staticVolume: "true"
clusterID: cephfs_csi_ssd_ec_4_1
fsName: cephfs
rootPath: /volumes/csi_ssd_ec_4_1/media-tv/4692957d-f5df-4f72-b9c9-56e4ee6d1333
+17
View File
@@ -0,0 +1,17 @@
---
# Jellyfin config: metadata images, plugins, subtitles and config XML. Shared
# ReadWriteMany across replicas (all pods read/write the same library metadata);
# the main library DB now lives in PostgreSQL, not here. Retain — this is state.
apiVersion: v1
kind: PersistentVolumeClaim
metadata:
name: jellyfin-config
namespace: jellyfin
spec:
accessModes:
- ReadWriteMany
resources:
requests:
storage: 20Gi
storageClassName: cephfs-raid5-retain
volumeMode: Filesystem
+24
View File
@@ -0,0 +1,24 @@
---
# Movie library, shared read-many across replicas. Statically bound to the
# jellyfin-media-movies PV (shared CephFS subvolume also used by arrstack/radarr).
# storageClassName "" + volumeName disables dynamic provisioning and binds the
# pre-created static PV.
apiVersion: v1
kind: PersistentVolumeClaim
metadata:
name: jellyfin-media-movies
namespace: jellyfin
annotations:
# Exclude from the jellyfin-config k8up Schedule (skipWithoutAnnotation is
# false cluster-wide, so unannotated PVCs are swept in). Only jellyfin-config
# is backed up; the media library is not restic-backup material.
k8up.io/backup: "false"
spec:
accessModes:
- ReadWriteMany
resources:
requests:
storage: 1Ti
storageClassName: ""
volumeName: jellyfin-media-movies
volumeMode: Filesystem
+24
View File
@@ -0,0 +1,24 @@
---
# TV library, shared read-many across replicas. Statically bound to the
# jellyfin-media-tv PV (shared CephFS subvolume also used by arrstack/sonarr).
# storageClassName "" + volumeName disables dynamic provisioning and binds the
# pre-created static PV.
apiVersion: v1
kind: PersistentVolumeClaim
metadata:
name: jellyfin-media-tv
namespace: jellyfin
annotations:
# Exclude from the jellyfin-config k8up Schedule (skipWithoutAnnotation is
# false cluster-wide, so unannotated PVCs are swept in). Only jellyfin-config
# is backed up; the media library is not restic-backup material.
k8up.io/backup: "false"
spec:
accessModes:
- ReadWriteMany
resources:
requests:
storage: 1Ti
storageClassName: ""
volumeName: jellyfin-media-tv
volumeMode: Filesystem
+23
View File
@@ -0,0 +1,23 @@
---
# Shared transcode scratch. ReadWriteMany is the hard requirement for the HA
# fork: a taking-over pod must read the in-flight HLS segments written by the
# pod it replaces. Scratch data (delete reclaim); raid5 avoids the raid6
# double-parity write penalty on the many small HLS segment writes.
apiVersion: v1
kind: PersistentVolumeClaim
metadata:
name: jellyfin-transcode
namespace: jellyfin
annotations:
# Exclude from the jellyfin-config k8up Schedule (skipWithoutAnnotation is
# false cluster-wide, so unannotated PVCs are swept in). Transcode is RWX
# scratch — nothing to back up.
k8up.io/backup: "false"
spec:
accessModes:
- ReadWriteMany
resources:
requests:
storage: 100Gi
storageClassName: cephfs-raid5-delete
volumeMode: Filesystem
+66
View File
@@ -0,0 +1,66 @@
---
apiVersion: apps/v1
kind: Deployment
metadata:
name: redis
namespace: jellyfin
spec:
replicas: 1
selector:
matchLabels:
app: redis
strategy:
type: Recreate
template:
metadata:
labels:
app: redis
spec:
containers:
- name: redis
image: redis:7-alpine
imagePullPolicy: IfNotPresent
command:
- redis-server
- --save
- "20"
- "1"
ports:
- containerPort: 6379
name: redis
protocol: TCP
livenessProbe:
exec:
command:
- redis-cli
- ping
failureThreshold: 3
initialDelaySeconds: 30
periodSeconds: 30
successThreshold: 1
timeoutSeconds: 5
readinessProbe:
exec:
command:
- redis-cli
- ping
failureThreshold: 3
initialDelaySeconds: 5
periodSeconds: 10
successThreshold: 1
timeoutSeconds: 5
resources:
limits:
cpu: 500m
memory: 512Mi
requests:
cpu: 50m
memory: 128Mi
volumeMounts:
- mountPath: /data
name: data
restartPolicy: Always
volumes:
- name: data
persistentVolumeClaim:
claimName: jellyfin-redis-data
+19
View File
@@ -0,0 +1,19 @@
---
apiVersion: v1
kind: PersistentVolumeClaim
metadata:
name: jellyfin-redis-data
namespace: jellyfin
annotations:
# Exclude from the jellyfin-config k8up Schedule (skipWithoutAnnotation is
# false cluster-wide, so unannotated PVCs are swept in). Redis holds only
# ephemeral transcode-lease state; RWO would also fail to mount while in use.
k8up.io/backup: "false"
spec:
accessModes:
- ReadWriteOnce
resources:
requests:
storage: 5Gi
storageClassName: cephrbd-fast-delete
volumeMode: Filesystem
+17
View File
@@ -0,0 +1,17 @@
---
apiVersion: v1
kind: Service
metadata:
name: redis
namespace: jellyfin
spec:
internalTrafficPolicy: Cluster
ports:
- name: redis
port: 6379
protocol: TCP
targetPort: redis
selector:
app: redis
sessionAffinity: None
type: ClusterIP
+58
View File
@@ -0,0 +1,58 @@
---
# k8up Schedule: restic backups of the jellyfin-config PVC (library metadata,
# plugins, config XML) to the dedicated Ceph RGW config-backup bucket. S3 creds
# come from the cephrgw BucketAccess Secret (jellyfin-config-backup-s3); the
# restic repo password comes from Vault via the jellyfin-k8up-restic Secret.
#
# s3.ceph.unkin.net presents the internal unkin.net CA, which the k8up/restic
# image does not trust by default, so the reflected vault-ca-cert Secret is
# mounted into every job pod and pointed at via backend.tlsOptions.caCert.
apiVersion: k8up.io/v1
kind: Schedule
metadata:
name: jellyfin-config
namespace: jellyfin
spec:
backend:
repoPasswordSecretRef:
name: jellyfin-k8up-restic
key: password
s3:
endpoint: https://s3.ceph.unkin.net
bucket: jellyfin-config-backup
accessKeyIDSecretRef:
name: jellyfin-config-backup-s3
key: AWS_ACCESS_KEY_ID
secretAccessKeySecretRef:
name: jellyfin-config-backup-s3
key: AWS_SECRET_ACCESS_KEY
tlsOptions:
caCert: /etc/k8up/ca/ca.crt
volumeMounts:
- name: vault-ca
mountPath: /etc/k8up/ca
readOnly: true
backup:
schedule: "0 2 * * *"
failedJobsHistoryLimit: 3
successfulJobsHistoryLimit: 3
volumes:
- name: vault-ca
secret:
secretName: vault-ca-cert
prune:
schedule: "0 3 * * 0"
retention:
keepDaily: 14
keepWeekly: 8
keepMonthly: 12
volumes:
- name: vault-ca
secret:
secretName: vault-ca-cert
check:
schedule: "0 4 * * 0"
volumes:
- name: vault-ca
secret:
secretName: vault-ca-cert
+18
View File
@@ -0,0 +1,18 @@
---
apiVersion: v1
kind: Service
metadata:
name: jellyfin
namespace: jellyfin
spec:
internalTrafficPolicy: Cluster
ports:
- name: http
port: 8096
protocol: TCP
targetPort: http
selector:
app: jellyfin
# Pin each client to one replica to reduce transcode-session churn/takeover.
sessionAffinity: ClientIP
type: ClusterIP
+245
View File
@@ -0,0 +1,245 @@
---
apiVersion: apps/v1
kind: StatefulSet
metadata:
name: jellyfin
namespace: jellyfin
spec:
# HA: two replicas coordinate transcode session ownership through Redis and
# resume each other's HLS segments off the shared RWX transcode PVC. Stable
# pod names (jellyfin-0/1) are the lease owner identity, hence StatefulSet.
replicas: 2
serviceName: jellyfin
podManagementPolicy: Parallel
updateStrategy:
type: RollingUpdate
selector:
matchLabels:
app: jellyfin
template:
metadata:
labels:
app: jellyfin
spec:
securityContext:
# Group-write the shared RWX volumes and grant the render/video groups so
# the runAsUser 1000 process can open the Intel DRI render node injected
# by the device plugin.
fsGroup: 1000
supplementalGroups:
- 44
- 105
- 109
seccompProfile:
type: RuntimeDefault
affinity:
# Spread the two replicas across nodes for node-level HA. Soft so a
# single-GPU-node cluster still schedules both (i915 has 4 shared slots).
podAntiAffinity:
preferredDuringSchedulingIgnoredDuringExecution:
- weight: 100
podAffinityTerm:
labelSelector:
matchLabels:
app: jellyfin
topologyKey: kubernetes.io/hostname
initContainers:
# Seed the fork's PostgreSQL provider (database.xml) and Intel iGPU
# hardware transcode settings (encoding.xml) before Jellyfin starts.
# Runs as root to chown into the shared config volume; mirrors the fork
# Helm chart's inject-db-config. Each file is written only when absent so
# admin changes persisted to the shared RWX /config survive pod restarts.
- name: inject-config
image: busybox:1.37.0
command:
- sh
- -c
- |
mkdir -p /config/config
chown 1000:1000 /config/config
chmod 775 /config/config
if [ ! -e /config/config/database.xml ]; then
cat > /config/config/database.xml << 'DBEOF'
<?xml version="1.0" encoding="utf-8"?>
<DatabaseConfigurationOptions>
<DatabaseType>Jellyfin-PostgreSQL</DatabaseType>
<LockingBehavior>NoLock</LockingBehavior>
</DatabaseConfigurationOptions>
DBEOF
chown 1000:1000 /config/config/database.xml
chmod 664 /config/config/database.xml
fi
# VAAPI on the Intel render node the device plugin injects
# (/dev/dri/renderD128 — ffmpeg's default DRM node, reachable via
# the render/video supplementalGroups). Without this the attached
# iGPU is idle and every transcode runs in software. Omitted
# elements fall back to the fork's EncodingOptions defaults.
if [ ! -e /config/config/encoding.xml ]; then
cat > /config/config/encoding.xml << 'ENCEOF'
<?xml version="1.0" encoding="utf-8"?>
<EncodingOptions xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" xmlns:xsd="http://www.w3.org/2001/XMLSchema">
<EncodingThreadCount>-1</EncodingThreadCount>
<HardwareAccelerationType>vaapi</HardwareAccelerationType>
<VaapiDevice>/dev/dri/renderD128</VaapiDevice>
<EnableHardwareEncoding>true</EnableHardwareEncoding>
<AllowHevcEncoding>true</AllowHevcEncoding>
<AllowAv1Encoding>false</AllowAv1Encoding>
<EnableIntelLowPowerH264HwEncoder>false</EnableIntelLowPowerH264HwEncoder>
<EnableIntelLowPowerHevcHwEncoder>false</EnableIntelLowPowerHevcHwEncoder>
<EnableTonemapping>false</EnableTonemapping>
<EnableVppTonemapping>false</EnableVppTonemapping>
<HardwareDecodingCodecs>
<string>h264</string>
<string>hevc</string>
<string>vc1</string>
<string>vp9</string>
</HardwareDecodingCodecs>
</EncodingOptions>
ENCEOF
chown 1000:1000 /config/config/encoding.xml
chmod 664 /config/config/encoding.xml
fi
resources:
requests:
cpu: 10m
memory: 32Mi
limits:
cpu: 100m
memory: 64Mi
volumeMounts:
- name: config
mountPath: /config
containers:
- name: jellyfin
image: artifactapi.k8s.syd1.au.unkin.net/docker-internal/jellyfin-ha:v0.1.0
imagePullPolicy: IfNotPresent
ports:
- name: http
containerPort: 8096
protocol: TCP
env:
# Pod identity for the Redis transcode lease owner. The fork reads
# JELLYFIN_INSTANCE_ID (falling back to MachineName); the stable
# StatefulSet pod name gives each replica a unique lease identity so
# takeover can target a dead replica. JELLYFIN_HA_POD_NAME is set for
# parity with the fork Helm chart (nothing currently reads it).
- name: JELLYFIN_INSTANCE_ID
valueFrom:
fieldRef:
fieldPath: metadata.name
- name: JELLYFIN_HA_POD_NAME
valueFrom:
fieldRef:
fieldPath: metadata.name
# Multiple replicas must not each answer UDP auto-discovery.
- name: JELLYFIN_Network__AutoDiscovery
value: "false"
# Config dir must differ from the data root (Jellyfin sanity check).
- name: JELLYFIN_CONFIG_DIR
value: /config/config
# Distributed transcode session store (jellyfin-ha additions).
- name: Jellyfin__TranscodeStore__RedisConnectionString
value: "redis:6379,abortConnect=false"
- name: Jellyfin__TranscodeStore__LeaseDurationSeconds
value: "30"
# PostgreSQL main DB via the CNPG-generated app secret, routed through
# the PgBouncer pooler. Composed with $(VAR) expansion so the password
# is never rendered into the manifest; CNPG passwords are URL-safe.
- name: PGUSER
valueFrom:
secretKeyRef:
name: jellyfin-postgres-app
key: username
- name: PGPASSWORD
valueFrom:
secretKeyRef:
name: jellyfin-postgres-app
key: password
- name: PGDB
valueFrom:
secretKeyRef:
name: jellyfin-postgres-app
key: dbname
- name: POSTGRES_CONNECTION_STRING
value: "postgresql://$(PGUSER):$(PGPASSWORD)@jellyfin-postgres-pooler:5432/$(PGDB)"
- name: DATABASE_URL
value: "postgresql://$(PGUSER):$(PGPASSWORD)@jellyfin-postgres-pooler:5432/$(PGDB)"
livenessProbe:
httpGet:
path: /health
port: http
initialDelaySeconds: 30
periodSeconds: 30
timeoutSeconds: 5
failureThreshold: 3
readinessProbe:
httpGet:
path: /health
port: http
initialDelaySeconds: 10
periodSeconds: 10
timeoutSeconds: 5
failureThreshold: 3
resources:
requests:
cpu: "1"
memory: 1Gi
gpu.intel.com/i915: "1"
limits:
cpu: "4"
memory: 6Gi
# Intel iGPU (QSV/VA-API) slot. Requesting it pins the pod to a
# GPU-labelled node and injects /dev/dri/renderD* automatically, so
# no /dev/dri hostPath or privileged container is needed. VA-API is
# pre-enabled via the seeded encoding.xml (see inject-config), so
# transcodes use the iGPU on first boot with no manual UI step.
gpu.intel.com/i915: "1"
securityContext:
runAsUser: 1000
runAsGroup: 1000
volumeMounts:
- name: config
mountPath: /config
- name: transcode
# Fork's real transcode temp path. RWX so a surviving pod reads the
# in-flight .ts/.m3u8 segments of the pod it takes over. A per-pod
# volume here silently breaks HA takeover.
mountPath: /config/transcodes
- name: cache
mountPath: /cache
- name: media-tv
mountPath: /media/tv
readOnly: true
- name: media-movies
mountPath: /media/movies
readOnly: true
volumes:
- name: config
persistentVolumeClaim:
claimName: jellyfin-config
- name: transcode
persistentVolumeClaim:
claimName: jellyfin-transcode
- name: media-tv
persistentVolumeClaim:
claimName: jellyfin-media-tv
- name: media-movies
persistentVolumeClaim:
claimName: jellyfin-media-movies
volumeClaimTemplates:
# Per-pod scratch cache — RWO, disposable, one PVC per replica.
- metadata:
name: cache
annotations:
# Exclude the per-pod cache PVCs from the jellyfin-config k8up Schedule
# (skipWithoutAnnotation is false cluster-wide). Cache is disposable and
# RWO — it would also fail to mount into the backup pod while in use.
k8up.io/backup: "false"
spec:
accessModes:
- ReadWriteOnce
resources:
requests:
storage: 30Gi
storageClassName: cephrbd-fast-delete
volumeMode: Filesystem
+20
View File
@@ -0,0 +1,20 @@
---
apiVersion: secrets.hashicorp.com/v1beta1
kind: VaultAuth
metadata:
name: default
namespace: jellyfin
annotations:
argocd.argoproj.io/sync-wave: "0"
spec:
allowedNamespaces:
- jellyfin
kubernetes:
audiences:
- vault
role: default
serviceAccount: default
tokenExpirationSeconds: 600
method: kubernetes
mount: k8s/au/syd1
vaultConnectionRef: vso-system/default
+24
View File
@@ -0,0 +1,24 @@
---
# restic repository password for the k8up jellyfin-config backups. Seeded at
# kv/kubernetes/namespace/jellyfin/default/k8up-restic (key: password); the
# default k8s role's templated policy already grants read here, so no
# terraform-vault change is needed. VSO syncs it into the jellyfin-k8up-restic
# Secret that the Schedule references via backend.repoPasswordSecretRef.
apiVersion: secrets.hashicorp.com/v1beta1
kind: VaultStaticSecret
metadata:
name: jellyfin-k8up-restic
namespace: jellyfin
annotations:
argocd.argoproj.io/sync-wave: "0"
spec:
destination:
create: true
name: jellyfin-k8up-restic
overwrite: true
hmacSecretData: true
mount: kv
path: kubernetes/namespace/jellyfin/default/k8up-restic
refreshAfter: 5m
type: kv-v2
vaultAuthRef: default
@@ -0,0 +1,6 @@
---
apiVersion: kustomize.config.k8s.io/v1beta1
kind: Kustomization
resources:
- ../../../base/jellyfin
@@ -5,6 +5,7 @@ kind: Kustomization
resources:
- aitooling.yaml
- logging.yaml
- media.yaml
- observability.yaml
- platform.yaml
- storage.yaml
+31
View File
@@ -0,0 +1,31 @@
---
apiVersion: argoproj.io/v1alpha1
kind: ApplicationSet
metadata:
name: media-apps
namespace: argocd
spec:
generators:
- git:
repoURL: https://git.unkin.net/unkin/argocd-apps
revision: HEAD
directories:
- path: apps/overlays/*/jellyfin
template:
metadata:
name: 'media-{{path[3]}}'
spec:
project: media
source:
repoURL: https://git.unkin.net/unkin/argocd-apps
targetRevision: HEAD
path: '{{path}}'
destination:
server: https://kubernetes.default.svc
namespace: '{{path[3]}}'
syncPolicy:
automated:
prune: true
selfHeal: true
syncOptions:
- ServerSideApply=true
+1
View File
@@ -5,6 +5,7 @@ kind: Kustomization
resources:
- aitooling.yaml
- logging.yaml
- media.yaml
- observability.yaml
- platform.yaml
- storage.yaml
+23
View File
@@ -0,0 +1,23 @@
---
apiVersion: argoproj.io/v1alpha1
kind: AppProject
metadata:
name: media
namespace: argocd
spec:
description: Media services
sourceRepos:
- https://git.unkin.net/unkin/argocd-apps
destinations:
- namespace: 'jellyfin'
server: https://kubernetes.default.svc
- namespace: 'arrstack'
server: https://kubernetes.default.svc
clusterResourceWhitelist:
- group: ''
kind: Namespace
- group: ''
kind: PersistentVolume
namespaceResourceWhitelist:
- group: '*'
kind: '*'