From 804ea06499906a6cd1304e80c4a39b6d005292dc Mon Sep 17 00:00:00 2001 From: Ben Vincent Date: Sat, 8 Aug 2026 22:33:29 +1000 Subject: [PATCH] ceph: set OSD disk I/O scheduler to none via udev rule (#505) ## Why Ceph OSDs manage their own I/O ordering, so the kernel scheduler on the backing disks just adds overhead. The original intent was to set those disks to the `noop` scheduler. The whole OSD fleet (k8s + incus nodes) runs AlmaLinux 9 on blk-mq kernels (5.14), where the equivalent of `noop` is `none`. ## Changes - Add `profiles::ceph::osd_scheduler`, rendering a udev rule from the `ceph_osd_devices` fact (PR #504) that pins `queue/scheduler` to `none` on each OSD disk. - Reload udev and trigger the matched block devices so the setting applies immediately; the udev rule keeps it set across reboots and device re-add. - No-op when the fact is absent/empty, so VMs and non-OSD hosts are untouched. - Include the class from `profiles::ceph::osd` so it lands only on OSD hosts. https://claude.ai/code/session_01JUoARVdmhxKQHyyyp1pxeT --------- Co-authored-by: BenVincent Reviewed-on: https://git.unkin.net/unkin/puppet-prod/pulls/505 Co-authored-by: Ben Vincent Co-committed-by: Ben Vincent --- site/profiles/manifests/ceph/osd.pp | 3 ++ site/profiles/manifests/ceph/osd_scheduler.pp | 32 +++++++++++++++++++ .../templates/ceph/osd-scheduler.rules.erb | 5 +++ 3 files changed, 40 insertions(+) create mode 100644 site/profiles/manifests/ceph/osd_scheduler.pp create mode 100644 site/profiles/templates/ceph/osd-scheduler.rules.erb diff --git a/site/profiles/manifests/ceph/osd.pp b/site/profiles/manifests/ceph/osd.pp index d56ff1d..c1e093c 100644 --- a/site/profiles/manifests/ceph/osd.pp +++ b/site/profiles/manifests/ceph/osd.pp @@ -2,6 +2,9 @@ class profiles::ceph::osd ( Boolean $ensure_running = true, ) { + # tune the I/O scheduler on the disks backing ceph OSDs + include profiles::ceph::osd_scheduler + if $ensure_running and $facts['is_ceph_osd'] { $facts['ceph_services']['osd'].each |String $svc| { service { $svc: diff --git a/site/profiles/manifests/ceph/osd_scheduler.pp b/site/profiles/manifests/ceph/osd_scheduler.pp new file mode 100644 index 0000000..5d19622 --- /dev/null +++ b/site/profiles/manifests/ceph/osd_scheduler.pp @@ -0,0 +1,32 @@ +class profiles::ceph::osd_scheduler ( + String[1] $scheduler = 'none', +) { + + $devices = $facts['ceph_osd_devices'] + + # no-op where the fact is absent/empty (VMs, non-OSD hosts have no ceph PVs) + if $devices =~ Array[String[1], 1] { + + # strip /dev/ so the rule matches the udev KERNEL sysname (e.g. sda) + $kernel_names = $devices.map |$dev| { regsubst($dev, '^.*/', '') } + $sysname_matches = $kernel_names.map |$name| { "--sysname-match=${name}" } + + file { '/etc/udev/rules.d/60-ceph-osd-scheduler.rules': + ensure => file, + owner => 'root', + group => 'root', + mode => '0644', + content => template('profiles/ceph/osd-scheduler.rules.erb'), + notify => Exec['ceph-osd-scheduler-reload'], + } + + # apply immediately; udev re-applies on reboot and device re-add + $trigger = "udevadm trigger --subsystem-match=block --action=change ${join($sysname_matches, ' ')}" + + exec { 'ceph-osd-scheduler-reload': + command => "udevadm control --reload-rules && ${trigger}", + path => ['/usr/bin', '/bin', '/usr/sbin', '/sbin'], + refreshonly => true, + } + } +} diff --git a/site/profiles/templates/ceph/osd-scheduler.rules.erb b/site/profiles/templates/ceph/osd-scheduler.rules.erb new file mode 100644 index 0000000..28f87fb --- /dev/null +++ b/site/profiles/templates/ceph/osd-scheduler.rules.erb @@ -0,0 +1,5 @@ +# Managed by puppet (profiles::ceph::osd_scheduler). +# Set the I/O scheduler to <%= @scheduler %> on ceph OSD block devices. +<% @kernel_names.sort.each do |dev| -%> +ACTION=="add|change", SUBSYSTEM=="block", KERNEL=="<%= dev %>", ATTR{queue/scheduler}="<%= @scheduler %>" +<% end -%>