ceph: set OSD disk I/O scheduler to none via udev rule (#505)
## Why Ceph OSDs manage their own I/O ordering, so the kernel scheduler on the backing disks just adds overhead. The original intent was to set those disks to the `noop` scheduler. The whole OSD fleet (k8s + incus nodes) runs AlmaLinux 9 on blk-mq kernels (5.14), where the equivalent of `noop` is `none`. ## Changes - Add `profiles::ceph::osd_scheduler`, rendering a udev rule from the `ceph_osd_devices` fact (PR #504) that pins `queue/scheduler` to `none` on each OSD disk. - Reload udev and trigger the matched block devices so the setting applies immediately; the udev rule keeps it set across reboots and device re-add. - No-op when the fact is absent/empty, so VMs and non-OSD hosts are untouched. - Include the class from `profiles::ceph::osd` so it lands only on OSD hosts. https://claude.ai/code/session_01JUoARVdmhxKQHyyyp1pxeT --------- Co-authored-by: BenVincent <benvin@main.unkin.net> Reviewed-on: #505 Co-authored-by: Ben Vincent <ben@unkin.net> Co-committed-by: Ben Vincent <ben@unkin.net>
This commit was merged in pull request #505.
This commit is contained in:
@@ -2,6 +2,9 @@ class profiles::ceph::osd (
|
||||
Boolean $ensure_running = true,
|
||||
) {
|
||||
|
||||
# tune the I/O scheduler on the disks backing ceph OSDs
|
||||
include profiles::ceph::osd_scheduler
|
||||
|
||||
if $ensure_running and $facts['is_ceph_osd'] {
|
||||
$facts['ceph_services']['osd'].each |String $svc| {
|
||||
service { $svc:
|
||||
|
||||
@@ -0,0 +1,32 @@
|
||||
class profiles::ceph::osd_scheduler (
|
||||
String[1] $scheduler = 'none',
|
||||
) {
|
||||
|
||||
$devices = $facts['ceph_osd_devices']
|
||||
|
||||
# no-op where the fact is absent/empty (VMs, non-OSD hosts have no ceph PVs)
|
||||
if $devices =~ Array[String[1], 1] {
|
||||
|
||||
# strip /dev/ so the rule matches the udev KERNEL sysname (e.g. sda)
|
||||
$kernel_names = $devices.map |$dev| { regsubst($dev, '^.*/', '') }
|
||||
$sysname_matches = $kernel_names.map |$name| { "--sysname-match=${name}" }
|
||||
|
||||
file { '/etc/udev/rules.d/60-ceph-osd-scheduler.rules':
|
||||
ensure => file,
|
||||
owner => 'root',
|
||||
group => 'root',
|
||||
mode => '0644',
|
||||
content => template('profiles/ceph/osd-scheduler.rules.erb'),
|
||||
notify => Exec['ceph-osd-scheduler-reload'],
|
||||
}
|
||||
|
||||
# apply immediately; udev re-applies on reboot and device re-add
|
||||
$trigger = "udevadm trigger --subsystem-match=block --action=change ${join($sysname_matches, ' ')}"
|
||||
|
||||
exec { 'ceph-osd-scheduler-reload':
|
||||
command => "udevadm control --reload-rules && ${trigger}",
|
||||
path => ['/usr/bin', '/bin', '/usr/sbin', '/sbin'],
|
||||
refreshonly => true,
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,5 @@
|
||||
# Managed by puppet (profiles::ceph::osd_scheduler).
|
||||
# Set the I/O scheduler to <%= @scheduler %> on ceph OSD block devices.
|
||||
<% @kernel_names.sort.each do |dev| -%>
|
||||
ACTION=="add|change", SUBSYSTEM=="block", KERNEL=="<%= dev %>", ATTR{queue/scheduler}="<%= @scheduler %>"
|
||||
<% end -%>
|
||||
Reference in New Issue
Block a user