Files
fan_speed/fan_speed.sh
Zeb Hering 4dffbb070c Let the fan settle downward on a re-assert
A symmetric deadband catches the fan on the way up and holds it there:
on iz-pve0 under load, 17 of 20 samples ran at 38% while the curve
wanted 34-36%, and nothing would bring it down until the request fell a
full 8 points. Damping was turning into a permanently louder machine.

A re-assert may now settle the speed downward to the current request.
Upward moves still require a full deadband crossing, so the drift this
was built to prevent cannot come back - covered by a test.

Verified under a live migration: settled 38 -> 36 on the first
re-assert, then one change in ten minutes while the request wandered
36-42, CPU stable at 66-69c and disks a degree cooler.

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
2026-08-28 22:33:38 -07:00

401 lines
19 KiB
Bash
Executable File

#!/bin/bash
#
# Dell PowerEdge fan control driven by CPU, GPU and disk temperature, with
# trend tracking and email alarms.
#
# Usage: fan_speed.sh [run|once|disks|selftest] (default: run)
SCRIPT_DIR=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)
source "$SCRIPT_DIR/functions.sh"
source "$SCRIPT_DIR/monitor.sh"
# Trap the signals for container exit and run graceful_exit function
trap 'graceful_exit' SIGINT SIGQUIT SIGTERM
# Variable defintions
# LOW_FAN_SPEED = Lowest fan speed
# HIGH_FAN_SPEED = max fan speed. (0-100)
# IDRAC_HOST = local or IP of iDRAC
# IDRAC_USERNAME = username of iDRAC
# IDRAC_PASSWORD = password if iDRAC
# CHECK_INTERVAL = time to wait in seconds before performing another temp check (if doing GPU might want to keep this low like 5 seconds)
# CPU_TEMPERATURE_THRESHOLD = max CPU temp before disabling this automation and setting idrac back to auto
# GPU_TEMPERATURE_THRESHOLD = max GPU TEMP before running a kill on the container running LLM
# TABLE_HEADER_PRINT_INTERVAL = Number of loops before we output temps
IDRAC_HOST=local
IDRAC_USERNAME=root
IDRAC_PASSWORD=${IDRAC_PASSWORD:-calvin}
CHECK_INTERVAL=10 #takes about 8 seconds to query all data
CPU_TEMPERATURE_THRESHOLD=90 #only decimal numbers
GPU_TEMPERATURE_THRESHOLD=75 #only decimal numbers
# Per-chassis: the PWM that matches what Dell's own profile runs at idle. The
# PWM->RPM relationship is chassis specific, so this is measured, not guessed -
# see "Calibrating a host" in the README. 18/50 suits an R730xd that idles at
# 48c; an R720xd idling at 60c needs a floor of 28 to match stock.
LOW_FAN_SPEED=${LOW_FAN_SPEED:-18}
HIGH_FAN_SPEED=${HIGH_FAN_SPEED:-50}
TABLE_HEADER_PRINT_INTERVAL=2
# The CPU ramp is derived from the CPU sensor's own upper-non-critical
# threshold as reported by iDRAC, exactly like the disks are derived from
# theirs. A fixed 45c start was calibrated on a chassis that idles at 48c; on
# one that idles at 58c it sat 40% up the ramp doing nothing useful, and ran
# ~10% more airflow than Dell's own profile chose at the same temperature
# (6878 vs 6240 RPM measured over 8 minutes on an R720xd).
#
# Offsets chosen so a 58c idle lands near what Dell picks. Both an R720xd and
# an R730xd report upper-non-critical 77c, so this resolves to a 53->69c ramp.
CPU_RAMP_LOW_OFFSET=${CPU_RAMP_LOW_OFFSET:-24}
CPU_RAMP_HIGH_OFFSET=${CPU_RAMP_HIGH_OFFSET:-8}
CPU_LIMIT_FALLBACK=${CPU_LIMIT_FALLBACK:-77}
# Disk thresholds, alarm routing and trend window all live in monitor.sh.
# iDRAC's declared upper-non-critical temperature for the CPU sensor.
cpu_upper_non_critical() {
ipmitool -I $IDRAC_LOGIN_STRING sdr get "Temp" 2>/dev/null |
awk -F: '/Upper non-critical/ {gsub(/[^0-9.]/,"",$2); print int($2); exit}'
}
#######################################
# Self test - pure logic only, no IPMI and no drives touched.
selftest() {
local tmp; tmp=$(mktemp -d)
export STATE_DIR="$tmp/state" TREND_FILE="$tmp/temps.csv" LOG_FILE="$tmp/log" DRY_RUN=1
local fail=0
check() { # check <description> <expected> <actual>
if [ "$2" = "$3" ]; then echo " ok $1"; else echo " FAIL $1: expected '$2' got '$3'"; fail=1; fi
}
echo "temperature parsing:"
check "SATA Temperature_Celsius" 35 \
"$(printf '194 Temperature_Celsius 0x0022 115 099 000 Old_age Always - 35\n' | parse_disk_temperature)"
check "Seagate Airflow_Temperature_Cel" 41 \
"$(printf '190 Airflow_Temperature_Cel 0x0032 059 051 000 Old_age Always - 41\n' | parse_disk_temperature)"
check "Intel Temperature_Internal" 46 \
"$(printf '194 Temperature_Internal 0x0022 100 100 000 Old_age Always - 46\n' | parse_disk_temperature)"
check "SAS Current Drive Temperature" 46 \
"$(printf 'Current Drive Temperature: 46 C\n' | parse_disk_temperature)"
check "Temperature_Difference ignored" "" \
"$(printf '190 Temperature_Difference_from_100 0x0022 059 051 000 Old_age Always - 41\n' | parse_disk_temperature)"
check "no temperature reported" "" "$(printf 'Device Model: whatever\n' | parse_disk_temperature)"
echo "limit parsing:"
check "SATA Min/Max limit" 70 "$(printf 'Min/Max Temperature Limit: 0/70 Celsius\n' | parse_disk_limit)"
check "SATA negative min" 85 "$(printf 'Min/Max Temperature Limit: -41/85 Celsius\n' | parse_disk_limit)"
check "SAS trip temperature" 60 "$(printf 'Drive Trip Temperature: 60 C\n' | parse_disk_limit)"
echo "fan curve (limit 70 -> ramp 52..62, limit 60 -> ramp 42..52):"
check "SSD at 47c is below its ramp" 18 "$(calculate_interpolated_fan_speed 47 52 62 18 50)"
check "SSD at 57c is mid ramp" 34 "$(calculate_interpolated_fan_speed 57 52 62 18 50)"
check "SSD at 62c is flat out" 50 "$(calculate_interpolated_fan_speed 62 52 62 18 50)"
check "HDD at 47c is mid ramp" 34 "$(calculate_interpolated_fan_speed 47 42 52 18 50)"
echo "trend detection:"
TREND_SAMPLES=5
: > "$TREND_FILE"
for t in 40 40 41 40 40; do echo "0,50,0,$t,20" >> "$TREND_FILE"; done
check "flat series does not alarm" 0 "$(rise_over_window 4)"
: > "$TREND_FILE"
for t in 38 41 44 47 50; do echo "0,50,0,$t,20" >> "$TREND_FILE"; done
check "rising series reports climb" 12 "$(rise_over_window 4)"
: > "$TREND_FILE"
echo "0,50,0,38,20" >> "$TREND_FILE"
check "partial window stays silent" 0 "$(rise_over_window 4)"
echo "fan speed deadband (hold unless the change is worth making):"
HIGH_FAN_SPEED=50
check "1% drift holds" 20 "$(next_fan_speed 21 20)"
check "3% drift holds" 20 "$(next_fan_speed 23 20)"
check "4% drift holds" 32 "$(next_fan_speed 36 32)"
check "5% rise is adopted" 25 "$(next_fan_speed 25 20)"
check "5% fall is adopted" 15 "$(next_fan_speed 15 20)"
check "full speed always adopted" 50 "$(next_fan_speed 50 20)"
check "first pass adopts the request" 21 "$(next_fan_speed 21 '')"
# A re-assert must re-push what is held, not adopt the current want - that
# leak is what let the setpoint drift 32 -> 36 -> 31 on iz-pve0.
check "re-assert re-pushes held value" 32 "$(next_fan_speed 34 32)"
echo "settling downward (damping must not make the machine permanently louder):"
check "holds above the want between re-asserts" 38 "$(settle_fan_speed 34 38 0)"
check "settles down to the want on re-assert" 34 "$(settle_fan_speed 34 38 1)"
check "small rise still held on re-assert" 38 "$(settle_fan_speed 42 38 1)"
check "real rise still adopted" 46 "$(settle_fan_speed 46 38 1)"
# The old upward leak: a re-assert must never drift the setpoint up.
check "no upward drift on re-assert" 32 "$(settle_fan_speed 36 32 1)"
echo "trend verdict (load is not a fault - only cooling that cannot keep up is):"
TREND_SAMPLES=5
HIGH_FAN_SPEED=50
mkwindow() { : > "$TREND_FILE"; for pair in $@; do echo "0,${pair%%:*},0,${pair%%:*},${pair##*:}" >> "$TREND_FILE"; done; }
mkwindow 50:20 50:20 51:20 50:20 50:20
check "flat temps, fans steady" quiet "$(trend_verdict 2 5)"
mkwindow 50:50 54:50 57:50 59:50 60:50
check "climbing with fans flat out" no_headroom "$(trend_verdict 2 5)"
mkwindow 50:20 53:25 56:30 59:35 62:40
check "climbing as fast as the fans ramp" not_converging "$(trend_verdict 2 5)"
mkwindow 50:20 55:30 58:35 59:35 59:35
check "load step that plateaued" quiet "$(trend_verdict 2 5)"
mkwindow 50:20 54:20 57:20 59:20 60:20
check "climbing but fans never moved" quiet "$(trend_verdict 2 5)"
: > "$TREND_FILE"; echo "0,50,0,50,20" >> "$TREND_FILE"
check "partial window stays silent" quiet "$(trend_verdict 2 5)"
echo "ramp/alarm ordering (full cooling must arrive no later than the alarm):"
ramp_ordering_ok 8 5 && check "full at limit-8, alarm at limit-5" ok ok || check "full at limit-8, alarm at limit-5" ok bad
ramp_ordering_ok 5 5 && check "full and alarm coincide" ok ok || check "full and alarm coincide" ok bad
ramp_ordering_ok 2 5 && check "full at limit-2 is rejected" bad ok || check "full at limit-2 is rejected" bad bad
echo "alert rate limiting:"
local out
out=$( { raise_alert testkey "first" "body"; raise_alert testkey "second" "body"; } | grep -c '^MAIL\[' )
check "second alert inside cooldown suppressed" 1 "$out"
out=$(clear_alert testkey "recovered" | grep -c '^MAIL\[')
check "recovery notice sent once" 1 "$out"
out=$(clear_alert testkey "recovered" | grep -c '^MAIL\[')
check "recovery not repeated" 0 "$out"
rm -rf "$tmp"
[ $fail -eq 0 ] && echo "selftest OK" || { echo "selftest FAILED"; return 1; }
}
if [ "$1" = "selftest" ]; then selftest; exit $?; fi
#######################################
#######################################
# Check if the iDRAC host is set to 'local' or not then set the IDRAC_LOGIN_STRING accordingly
if [[ $IDRAC_HOST == "local" ]]; then
# Check that the Docker host IPMI device (the iDRAC) has been exposed to the Docker container
if [ ! -e "/dev/ipmi0" ] && [ ! -e "/dev/ipmi/0" ] && [ ! -e "/dev/ipmidev/0" ]; then
echo "/!\ Could not open device at /dev/ipmi0 or /dev/ipmi/0 or /dev/ipmidev/0, check that you added the device to your Docker container or stop using local mode. Exiting." >&2
exit 1
fi
IDRAC_LOGIN_STRING='open'
else
echo "iDRAC/IPMI username: $IDRAC_USERNAME"
IDRAC_LOGIN_STRING="lanplus -H $IDRAC_HOST -U $IDRAC_USERNAME -P $IDRAC_PASSWORD"
fi
#######################################
#######################################
# This script ONLY runs on Dell Servers
get_Dell_server_model
if [[ ! $SERVER_MANUFACTURER == "DELL" ]]; then
echo "/!\ Your server isn't a Dell product. Exiting." >&2
exit 1
fi
#######################################
#######################################
# Prepare, format and define initial variables
# Check if LOW_FAN_SPEED variable is in hexadecimal format. If not, convert it to hexadecimal
if [[ $LOW_FAN_SPEED == 0x* ]]; then
readonly DECIMAL_LOW_FAN_SPEED=$(printf '%d' $LOW_FAN_SPEED)
readonly HEXADECIMAL_LOW_FAN_SPEED=$LOW_FAN_SPEED
else
readonly DECIMAL_LOW_FAN_SPEED=$LOW_FAN_SPEED
readonly HEXADECIMAL_LOW_FAN_SPEED=$(convert_decimal_value_to_hexadecimal $LOW_FAN_SPEED)
fi
#######################################
#######################################
# Derive the CPU ramp from what iDRAC says this CPU can actually take.
CPU_LIMIT=$(cpu_upper_non_critical)
[ -n "$CPU_LIMIT" ] || CPU_LIMIT=$CPU_LIMIT_FALLBACK
LOW_TEMPERATURE_THRESHOLD=$((CPU_LIMIT - CPU_RAMP_LOW_OFFSET))
HIGH_TEMPERATURE_THRESHOLD=$((CPU_LIMIT - CPU_RAMP_HIGH_OFFSET))
echo "CPU limit ${CPU_LIMIT}c -> fan ramp ${LOW_TEMPERATURE_THRESHOLD}c..${HIGH_TEMPERATURE_THRESHOLD}c"
#######################################
#######################################
# Determine where to get CPU temps from iDRAC
# If server model is Gen 14 (*40) or newer
if [[ $SERVER_MODEL =~ .*[RT][[:space:]]?[0-9][4-9]0.* ]]; then
DELL_POWEREDGE_GEN_14_OR_NEWER=true
CPU1_TEMPERATURE_INDEX=2
CPU2_TEMPERATURE_INDEX=4
else
DELL_POWEREDGE_GEN_14_OR_NEWER=false
CPU1_TEMPERATURE_INDEX=1
CPU2_TEMPERATURE_INDEX=2
fi
#######################################
#######################################
# Check if sensors are present on the server
IS_EXHAUST_TEMPERATURE_SENSOR_PRESENT=true
IS_CPU2_TEMPERATURE_SENSOR_PRESENT=true
retrieve_cpu_temperatures $IS_EXHAUST_TEMPERATURE_SENSOR_PRESENT $IS_CPU2_TEMPERATURE_SENSOR_PRESENT
if [ -z "$EXHAUST_TEMPERATURE" ]; then
echo "No exhaust temperature sensor detected."
IS_EXHAUST_TEMPERATURE_SENSOR_PRESENT=false
fi
if [ -z "$CPU2_TEMPERATURE" ]; then
echo "No CPU2 temperature sensor detected."
IS_CPU2_TEMPERATURE_SENSOR_PRESENT=false
fi
#######################################
#######################################
# Read every drive's own temperature limit once - they do not change, and
# smartctl -x is far heavier than the -A used on the fast loop.
check_ramp_ordering
cache_disk_limits
echo "Monitoring $DISK_COUNT_EXPECTED disk(s)."
LAST_SMART_CHECK=0
APPLIED_FAN_SPEED=""
APPLIED_FAN_SPEED_AT=0
#######################################
#######################################
# One measurement pass: read everything, decide a fan speed, raise alarms.
one_pass() {
retrieve_cpu_temperatures $IS_EXHAUST_TEMPERATURE_SENSOR_PRESENT $IS_CPU2_TEMPERATURE_SENSOR_PRESENT
retrieve_gpu_temperature
retrieve_disk_temperatures
# Get highest CPU Temp and check if any of them are over heating
HIGHEST_CPU_TEMPERATURE=$CPU1_TEMPERATURE
if $IS_CPU2_TEMPERATURE_SENSOR_PRESENT; then
if [ "$CPU2_TEMPERATURE" -gt "$CPU1_TEMPERATURE" ]; then
HIGHEST_CPU_TEMPERATURE=$CPU2_TEMPERATURE
fi
fi
if [ "$HIGHEST_CPU_TEMPERATURE" -gt "$CPU_TEMPERATURE_THRESHOLD" ]; then
# CPU is overheating - hand cooling back to Dell and say so out loud.
apply_Dell_fan_control_profile
IS_DELL_FAN_CONTROL_PROFILE_APPLIED=true
COMMENT="CPU temperature is too high, Dell default dynamic fan control profile applied for safety"
raise_alert cpu "Fan control disengaged on $(hostname)" \
"CPU at ${HIGHEST_CPU_TEMPERATURE}c exceeds the ${CPU_TEMPERATURE_THRESHOLD}c threshold. Dell's default profile has been restored."
record_sample "$HIGHEST_CPU_TEMPERATURE" "$GPU_TEMPERATURE" "$HOTTEST_DISK_TEMPERATURE" 0
return
fi
IS_DELL_FAN_CONTROL_PROFILE_APPLIED=false
clear_alert cpu "Fan control restored on $(hostname)"
# CPU and GPU share one curve; disks are interpolated against their own limits
# in retrieve_disk_temperatures. Each source asks for a speed, loudest wins -
# temperatures from different classes of hardware are not comparable directly.
CPU_GPU_FAN_SPEED=$(calculate_interpolated_fan_speed "$HIGHEST_CPU_TEMPERATURE" \
$LOW_TEMPERATURE_THRESHOLD $HIGH_TEMPERATURE_THRESHOLD $LOW_FAN_SPEED $HIGH_FAN_SPEED)
# The GPU has its own limit and so gets its own ramp, rather than being
# compared against the CPU's on a shared scale.
if [ "$GPU_TEMPERATURE" -gt 0 ]; then
local gpu_speed
gpu_speed=$(calculate_interpolated_fan_speed "$GPU_TEMPERATURE" \
$((GPU_TEMPERATURE_THRESHOLD - CPU_RAMP_LOW_OFFSET)) $GPU_TEMPERATURE_THRESHOLD \
$LOW_FAN_SPEED $HIGH_FAN_SPEED)
[ "$gpu_speed" -gt "$CPU_GPU_FAN_SPEED" ] && CPU_GPU_FAN_SPEED=$gpu_speed
fi
DECIMAL_CURRENT_FAN_SPEED=$CPU_GPU_FAN_SPEED
FAN_SPEED_DRIVER="cpu/gpu"
if [ "$DISK_FAN_SPEED" -gt "$DECIMAL_CURRENT_FAN_SPEED" ]; then
DECIMAL_CURRENT_FAN_SPEED=$DISK_FAN_SPEED
FAN_SPEED_DRIVER="disk"
fi
# Hold the current speed unless the change is worth making - see the deadband
# note in monitor.sh. Chasing every degree is what made this louder than stock.
local target now due=0 held=""
now=$(date +%s)
[ $((now - APPLIED_FAN_SPEED_AT)) -ge "$FAN_REASSERT_INTERVAL" ] && due=1
target=$(settle_fan_speed "$DECIMAL_CURRENT_FAN_SPEED" "$APPLIED_FAN_SPEED" "$due")
[ "$target" = "$DECIMAL_CURRENT_FAN_SPEED" ] || held=" held, want $DECIMAL_CURRENT_FAN_SPEED"
if [ "$target" != "$APPLIED_FAN_SPEED" ] || [ "$due" = 1 ]; then
apply_user_fan_control "$target"
APPLIED_FAN_SPEED=$target
APPLIED_FAN_SPEED_AT=$now
fi
COMMENT="CPU1:$CPU1_TEMPERATURE | CPU2:$CPU2_TEMPERATURE | GPU:$GPU_TEMPERATURE | Inlet:$INLET_TEMPERATURE | Exhaust:$EXHAUST_TEMPERATURE | Disk:$HOTTEST_DISK_TEMPERATURE/$HOTTEST_DISK_LIMIT($(basename $HOTTEST_DISK_DEVICE)) | Fan Speed:$APPLIED_FAN_SPEED($FAN_SPEED_DRIVER$held)"
# Record what the fans are actually doing, not what was merely requested.
record_sample "$HIGHEST_CPU_TEMPERATURE" "$GPU_TEMPERATURE" "$HOTTEST_DISK_TEMPERATURE" "$APPLIED_FAN_SPEED"
check_alarms
}
check_alarms() {
# A drive within DISK_ALARM_OFFSET of its own limit: fans are already flat out
# for it and it is still climbing.
if [ "$HOTTEST_DISK_TEMPERATURE" -gt 0 ] &&
[ "$HOTTEST_DISK_TEMPERATURE" -ge $((HOTTEST_DISK_LIMIT - DISK_ALARM_OFFSET)) ]; then
raise_alert disk_temp "Disk over temperature on $(hostname)" \
"$HOTTEST_DISK_DEVICE at ${HOTTEST_DISK_TEMPERATURE}c, its own limit is ${HOTTEST_DISK_LIMIT}c. Fans at ${DECIMAL_CURRENT_FAN_SPEED}%."
else
clear_alert disk_temp "Disk temperature normal on $(hostname)"
fi
if [ "$GPU_TEMPERATURE" -gt "$GPU_TEMPERATURE_THRESHOLD" ]; then
raise_alert gpu "GPU over temperature on $(hostname)" \
"GPU at ${GPU_TEMPERATURE}c exceeds the ${GPU_TEMPERATURE_THRESHOLD}c threshold."
else
clear_alert gpu "GPU temperature normal on $(hostname)"
fi
# Climbing is only a fault if the fans are not answering it. A busy machine
# heats up and plateaus; a dying fan or blocked intake keeps climbing. See
# trend_verdict() - this stays quiet through hours of legitimate load.
local cpu_verdict disk_verdict cpu_rise disk_rise minutes
cpu_verdict=$(trend_verdict 2 5)
disk_verdict=$(trend_verdict 4 5)
cpu_rise=$(rise_over_window 2)
disk_rise=$(rise_over_window 4)
minutes=$((TREND_SAMPLES * CHECK_INTERVAL / 60))
if [ "$cpu_verdict" = no_headroom ] || [ "$disk_verdict" = no_headroom ]; then
raise_alert trend "Out of cooling headroom on $(hostname)" \
"Fans are at ${HIGH_FAN_SPEED}% and temperatures are still climbing. Over the last ${minutes} minutes: CPU +${cpu_rise}c, hottest disk +${disk_rise}c ($HOTTEST_DISK_DEVICE at ${HOTTEST_DISK_TEMPERATURE}c of ${HOTTEST_DISK_LIMIT}c). There is no cooling left to apply."
elif [ "$cpu_verdict" = not_converging ] || [ "$disk_verdict" = not_converging ]; then
raise_alert trend "Temperature outrunning the fans on $(hostname)" \
"Temperatures are climbing and the fans are ramping with them, but the rise is not slowing. Over the last ${minutes} minutes: CPU +${cpu_rise}c, hottest disk +${disk_rise}c, fans now ${APPLIED_FAN_SPEED}%. Check airflow, intake and fan health."
else
clear_alert trend "Temperature stabilised on $(hostname)"
fi
# A drive that stops answering is either asleep or gone. Worth knowing which.
if [ "$DISKS_READ" -lt "$DISK_COUNT_EXPECTED" ]; then
raise_alert disk_count "Disk missing on $(hostname)" \
"Read $DISKS_READ of $DISK_COUNT_EXPECTED disks. A drive is in standby, has dropped off the bus, or has failed."
else
clear_alert disk_count "All disks reporting on $(hostname)"
fi
# SMART is expensive across two dozen drives - hourly, not every loop.
local now; now=$(date +%s)
if [ $((now - LAST_SMART_CHECK)) -ge "$SMART_CHECK_INTERVAL" ]; then
LAST_SMART_CHECK=$now
check_disk_health
fi
}
#######################################
case "${1:-run}" in
once)
one_pass
echo "$COMMENT"
log_line "$COMMENT"
;;
disks)
for dev in $(disk_devices); do
printf '%-12s %-4s limit=%sc\n' "$dev" "$(disk_temperature "$dev")c" "$(disk_limit "$dev")"
done
;;
run)
while true; do
sleep $CHECK_INTERVAL &
SLEEP_PROCESS_PID=$!
one_pass
log_line "$COMMENT"
wait $SLEEP_PROCESS_PID
done
;;
*)
echo "usage: $0 [run|once|disks|selftest]" >&2
exit 1
;;
esac