From 099d92c81371214f4a0b26519f965a8dc3b0cde7 Mon Sep 17 00:00:00 2001 From: df Date: Sat, 15 Aug 2026 01:15:05 +0800 Subject: [PATCH 1/2] feat(temperature): normalize sources and add Linux disk support Introduce a shared validate, collect, and adjust contract for every temperature provider. Add smartctl-based Linux disk readings and SSH-based remote NVIDIA GPU readings so heterogeneous sensors can feed one decision chain. Update container dependencies, Compose guidance, TUI settings, tests, and bilingual operations documentation, including the TrueNAS CD6 deployment path. --- .env.example | 31 ++- .gitattributes | 4 + Dockerfile | 4 +- README.md | 60 +++-- README.zh-TW.md | 60 +++-- USAGE_GUIDE.md | 34 ++- docker-compose.yml | 4 + docs/TEMPERATURE_SOURCES.md | 88 +++++++ docs/TROUBLESHOOTING.md | 32 ++- src/FanControlWithEsxiSmart.sh | 426 ++++++++++++++++++++++++++------- src/fan-control-tui.sh | 97 ++++++-- tests/fan-control.test.sh | 133 +++++++++- tests/tui.test.sh | 12 + 13 files changed, 828 insertions(+), 157 deletions(-) create mode 100644 .gitattributes create mode 100644 docs/TEMPERATURE_SOURCES.md diff --git a/.env.example b/.env.example index 7c030f7..3118b0a 100644 --- a/.env.example +++ b/.env.example @@ -32,10 +32,13 @@ RESTORE_AUTO_ON_EXIT=true # ----------------------------------------------------------------------------- # Temperature sources # ----------------------------------------------------------------------------- -# Supported values: esxi, idrac, gpu. Use comma-separated values for hybrid mode. +# Supported values: esxi, idrac, gpu, linux_disk, remote_gpu. +# Use comma-separated values for hybrid mode. # esxi reads one NVMe SMART temperature through ESXi SSH. # idrac reads all iDRAC temperature sensors through ipmitool. -# gpu reads NVIDIA GPU temperatures through nvidia-smi. +# gpu reads NVIDIA GPU temperatures on the controller host. +# linux_disk reads one or more local Linux disks through smartctl JSON output. +# remote_gpu reads NVIDIA GPU temperatures from one or more Linux VMs over SSH. TEMPERATURE_SOURCES=esxi # Backward-compatible GPU switch. If true, gpu is appended to TEMPERATURE_SOURCES. @@ -56,10 +59,34 @@ ESXI_USERNAME=root ESXI_PASSWORD=change-me ESXI_SSH_KEY= ESXI_SSH_PORT=22 +# Shared by the ESXi and remote GPU SSH transports. SSH_CONNECT_TIMEOUT=10 SSH_STRICT_HOST_KEY_CHECKING=accept-new DRIVE_DEVICE=t10.NVMe____replace_with_your_device_identifier +# ----------------------------------------------------------------------------- +# Local Linux disk source settings +# Required only when TEMPERATURE_SOURCES includes linux_disk. +# ----------------------------------------------------------------------------- +# Use controller device nodes (for NVMe, /dev/nvmeN is preferred). In Docker, +# map every listed host device into the container with Compose `devices`. +LINUX_DISK_DEVICES=/dev/nvme1 +LINUX_DISK_TEMP_OFFSET=0 +# Set standby to avoid waking sleeping SATA/SAS disks. Use never for NVMe. +LINUX_DISK_NOCHECK=never + +# ----------------------------------------------------------------------------- +# Remote NVIDIA GPU source settings +# Required only when TEMPERATURE_SOURCES includes remote_gpu. +# ----------------------------------------------------------------------------- +# Every host must provide nvidia-smi. One SSH identity is shared by all hosts. +REMOTE_GPU_HOSTS=192.0.2.31,192.0.2.32 +REMOTE_GPU_USERNAME=root +REMOTE_GPU_PASSWORD= +REMOTE_GPU_SSH_KEY=/run/secrets/gpu_vms_ed25519 +REMOTE_GPU_SSH_PORT=22 +REMOTE_GPU_TEMP_OFFSET=15 + # ----------------------------------------------------------------------------- # Fan curve # ----------------------------------------------------------------------------- diff --git a/.gitattributes b/.gitattributes new file mode 100644 index 0000000..3981d63 --- /dev/null +++ b/.gitattributes @@ -0,0 +1,4 @@ +*.sh text eol=lf +Dockerfile text eol=lf +*.yml text eol=lf +*.yaml text eol=lf diff --git a/Dockerfile b/Dockerfile index e0fab70..2f95ecc 100644 --- a/Dockerfile +++ b/Dockerfile @@ -2,7 +2,7 @@ ARG BASE_IMAGE=nvidia/cuda:12.9.0-runtime-ubuntu24.04 FROM ${BASE_IMAGE} LABEL org.opencontainers.image.title="iDRAC Fan Speed Control" \ - org.opencontainers.image.description="Automatic Dell iDRAC fan control using IPMI with ESXi, iDRAC, and optional NVIDIA GPU temperature sources" \ + org.opencontainers.image.description="Automatic Dell iDRAC fan control with pluggable disk, iDRAC, and NVIDIA GPU temperature sources" \ org.opencontainers.image.source="https://github.com/DF-wu/iDRACFanSpeedControl" ENV DEBIAN_FRONTEND=noninteractive \ @@ -13,8 +13,10 @@ RUN apt-get update && apt-get install -y --no-install-recommends \ ca-certificates \ coreutils \ ipmitool \ + jq \ openssh-client \ procps \ + smartmontools \ sshpass \ tini \ && rm -rf /var/lib/apt/lists/* diff --git a/README.md b/README.md index 8d02a0c..e314cc0 100644 --- a/README.md +++ b/README.md @@ -6,7 +6,7 @@ [![Container](https://img.shields.io/badge/container-GHCR-blue)](https://github.com/DF-wu/iDRACFanSpeedControl/pkgs/container/idrac-fan-control) [![License](https://img.shields.io/badge/license-MIT-green)](LICENSE) -A fan controller for Dell PowerEdge servers. It sets the fan duty cycle through iDRAC IPMI OEM raw commands, then selects a fan-curve level using one or more temperature sources: ESXi NVMe SMART, iDRAC temperature sensors, or NVIDIA GPUs. +A fan controller for Dell PowerEdge servers. It sets the fan duty cycle through iDRAC IPMI OEM raw commands, then selects a fan-curve level from pluggable temperature sources: ESXi NVMe SMART, iDRAC sensors, local Linux disks, and local or remote NVIDIA GPUs. > [!CAUTION] > This program temporarily overrides Dell's automatic fan control. During initial setup, keep the iDRAC Web UI or a physical console available. Verify `manual`, `restore`, and `diagnose` before running `auto` for an extended period. If the server overheats, readings become unreliable, or behavior is abnormal, run `restore` immediately and stop the container. @@ -21,11 +21,12 @@ flowchart LR CTRL --> IPMI[iDRAC / IPMI\nfan raw command] CTRL --> ESXI[ESXi\nesxcli SMART] CTRL --> SDR[iDRAC\nTemperature SDR] - CTRL --> GPU[NVIDIA\nnvidia-smi] + CTRL --> DISK[Linux disks\nsmartctl JSON] + CTRL --> GPU[Local/remote NVIDIA\nnvidia-smi] CTRL --> LOG[logs/fan_control.log\nhealthcheck] ``` -The controller sends fan commands only to iDRAC; all temperature sources are read-only. When multiple sources are enabled, it uses the highest valid decision temperature. The GPU reading is adjusted by `GPU_TEMP_OFFSET` first, preventing the GPU package temperature from driving chassis fans unnecessarily fast. +The controller sends fan commands only to iDRAC; all temperature sources are read-only. Every provider implements the same `validate`, `collect`, and `adjust` interface. When multiple sources are enabled, the controller uses the highest valid adjusted temperature. GPU and disk offsets let unlike sensors share one fan curve without source-specific decision code. ![iDRAC IPMI over LAN settings](images/image.png) @@ -50,12 +51,14 @@ The main menu looks like this. Each submenu writes changes back to `.env`, and t 1) Quick setup wizard 6) Safety, timing, and logging 2) iDRAC / IPMI settings 7) Review redacted configuration - 3) Temperature source 8) Validate configuration - 4) ESXi NVMe source 9) Run read-only diagnostics - 5) Fan curve 0) Save and exit + 3) Temperature source 8) Safety, timing, and logging + 4) ESXi NVMe source 9) Review redacted configuration + 5) Local Linux disks 10) Validate configuration + 6) Remote NVIDIA GPUs 11) Run read-only diagnostics + 7) Fan curve 0) Save and exit ``` -For your first setup, select `iDRAC sensors only` and leave ESXi and GPU sources disabled. After completing the TUI, run: +For your first setup, select `iDRAC sensors only` and leave external sources disabled. After completing the TUI, run: ```bash make validate @@ -125,7 +128,7 @@ src/FanControlWithEsxiSmart.sh diagnose src/FanControlWithEsxiSmart.sh config ``` -Local execution requires `bash`, `coreutils`, `ipmitool`, and `timeout`. ESXi password authentication also requires `sshpass`. The Docker image includes these dependencies. +Local execution requires `bash`, `coreutils`, `ipmitool`, and `timeout`. Linux disk mode also requires `smartctl` and `jq`; SSH password authentication requires `sshpass`. The Docker image includes these dependencies. ## Safe startup sequence @@ -157,13 +160,15 @@ Before unattended operation, confirm each item: ## Temperature sources and control decisions -`TEMPERATURE_SOURCES` is a comma-separated list containing `esxi`, `idrac`, and/or `gpu`. Sources can be combined, and the controller retains each source label for logging and diagnostics. +`TEMPERATURE_SOURCES` is a comma-separated list containing `esxi`, `idrac`, `gpu`, `linux_disk`, and/or `remote_gpu`. Sources can be combined, and the controller retains each source label for logging and diagnostics. See [Temperature source interface](docs/TEMPERATURE_SOURCES.md) for the extension contract and deployment examples. | Source | Reading | Requirements | Behavior on failure | | --- | --- | --- | --- | | `idrac` | Readable sensors from `ipmitool sdr type Temperature` | iDRAC IPMI | Marks this source as failed | | `esxi` | `esxcli storage core device smart get` for the configured NVMe device | SSH and `DRIVE_DEVICE` | Marks this source as failed | -| `gpu` | Temperature of each GPU reported by `nvidia-smi` | NVIDIA Container Toolkit and driver | Marks this source as failed | +| `gpu` | Temperature of each local GPU reported by `nvidia-smi` | NVIDIA Container Toolkit and driver | Marks this source as failed | +| `linux_disk` | SMART temperature for each configured Linux device | `smartctl`, `jq`, and device access | Fails unreadable devices; keeps other valid disks | +| `remote_gpu` | Every GPU reported by `nvidia-smi` on each configured VM | SSH credentials and remote NVIDIA driver | Fails unreachable hosts; keeps other valid VMs | ```mermaid flowchart TD @@ -191,7 +196,7 @@ The default curve is shown below. `validate` ensures that thresholds are strictl | `critical` | `>=80°C` | 60% | | `failsafe` | All sources fail | 70% | -Moving to a lower fan level requires the temperature to fall by `HYSTERESIS` degrees below the relevant threshold, preventing repeated speed changes near a boundary. Moving to a higher level is immediate. The adjusted GPU value is `GPU_TEMP - GPU_TEMP_OFFSET`, with a minimum of 0°C. +Moving to a lower fan level requires the temperature to fall by `HYSTERESIS` degrees below the relevant threshold, preventing repeated speed changes near a boundary. Moving to a higher level is immediate. Each source owns its adjustment: local and remote GPUs subtract their configured offset, Linux disks subtract `LINUX_DISK_TEMP_OFFSET`, and all results have a minimum of 0°C. ## Configuration reference @@ -214,7 +219,7 @@ Moving to a lower fan level requires the temperature to fall by `HYSTERESIS` deg | Variable | Default | Description | | --- | --- | --- | -| `TEMPERATURE_SOURCES` | `esxi` | Comma-separated list of `esxi,idrac,gpu` | +| `TEMPERATURE_SOURCES` | `esxi` | Comma-separated list of registered source IDs | | `WITH_GPU_TEMP` | `false` | Legacy compatibility switch; `true` appends `gpu` | | `GPU_TEMP_OFFSET` | `15` | Offset subtracted from GPU temperature | | `ESXI_HOST` / `ESXI_USERNAME` | Empty / `root` | ESXi SSH target | @@ -224,6 +229,13 @@ Moving to a lower fan level requires the temperature to fall by `HYSTERESIS` deg | `SSH_CONNECT_TIMEOUT` | `10` | SSH connection timeout in seconds | | `SSH_STRICT_HOST_KEY_CHECKING` | `accept-new` | `yes`, `no`, `ask`, or `accept-new` | | `DRIVE_DEVICE` | Empty | Full ID returned by `esxcli storage core device list` | +| `LINUX_DISK_DEVICES` | Empty | Comma-separated Linux device paths, such as `/dev/nvme1,/dev/sdb` | +| `LINUX_DISK_TEMP_OFFSET` | `0` | Offset subtracted from local disk temperatures | +| `LINUX_DISK_NOCHECK` | `never` | smartctl power-mode check; `standby` avoids waking sleeping disks | +| `REMOTE_GPU_HOSTS` | Empty | Comma-separated Linux VM hostnames or addresses | +| `REMOTE_GPU_USERNAME` / `REMOTE_GPU_SSH_PORT` | `root` / `22` | SSH identity shared by remote GPU hosts | +| `REMOTE_GPU_PASSWORD` / `REMOTE_GPU_SSH_KEY` | Empty / Empty | Remote GPU SSH authentication; keys are preferred | +| `REMOTE_GPU_TEMP_OFFSET` | `15` | Offset subtracted from remote GPU temperatures | | `IDRAC_SENSOR_INCLUDE_REGEX` | Empty | awk regex used to keep matching sensor names only | | `IDRAC_SENSOR_EXCLUDE_REGEX` | `no reading\|disabled\|not readable` | Excludes invalid SDR entries | @@ -241,7 +253,7 @@ Moving to a lower fan level requires the temperature to fall by `HYSTERESIS` deg | `LOG_LEVEL` | `INFO` | `DEBUG` adds command and source details but never logs passwords | | `HEALTHCHECK_MAX_AGE` | `0` | `0` means `CHECK_INTERVAL*3 + COMMAND_TIMEOUT` | -## Docker deployment and GPU support +## Docker deployment, Linux disks, and GPUs The Compose configuration uses the GHCR image, host networking, and a `./logs` volume by default: @@ -274,6 +286,17 @@ If GPU support is unnecessary, build locally without CUDA to reduce the image si docker build --build-arg BASE_IMAGE=ubuntu:24.04 -t idrac-fan-control:local . ``` +Linux disk mode needs a device mapping for every entry in `LINUX_DISK_DEVICES`. For example: + +```yaml +services: + idrac-fan-control: + devices: + - /dev/nvme1:/dev/nvme1 +``` + +For a TrueNAS CD6 plus NVIDIA GPUs in other VMs, use `TEMPERATURE_SOURCES=linux_disk,remote_gpu`, map the CD6 controller device, and mount a read-only SSH key for the GPU VMs. The complete example is in [docs/TEMPERATURE_SOURCES.md](docs/TEMPERATURE_SOURCES.md). + ## Debugging, logs, and troubleshooting Example normal log entry: @@ -312,7 +335,7 @@ docker compose run --rm idrac-fan-control diagnose ## Security and backups - Never commit `.env`, `logs/`, private keys, or incident dumps. The TUI sets configuration file permissions to `600`. -- Compose mounts the gitignored `./secrets` directory read-only at `/run/secrets`. Use the in-container path for ESXi keys, such as `/run/secrets/esxi_ed25519`. +- Compose mounts the gitignored `./secrets` directory read-only at `/run/secrets`. Use in-container paths such as `/run/secrets/esxi_ed25519` or `/run/secrets/gpu_vms_ed25519`. - Prefer `ESXI_SSH_KEY`. When password authentication is required, the controller uses the `SSHPASS` environment variable with `sshpass -e`, keeping the password out of argv. - The iDRAC password is supplied through the `IPMI_PASSWORD` environment variable and `ipmitool -E`. The `config` command and TUI review show only whether it is set and its character count. - Keep iDRAC and ESXi on an isolated management network. Never expose IPMI over LAN to the public internet. @@ -331,13 +354,13 @@ Copy the complete identifier into `DRIVE_DEVICE`; do not shorten it. If the SMAR ## Testing and quality gates ```bash -make test # bash -n + 37 core assertions + 10 TUI assertions +make test # bash -n + 52 core assertions + 14 TUI assertions make validate # Check the current .env with Docker dependencies; does not change fans make validate-example # DRY_RUN smoke test that does not require .env make docker-build ``` -Tests cover source normalization, SDR parsing, GPU offset, hysteresis, fail-safe behavior, curve/port/regex validation, credential redaction, health freshness, diagnostic previews, and safe round-tripping of TUI configuration files. Real hardware, ESXi SSH, and GPUs must still be verified with `diagnose` on your management network. +Tests cover the source interface, Linux SMART collection, multi-host remote GPUs, source offsets, SDR parsing, hysteresis, fail-safe behavior, configuration validation, credential redaction, health freshness, diagnostics, and safe TUI configuration round-trips. Real hardware, SSH targets, disks, iDRAC, and GPUs must still be verified with `diagnose` on your management network. ## Project structure @@ -351,6 +374,7 @@ Tests cover source normalization, SDR parsing, GPU offset, hysteresis, fail-safe │ ├── fan-control.test.sh # Core logic and safety tests │ └── tui.test.sh # .env parser/writer tests ├── docs/ +│ ├── TEMPERATURE_SOURCES.md # Source interface and deployment examples │ └── TROUBLESHOOTING.md # Symptom-based troubleshooting guide ├── images/image.png # iDRAC IPMI settings screenshot ├── .env.example # Fully commented configuration template @@ -364,10 +388,10 @@ Tests cover source normalization, SDR parsing, GPU offset, hysteresis, fail-safe ## Compatibility and limitations -The project primarily targets Dell PowerEdge R730/R730xd-class systems with iDRAC 8 OEM fan raw commands. Other generations may be compatible, but identical behavior must not be assumed. Complete `manual`, `restore`, and `diagnose` checks before unattended operation. Actual iDRAC, ESXi, and GPU sensor names and permissions vary by firmware and driver, so rely on diagnostic output from your own environment. +The project primarily targets Dell PowerEdge R730/R730xd-class systems with iDRAC 8 OEM fan raw commands. Other generations may be compatible, but identical behavior must not be assumed. Complete `manual`, `restore`, and `diagnose` checks before unattended operation. Sensor names, disk permissions, SSH access, and driver behavior vary by environment, so rely on diagnostic output from your own deployment. ## License MIT. See [LICENSE](LICENSE). -Documentation last reviewed: July 18, 2026. +Documentation last reviewed: August 15, 2026. diff --git a/README.zh-TW.md b/README.zh-TW.md index 585f537..f7870f5 100644 --- a/README.zh-TW.md +++ b/README.zh-TW.md @@ -6,7 +6,7 @@ [![Container](https://img.shields.io/badge/container-GHCR-blue)](https://github.com/DF-wu/iDRACFanSpeedControl/pkgs/container/idrac-fan-control) [![License](https://img.shields.io/badge/license-MIT-green)](LICENSE) -這是一個給 Dell PowerEdge 使用的風扇控制器。它透過 iDRAC 的 IPMI OEM raw command 設定 fan duty cycle,再用 ESXi NVMe SMART、iDRAC temperature sensor、NVIDIA GPU 中任一個或多個來源決定風扇曲線。 +這是一個給 Dell PowerEdge 使用的風扇控制器。它透過 iDRAC 的 IPMI OEM raw command 設定 fan duty cycle,再以可插拔溫度來源決定風扇曲線:ESXi NVMe SMART、iDRAC sensor、本機 Linux 磁碟,以及本機或遠端 NVIDIA GPU。 > [!CAUTION] > 這個程式會暫時覆寫 Dell 原廠風扇控制。第一次設定請保留 iDRAC Web UI 或實體主控台,先用 `manual`、`restore`、`diagnose` 驗證,再讓 `auto` 長時間執行。任何過熱、讀值不可信或行為異常時,立即執行 `restore` 並停止容器。 @@ -21,11 +21,12 @@ flowchart LR CTRL --> IPMI[iDRAC / IPMI\nfan raw command] CTRL --> ESXI[ESXi\nesxcli SMART] CTRL --> SDR[iDRAC\nTemperature SDR] - CTRL --> GPU[NVIDIA\nnvidia-smi] + CTRL --> DISK[Linux 磁碟\nsmartctl JSON] + CTRL --> GPU[本機/遠端 NVIDIA\nnvidia-smi] CTRL --> LOG[logs/fan_control.log\nhealthcheck] ``` -控制器只會對 iDRAC 發送風扇命令;溫度來源全部是讀取。多來源時採用最高的有效決策溫度,GPU 會先減去 `GPU_TEMP_OFFSET`,避免 GPU 的封裝溫度直接把機箱風扇推到不必要的高轉速。 +控制器只會對 iDRAC 發送風扇命令;溫度來源全部是讀取。每個 provider 都實作同一組 `validate`、`collect`、`adjust` 介面。多來源時採用最高的有效 adjusted temperature;GPU 與磁碟可各自套用 offset,不需在決策流程加入來源特例。 ![iDRAC IPMI over LAN 設定畫面](images/image.png) @@ -50,12 +51,14 @@ make tui 1) Quick setup wizard 6) Safety, timing, and logging 2) iDRAC / IPMI settings 7) Review redacted configuration - 3) Temperature source 8) Validate configuration - 4) ESXi NVMe source 9) Run read-only diagnostics - 5) Fan curve 0) Save and exit + 3) Temperature source 8) Safety, timing, and logging + 4) ESXi NVMe source 9) Review redacted configuration + 5) Local Linux disks 10) Validate configuration + 6) Remote NVIDIA GPUs 11) Run read-only diagnostics + 7) Fan curve 0) Save and exit ``` -第一次建議選 `iDRAC sensors only`,先不要把 ESXi 或 GPU 加入決策。TUI 完成後,依序執行: +第一次建議選 `iDRAC sensors only`,先不要把外部來源加入決策。TUI 完成後,依序執行: ```bash make validate @@ -125,7 +128,7 @@ src/FanControlWithEsxiSmart.sh diagnose src/FanControlWithEsxiSmart.sh config ``` -本機執行需要 `bash`、`coreutils`、`ipmitool`、`timeout`,ESXi password 模式另需 `sshpass`;Docker image 已包含這些依賴。 +本機執行需要 `bash`、`coreutils`、`ipmitool`、`timeout`。Linux 磁碟模式另需 `smartctl` 與 `jq`;SSH password 模式另需 `sshpass`。Docker image 已包含這些依賴。 ## 安全啟動順序 @@ -157,13 +160,15 @@ sequenceDiagram ## 溫度來源與控制決策 -`TEMPERATURE_SOURCES` 使用逗號分隔,可填 `esxi`、`idrac`、`gpu`。來源可以混合;控制器會保留每個來源的 label 供 log 與診斷追蹤。 +`TEMPERATURE_SOURCES` 使用逗號分隔,可填 `esxi`、`idrac`、`gpu`、`linux_disk`、`remote_gpu`。來源可以混合;控制器會保留每個來源的 label 供 log 與診斷追蹤。擴充介面與部署範例請見 [Temperature source interface](docs/TEMPERATURE_SOURCES.md)。 | 來源 | 讀取內容 | 必要條件 | 失敗時的行為 | | --- | --- | --- | --- | | `idrac` | `ipmitool sdr type Temperature` 中可讀的 sensor | iDRAC IPMI | 該來源標記失敗 | | `esxi` | 指定 NVMe 的 `esxcli storage core device smart get` | SSH、`DRIVE_DEVICE` | 該來源標記失敗 | -| `gpu` | `nvidia-smi` 的每張 GPU 溫度 | NVIDIA Container Toolkit/驅動 | 該來源標記失敗 | +| `gpu` | 本機 `nvidia-smi` 的每張 GPU 溫度 | NVIDIA Container Toolkit/驅動 | 該來源標記失敗 | +| `linux_disk` | 每個指定 Linux device 的 SMART 溫度 | `smartctl`、`jq`、device 權限 | 單一磁碟失敗,保留其他有效讀值 | +| `remote_gpu` | 每台指定 VM 上 `nvidia-smi` 回報的所有 GPU | SSH 與遠端 NVIDIA driver | 單一 VM 失敗,保留其他有效讀值 | ```mermaid flowchart TD @@ -191,7 +196,7 @@ flowchart TD | `critical` | `>=80°C` | 60% | | `failsafe` | 所有來源失敗 | 70% | -降檔會等待 `HYSTERESIS` 度,避免溫度在閾值附近來回跳速;升檔不延遲。GPU 的 adjusted 值為 `GPU_TEMP - GPU_TEMP_OFFSET`,最低不會小於 0°C。 +降檔會等待 `HYSTERESIS` 度,避免溫度在閾值附近來回跳速;升檔不延遲。每個來源自行調整讀值:本機與遠端 GPU 分別扣除各自 offset,Linux 磁碟扣除 `LINUX_DISK_TEMP_OFFSET`,最低都不會小於 0°C。 ## 設定參考 @@ -214,7 +219,7 @@ flowchart TD | 變數 | 預設 | 說明 | | --- | --- | --- | -| `TEMPERATURE_SOURCES` | `esxi` | `esxi,idrac,gpu` 的逗號清單 | +| `TEMPERATURE_SOURCES` | `esxi` | 已註冊 source ID 的逗號清單 | | `WITH_GPU_TEMP` | `false` | 舊版相容開關;true 會附加 `gpu` | | `GPU_TEMP_OFFSET` | `15` | GPU 溫度的補償度數 | | `ESXI_HOST` / `ESXI_USERNAME` | 空 / `root` | ESXi SSH 目標 | @@ -224,6 +229,13 @@ flowchart TD | `SSH_CONNECT_TIMEOUT` | `10` | SSH 建立連線上限(秒) | | `SSH_STRICT_HOST_KEY_CHECKING` | `accept-new` | `yes`、`no`、`ask` 或 `accept-new` | | `DRIVE_DEVICE` | 空 | `esxcli storage core device list` 找到的完整 ID | +| `LINUX_DISK_DEVICES` | 空 | Linux device path 逗號清單,例如 `/dev/nvme1,/dev/sdb` | +| `LINUX_DISK_TEMP_OFFSET` | `0` | 從本機磁碟溫度扣除的 offset | +| `LINUX_DISK_NOCHECK` | `never` | smartctl power mode check;`standby` 可避免喚醒休眠磁碟 | +| `REMOTE_GPU_HOSTS` | 空 | Linux GPU VM hostname 或 IP 逗號清單 | +| `REMOTE_GPU_USERNAME` / `REMOTE_GPU_SSH_PORT` | `root` / `22` | 遠端 GPU 主機共用的 SSH identity | +| `REMOTE_GPU_PASSWORD` / `REMOTE_GPU_SSH_KEY` | 空 / 空 | 遠端 GPU SSH 認證;建議使用 key | +| `REMOTE_GPU_TEMP_OFFSET` | `15` | 從遠端 GPU 溫度扣除的 offset | | `IDRAC_SENSOR_INCLUDE_REGEX` | 空 | 只保留符合 sensor 名稱的 awk regex | | `IDRAC_SENSOR_EXCLUDE_REGEX` | `no reading\|disabled\|not readable` | 排除無效 SDR | @@ -241,7 +253,7 @@ flowchart TD | `LOG_LEVEL` | `INFO` | `DEBUG` 會增加命令與來源細節,絕不列密碼 | | `HEALTHCHECK_MAX_AGE` | `0` | 0 代表 `CHECK_INTERVAL*3 + COMMAND_TIMEOUT` | -## Docker 部署與 GPU +## Docker 部署、Linux 磁碟與 GPU Compose 預設使用 GHCR image、host network 與 `./logs` volume: @@ -274,6 +286,17 @@ docker compose run --rm idrac-fan-control diagnose docker build --build-arg BASE_IMAGE=ubuntu:24.04 -t idrac-fan-control:local . ``` +Linux 磁碟模式必須將 `LINUX_DISK_DEVICES` 的每個 device 映射進容器,例如: + +```yaml +services: + idrac-fan-control: + devices: + - /dev/nvme1:/dev/nvme1 +``` + +若要在 TrueNAS 將 CD6 與其他 VM 的 NVIDIA GPU 納入同一決策鏈,設定 `TEMPERATURE_SOURCES=linux_disk,remote_gpu`、映射 CD6 controller device,並以唯讀 volume 掛載 GPU VM SSH key。完整範例請見 [docs/TEMPERATURE_SOURCES.md](docs/TEMPERATURE_SOURCES.md)。 + ## Debug、log 與故障排除 正常 log 範例: @@ -312,7 +335,7 @@ docker compose run --rm idrac-fan-control diagnose ## 安全與備份 - `.env`、`logs/`、私鑰與 incident dump 都不應提交到 Git;TUI 會將設定檔權限設為 `600`。 -- Compose 會以 read-only 掛載 gitignored 的 `./secrets` 到 `/run/secrets`;ESXi key 請使用容器內路徑(例如 `/run/secrets/esxi_ed25519`)。 +- Compose 會以 read-only 掛載 gitignored 的 `./secrets` 到 `/run/secrets`;請使用 `/run/secrets/esxi_ed25519` 或 `/run/secrets/gpu_vms_ed25519` 這類容器內路徑。 - 優先使用 `ESXI_SSH_KEY`;若必須用 password,控制器會以 `SSHPASS` environment 搭配 `sshpass -e`,不把密碼放進 argv。 - iDRAC 密碼透過 `IPMI_PASSWORD` environment 搭配 `ipmitool -E`;`config`/TUI review 只顯示 set 與字元數。 - iDRAC/ESXi 應位於隔離管理網路,避免把 IPMI over LAN 暴露到公網。 @@ -331,13 +354,13 @@ esxcli storage core device smart get -d 't10.NVMe____完整識別字串' ## 測試與品質門檻 ```bash -make test # bash -n + 37 個核心 assertions + 10 個 TUI assertions +make test # bash -n + 52 個核心 assertions + 14 個 TUI assertions make validate # 以 Docker 依賴檢查目前的 .env(不改變風扇) make validate-example # 不需要 .env 的 DRY_RUN smoke test make docker-build ``` -測試涵蓋 source normalization、SDR parsing、GPU offset、hysteresis、fail-safe、曲線/port/regex 驗證、credential redaction、health freshness、診斷 preview,以及 TUI 設定檔的安全 round-trip。真實硬體、ESXi SSH 和 GPU 仍須在你的管理網路上以 `diagnose` 驗證。 +測試涵蓋統一 source 介面、Linux SMART 收集、多主機 remote GPU、source offset、SDR parsing、hysteresis、fail-safe、設定驗證、credential redaction、health freshness、診斷 preview,以及 TUI 設定檔安全 round-trip。真實硬體、SSH target、磁碟、iDRAC 與 GPU 仍須在管理網路上以 `diagnose` 驗證。 ## 專案結構 @@ -351,6 +374,7 @@ make docker-build │ ├── fan-control.test.sh # 核心邏輯與安全測試 │ └── tui.test.sh # .env parser / writer 測試 ├── docs/ +│ ├── TEMPERATURE_SOURCES.md # Source 介面與部署範例 │ └── TROUBLESHOOTING.md # 症狀導向除錯手冊 ├── images/image.png # iDRAC IPMI 設定畫面 ├── .env.example # 帶完整註解的設定模板 @@ -364,10 +388,10 @@ make docker-build ## 相容性與限制 -專案以 Dell PowerEdge R730/R730xd、iDRAC 8 類機型的 OEM fan raw command 為主要驗證目標。其他世代可能相容,但不能假設相同;請在無人值守前完成 `manual`/`restore`/`diagnose`。iDRAC、ESXi、GPU 的實際 sensor 名稱與權限會依韌體和驅動改變,應以你自己的 diagnostic output 為準。 +專案以 Dell PowerEdge R730/R730xd、iDRAC 8 類機型的 OEM fan raw command 為主要驗證目標。其他世代可能相容,但不能假設相同;請在無人值守前完成 `manual`/`restore`/`diagnose`。Sensor 名稱、磁碟權限、SSH access 與 driver 行為會依環境改變,應以實際 diagnostic output 為準。 ## 授權 MIT,詳見 [LICENSE](LICENSE)。 -文件最後檢視:2026-07-18。 +文件最後檢視:2026-08-15。 diff --git a/USAGE_GUIDE.md b/USAGE_GUIDE.md index 1f8d5e0..0cd754c 100644 --- a/USAGE_GUIDE.md +++ b/USAGE_GUIDE.md @@ -124,6 +124,38 @@ cp ~/.ssh/esxi_ed25519 secrets/esxi_ed25519 chmod 600 secrets/esxi_ed25519 ``` +### TrueNAS CD6 與其他 VM 的 GPU + +目前 TrueNAS 25.10 主機上的 Kioxia CD6 controller device 是 `/dev/nvme1`。先在 host 唯讀確認: + +```bash +sudo smartctl -A -j /dev/nvme1 | jq '.temperature.current' +``` + +設定 CD6 與多台 NVIDIA VM: + +```dotenv +TEMPERATURE_SOURCES=linux_disk,remote_gpu +LINUX_DISK_DEVICES=/dev/nvme1 +LINUX_DISK_TEMP_OFFSET=0 + +REMOTE_GPU_HOSTS=gpu-vm-1,gpu-vm-2 +REMOTE_GPU_USERNAME=monitor +REMOTE_GPU_SSH_KEY=/run/secrets/gpu_vms_ed25519 +REMOTE_GPU_PASSWORD= +REMOTE_GPU_SSH_PORT=22 +REMOTE_GPU_TEMP_OFFSET=15 +``` + +在 `docker-compose.yml` 的 service 加入精確 device mapping: + +```yaml +devices: + - /dev/nvme1:/dev/nvme1 +``` + +每台 GPU VM 必須讓該 SSH 帳號可執行唯讀的 `nvidia-smi --query-gpu=index,temperature.gpu --format=csv,noheader,nounits`。設定後先跑 `validate` 與 `diagnose`;預期同時看到 `source:linux_disk`、`source:remote_gpu` 和 decision preview。詳見 [Temperature source interface](docs/TEMPERATURE_SOURCES.md)。 + ### 啟用 GPU ```dotenv @@ -178,4 +210,4 @@ make docker-build 長時間運行請確認 `logs/` 有輪替策略;專案只負責寫入單一 log,不會替主機設定 logrotate。建議以主機的 logrotate 或 journald retention 管理容量。 -文件最後檢視:2026-07-18。 +文件最後檢視:2026-08-15。 diff --git a/docker-compose.yml b/docker-compose.yml index 374b065..b17b84a 100644 --- a/docker-compose.yml +++ b/docker-compose.yml @@ -8,6 +8,10 @@ services: - ./logs:/var/log/fan-control # Put an ESXi private key in ./secrets and use /run/secrets/ in .env. - ./secrets:/run/secrets:ro + # Map each device listed in LINUX_DISK_DEVICES. Device mappings grant raw + # hardware access, so expose only the disks used by the controller. + # devices: + # - /dev/nvme1:/dev/nvme1 network_mode: host restart: unless-stopped init: true diff --git a/docs/TEMPERATURE_SOURCES.md b/docs/TEMPERATURE_SOURCES.md new file mode 100644 index 0000000..ada5d9d --- /dev/null +++ b/docs/TEMPERATURE_SOURCES.md @@ -0,0 +1,88 @@ +# Temperature source interface + +The controller treats every temperature provider through the same interface. `collect_temperature_readings` and `diagnose_mode` never branch on a source ID; they call the registered source methods instead. + +## Interface contract + +Register a source ID in `TEMPERATURE_SOURCE_IDS`, then implement these Bash functions in `src/FanControlWithEsxiSmart.sh`: + +```bash +temperature_source__validate() { ...; } +temperature_source__collect() { ...; } +temperature_source__adjust() { ...; } +``` + +- `validate` checks required configuration and runtime commands. It returns nonzero on an invalid setup. +- `collect` prints one or more tab-separated records and returns zero when at least one reading is valid. +- `adjust` receives one integer Celsius value and prints the integer used by the fan curve. + +Each collection record has this format: + +```text +source_idsensor_labelraw_temperature_celsius +``` + +Labels must not contain tabs or newlines. A source must reject missing, malformed, or unsupported readings instead of emitting `0`. The controller applies every source's `adjust` method, logs both raw and adjusted values when they differ, and selects the highest adjusted temperature. + +`temperature_source_interface_complete` checks the contract during validation. The supported source list is an allowlist, so an environment value cannot invoke an arbitrary shell function. + +## Built-in sources + +| ID | Provider | Adjustment | +| --- | --- | --- | +| `idrac` | All readable `ipmitool sdr type Temperature` sensors | None | +| `esxi` | One ESXi storage device queried through SSH and `esxcli` | None | +| `gpu` | All local NVIDIA GPUs queried with `nvidia-smi` | `GPU_TEMP_OFFSET` | +| `linux_disk` | One or more local Linux devices queried with `smartctl -A -j` | `LINUX_DISK_TEMP_OFFSET` | +| `remote_gpu` | All NVIDIA GPUs on one or more Linux hosts queried through SSH | `REMOTE_GPU_TEMP_OFFSET` | + +## Linux disk behavior + +Set `LINUX_DISK_DEVICES` to comma-separated `/dev` paths. For NVMe, prefer the controller node such as `/dev/nvme1`; namespace paths such as `/dev/nvme1n1` also work when smartctl supports them. + +The parser first reads smartctl's generic `temperature.current`, then checks the NVMe health log and ATA temperature attributes. It accepts a valid temperature even when smartctl's exit status reports disk health bits. A timeout, invalid JSON, inaccessible device, or absent temperature fails that device only. Set `LINUX_DISK_NOCHECK=standby` to avoid waking sleeping SATA/SAS disks; keep the default `never` for always-on devices such as the CD6. + +The source requires `smartctl`, `jq`, and `timeout`. The container image includes them. Containers also need an explicit device mapping for every configured disk: + +```yaml +services: + idrac-fan-control: + devices: + - /dev/nvme1:/dev/nvme1 +``` + +Grant access only to selected devices. Raw disk access is sensitive even though this controller invokes only the read-only `smartctl -A -j` query. + +## TrueNAS CD6 plus remote GPU VMs + +The following configuration combines a TrueNAS-hosted Kioxia CD6 with GPUs in two Linux VMs: + +```dotenv +TEMPERATURE_SOURCES=linux_disk,remote_gpu + +LINUX_DISK_DEVICES=/dev/nvme1 +LINUX_DISK_TEMP_OFFSET=0 +LINUX_DISK_NOCHECK=never + +REMOTE_GPU_HOSTS=gpu-vm-1,gpu-vm-2 +REMOTE_GPU_USERNAME=monitor +REMOTE_GPU_PASSWORD= +REMOTE_GPU_SSH_KEY=/run/secrets/gpu_vms_ed25519 +REMOTE_GPU_SSH_PORT=22 +REMOTE_GPU_TEMP_OFFSET=15 +``` + +Each remote host must have `nvidia-smi` in its non-interactive SSH `PATH`. Use a restricted monitoring account and SSH key where practical. The controller runs only this remote command: + +```bash +nvidia-smi --query-gpu=index,temperature.gpu --format=csv,noheader,nounits +``` + +Run `validate`, then `diagnose`. A healthy result includes one `source:linux_disk` row, one combined `source:remote_gpu` row, and a decision preview. A multi-device source passes when at least one configured target returns a reading and logs a warning for each failed target. `diagnose` returns nonzero when an enabled source returns no readings. Automatic control continues with remaining valid sources; if every source fails, the configured fail-safe applies. + +## Upstream references + +- [smartctl manual](https://github.com/smartmontools/smartmontools/blob/main/src/smartctl.8.in): Linux device forms, JSON output, and exit-status bitmask. +- [Docker Compose service `devices`](https://docs.docker.com/reference/compose-file/services/#devices): host-to-container device mapping syntax. +- [NVIDIA System Management Interface](https://docs.nvidia.com/deploy/nvidia-smi/index.html): selective GPU queries, CSV formatting, and Celsius temperature fields. +- [TrueNAS 25.10 disk API](https://api.truenas.com/v25.10/api_methods_disk.html): platform disk inventory and temperature methods used for deployment cross-checks. diff --git a/docs/TROUBLESHOOTING.md b/docs/TROUBLESHOOTING.md index 6e9bc94..8c8e85d 100644 --- a/docs/TROUBLESHOOTING.md +++ b/docs/TROUBLESHOOTING.md @@ -36,7 +36,7 @@ docker logs --tail 200 idrac-fan-control tail -n 200 logs/fan_control.log ``` -需要命令階段與來源選擇細節時,在 `.env` 設 `LOG_LEVEL=DEBUG` 後重跑 `diagnose`。Debug log 不會列出 iDRAC/ESXi 密碼;提交 issue 前仍應檢查主機名稱、IP、device identifier 是否需要遮罩。 +需要命令階段與來源選擇細節時,在 `.env` 設 `LOG_LEVEL=DEBUG` 後重跑 `diagnose`。Debug log 不會列出 iDRAC、ESXi 或 remote GPU 密碼;提交 issue 前仍應檢查主機名稱、IP、device identifier 是否需要遮罩。 ## Diagnostic output 怎麼讀 @@ -67,6 +67,8 @@ Result: PASS - controller dependencies are ready. | `source:idrac ... no valid` | SDR 無 readable temperature 或 regex 過濾全部 | `status`、清空 include regex | 修正 filter;保留預設 exclude | | `source:esxi ... no valid` | SSH、device ID 或 SMART 格式錯 | ESXi 上直接跑 `esxcli ... smart get` | 修復 key/password、port、完整 ID | | `source:gpu ... no valid` | 容器無 GPU、驅動/Toolkit 未配置 | `docker ... nvidia-smi` | 啟用 `gpus: all` 與 Toolkit,或移除 gpu source | +| `source:linux_disk ... no valid` | device 未映射、權限不足、SMART 無溫度 | host 與容器內分別執行 `smartctl -A -j` | 加入精確 `devices` mapping,確認 `jq` 與路徑 | +| `source:remote_gpu ... no valid` | VM 不可達、SSH 認證或遠端 `nvidia-smi` 失敗 | 以相同 key 手動 SSH 執行 query | 修復 host/key/known_hosts/driver,或移除失敗 target | | `No valid temperature readings` | 所有 selected sources 都失敗 | `diagnose` | 修復至少一個 source;先保留 fail-safe | | `Fail-safe fan speed applied` | 上述全部失敗且 fail-safe 開啟 | 同時看前面的 source WARN | 修復來源;這不是曲線 level | | `Fan speeds must not decrease` | 高溫檔位比低溫檔位低 | TUI Review | 讓 idle ≤ low ≤ medium ≤ high ≤ critical | @@ -130,6 +132,23 @@ ssh -i /path/to/key -o BatchMode=yes root@ESXI_HOST true 容器要使用 host key file 時,還必須把 key 以 read-only volume 掛進 container,且 `.env` 的 `ESXI_SSH_KEY` 要填 container 內路徑;只填 host path 不會自動掛載。 +## Linux 磁碟深入檢查 + +先在 host 上執行與控制器相同的唯讀查詢: + +```bash +sudo smartctl -A -j /dev/nvme1 | jq '.temperature.current // .nvme_smart_health_information_log.temperature' +``` + +再確認容器看到同一個 device: + +```bash +docker compose run --rm --entrypoint sh idrac-fan-control -c \ + 'ls -l /dev/nvme1 && smartctl -A -j /dev/nvme1 | jq .temperature' +``` + +Host 成功、容器失敗通常代表 `docker-compose.yml` 未加入 `/dev/nvme1:/dev/nvme1`。`LINUX_DISK_DEVICES` 必須是 `/dev/...` 逗號清單;控制器會拒絕空白、shell metacharacter 與相對路徑。smartctl 的非零狀態可能是 SMART health bitmask,因此只要 JSON 仍含有效溫度,控制器會保留該讀值;timeout、無效 JSON 或沒有溫度才使該 device 失敗。 + ## NVIDIA GPU 深入檢查 ```bash @@ -140,6 +159,15 @@ docker compose run --rm idrac-fan-control diagnose 第一個失敗代表主機驅動問題;第一個成功、第二個失敗通常是 Container Toolkit;兩者成功但 Compose 失敗則檢查 `docker-compose.yml` 的 `gpus: all`。不用 GPU 做控制時,最安全的修復是從 `TEMPERATURE_SOURCES` 移除 `gpu`,而不是把錯誤隱藏。 +遠端 GPU 來源請從 controller host 或容器測試完全相同的命令: + +```bash +ssh -i /run/secrets/gpu_vms_ed25519 monitor@gpu-vm-1 \ + 'nvidia-smi --query-gpu=index,temperature.gpu --format=csv,noheader,nounits' +``` + +若互動式 SSH 成功但 controller 失敗,檢查 key 的容器內路徑、`REMOTE_GPU_USERNAME`、port 與 `SSH_STRICT_HOST_KEY_CHECKING`。遠端登入 shell 必須能在非互動 `PATH` 找到 `nvidia-smi`。 + ## Healthcheck 與 log auto mode 每次成功決策或套用 fail-safe 都會更新 `${LOG_DIR}/${LOG_FILE}`。預設允許的最大年齡為: @@ -173,4 +201,4 @@ docker inspect idrac-fan-control --format '{{json .Mounts}}' 請勿附上 `.env`、密碼、SSH private key、公開可達的管理 IP 或完整 service tag。 -文件最後檢視:2026-07-18。 +文件最後檢視:2026-08-15。 diff --git a/src/FanControlWithEsxiSmart.sh b/src/FanControlWithEsxiSmart.sh index e726b39..3cd1474 100755 --- a/src/FanControlWithEsxiSmart.sh +++ b/src/FanControlWithEsxiSmart.sh @@ -22,6 +22,17 @@ PROGRAM_NAME="$(basename "$0")" : "${SSH_STRICT_HOST_KEY_CHECKING:=accept-new}" : "${DRIVE_DEVICE:=}" +: "${LINUX_DISK_DEVICES:=}" +: "${LINUX_DISK_TEMP_OFFSET:=0}" +: "${LINUX_DISK_NOCHECK:=never}" + +: "${REMOTE_GPU_HOSTS:=}" +: "${REMOTE_GPU_USERNAME:=root}" +: "${REMOTE_GPU_PASSWORD:=}" +: "${REMOTE_GPU_SSH_KEY:=}" +: "${REMOTE_GPU_SSH_PORT:=22}" +: "${REMOTE_GPU_TEMP_OFFSET:=15}" + : "${OPERATION_MODE:=manual}" : "${TEMPERATURE_SOURCES:=esxi}" : "${WITH_GPU_TEMP:=false}" @@ -54,6 +65,9 @@ PROGRAM_NAME="$(basename "$0")" : "${HEALTHCHECK_MAX_AGE:=0}" : "${DRY_RUN:=false}" +readonly TEMPERATURE_SOURCE_INTERFACE_VERSION=1 +readonly TEMPERATURE_SOURCE_IDS="esxi idrac gpu linux_disk remote_gpu" + LAST_LEVEL="" LAST_SPEED="" @@ -234,6 +248,57 @@ source_enabled_in_list() { return 1 } +normalize_csv_list() { + local raw="${1:-}" + local normalized="" + local item + local -a items + + IFS=',' read -r -a items <<< "$raw" + for item in "${items[@]}"; do + item="$(trim "$item")" + [[ -z "$item" ]] && continue + if ! source_enabled_in_list "$item" "$normalized"; then + normalized="${normalized:+${normalized},}${item}" + fi + done + + printf '%s' "$normalized" +} + +temperature_source_supported() { + local wanted="$1" + local source + + for source in $TEMPERATURE_SOURCE_IDS; do + [[ "$source" == "$wanted" ]] && return 0 + done + return 1 +} + +temperature_source_method() { + local source="$1" + local method="$2" + shift 2 + local function_name="temperature_source_${source}_${method}" + + temperature_source_supported "$source" || return 1 + declare -F "$function_name" >/dev/null 2>&1 || { + log "ERROR" "Temperature source '${source}' does not implement ${method}() for interface v${TEMPERATURE_SOURCE_INTERFACE_VERSION}" + return 1 + } + "$function_name" "$@" +} + +temperature_source_interface_complete() { + local source="$1" + local method + + for method in validate collect adjust; do + declare -F "temperature_source_${source}_${method}" >/dev/null 2>&1 || return 1 + done +} + command_needs_ipmi() { case "$1" in auto|once|manual|restore|status|diagnose) return 0 ;; @@ -267,19 +332,19 @@ validate_sources() { sources="$(normalize_sources)" if [[ -z "$sources" ]]; then - log "ERROR" "TEMPERATURE_SOURCES must contain at least one source: esxi, idrac, gpu" + log "ERROR" "TEMPERATURE_SOURCES must contain at least one source: ${TEMPERATURE_SOURCE_IDS// /, }" return 1 fi IFS=',' read -r -a source_list <<< "$sources" for source in "${source_list[@]}"; do - case "$source" in - esxi|idrac|gpu) ;; - *) - log "ERROR" "Unsupported temperature source '${source}'. Use esxi, idrac, or gpu." - error=1 - ;; - esac + if ! temperature_source_supported "$source"; then + log "ERROR" "Unsupported temperature source '${source}'. Use ${TEMPERATURE_SOURCE_IDS// /, }." + error=1 + elif ! temperature_source_interface_complete "$source"; then + log "ERROR" "Temperature source '${source}' has an incomplete interface" + error=1 + fi done return "$error" @@ -332,7 +397,10 @@ validate_config() { validate_integer_range "CHECK_INTERVAL" "$CHECK_INTERVAL" 1 86400 || error=1 validate_integer_range "SSH_CONNECT_TIMEOUT" "$SSH_CONNECT_TIMEOUT" 1 300 || error=1 validate_integer_range "ESXI_SSH_PORT" "$ESXI_SSH_PORT" 1 65535 || error=1 + validate_integer_range "REMOTE_GPU_SSH_PORT" "$REMOTE_GPU_SSH_PORT" 1 65535 || error=1 validate_integer_range "GPU_TEMP_OFFSET" "$GPU_TEMP_OFFSET" 0 120 || error=1 + validate_integer_range "LINUX_DISK_TEMP_OFFSET" "$LINUX_DISK_TEMP_OFFSET" 0 120 || error=1 + validate_integer_range "REMOTE_GPU_TEMP_OFFSET" "$REMOTE_GPU_TEMP_OFFSET" 0 120 || error=1 validate_integer_range "HYSTERESIS" "$HYSTERESIS" 0 30 || error=1 validate_integer_range "HEALTHCHECK_MAX_AGE" "$HEALTHCHECK_MAX_AGE" 0 86400 || error=1 @@ -389,39 +457,12 @@ validate_config() { if command_needs_temperature "$command"; then validate_sources || error=1 sources="$(normalize_sources)" - - if source_enabled_in_list "esxi" "$sources"; then - require_value "ESXI_HOST" "$ESXI_HOST" || error=1 - require_value "ESXI_USERNAME" "$ESXI_USERNAME" || error=1 - require_value "DRIVE_DEVICE" "$DRIVE_DEVICE" || error=1 - if [[ -z "$ESXI_PASSWORD" && -z "$ESXI_SSH_KEY" ]]; then - log "ERROR" "Set ESXI_PASSWORD or ESXI_SSH_KEY when TEMPERATURE_SOURCES includes esxi" - error=1 - fi - if [[ -z "$ESXI_SSH_KEY" ]]; then - require_value "ESXI_PASSWORD" "$ESXI_PASSWORD" || error=1 - elif is_placeholder_value "$ESXI_SSH_KEY"; then - log "ERROR" "Replace placeholder value for ESXI_SSH_KEY" - error=1 - fi - if [[ -n "$ESXI_SSH_KEY" && ! -r "$ESXI_SSH_KEY" ]]; then - log "ERROR" "ESXI_SSH_KEY is not readable: ${ESXI_SSH_KEY}" - error=1 - fi - if ! is_true "$DRY_RUN"; then - require_command "ssh" || error=1 - require_command "timeout" || error=1 - if [[ -n "$ESXI_PASSWORD" && -z "$ESXI_SSH_KEY" ]]; then - require_command "sshpass" || error=1 - fi - fi - fi - - if source_enabled_in_list "gpu" "$sources" && ! is_true "$DRY_RUN"; then - if ! command -v nvidia-smi >/dev/null 2>&1; then - log "WARN" "GPU source is enabled but nvidia-smi is not available; fail-safe will be used if no other source succeeds" - fi - fi + local source + local -a source_list + IFS=',' read -r -a source_list <<< "$sources" + for source in "${source_list[@]}"; do + temperature_source_method "$source" validate || error=1 + done fi return "$error" @@ -468,24 +509,35 @@ remote_quote() { printf "'%s'" "${value//\'/\'\\\'\'}" } -run_esxi_command() { - local remote_command="$1" - local ssh_command=(ssh -p "$ESXI_SSH_PORT" -o "StrictHostKeyChecking=${SSH_STRICT_HOST_KEY_CHECKING}" -o "ConnectTimeout=${SSH_CONNECT_TIMEOUT}") - - if [[ -n "$ESXI_SSH_KEY" ]]; then - ssh_command+=(-i "$ESXI_SSH_KEY" -o BatchMode=yes) +run_ssh_command() { + local context="$1" + local host="$2" + local username="$3" + local password="$4" + local key="$5" + local port="$6" + local remote_command="$7" + local ssh_command=(ssh -p "$port" -o "StrictHostKeyChecking=${SSH_STRICT_HOST_KEY_CHECKING}" -o "ConnectTimeout=${SSH_CONNECT_TIMEOUT}") + + if [[ -n "$key" ]]; then + ssh_command+=(-i "$key" -o BatchMode=yes) fi - ssh_command+=("${ESXI_USERNAME}@${ESXI_HOST}" "$remote_command") - log "DEBUG" "Running ESXi command against ${ESXI_HOST}:${ESXI_SSH_PORT} as ${ESXI_USERNAME}" + ssh_command+=("${username}@${host}" "$remote_command") + log "DEBUG" "Running ${context} SSH query against ${host}:${port} as ${username}" - if [[ -n "$ESXI_PASSWORD" && -z "$ESXI_SSH_KEY" ]]; then - SSHPASS="$ESXI_PASSWORD" timeout "$COMMAND_TIMEOUT" sshpass -e "${ssh_command[@]}" + if [[ -n "$password" && -z "$key" ]]; then + SSHPASS="$password" timeout "$COMMAND_TIMEOUT" sshpass -e "${ssh_command[@]}" else timeout "$COMMAND_TIMEOUT" "${ssh_command[@]}" fi } +run_esxi_command() { + run_ssh_command "ESXi" "$ESXI_HOST" "$ESXI_USERNAME" "$ESXI_PASSWORD" \ + "$ESXI_SSH_KEY" "$ESXI_SSH_PORT" "$1" +} + get_esxi_drive_temperature() { local remote_device local output @@ -567,9 +619,6 @@ get_idrac_temperatures() { get_gpu_temperatures() { local output - local index - local temp - local found=1 command -v nvidia-smi >/dev/null 2>&1 || return 1 @@ -578,18 +627,229 @@ get_gpu_temperatures() { return 1 fi + parse_nvidia_smi_temperatures "gpu" "" <<< "$output" +} + +parse_nvidia_smi_temperatures() { + local source="$1" + local label_prefix="$2" + local index + local temp + local found=1 + while IFS=',' read -r index temp; do index="$(trim "$index")" temp="$(trim "$temp")" if is_integer "$temp"; then - printf 'gpu\tgpu%s\t%s\n' "$index" "$temp" + printf '%s\t%sgpu%s\t%s\n' "$source" "$label_prefix" "$index" "$temp" + found=0 + fi + done + + return "$found" +} + +parse_smartctl_temperature() { + jq -er ' + [ + .temperature.current?, + .nvme_smart_health_information_log.temperature?, + ( + .ata_smart_attributes.table[]? + | select((.name // "" | ascii_downcase) | contains("temperature")) + | .raw.value? + ) + ] + | map(if type == "string" then (tonumber? // empty) else . end) + | map(select(type == "number")) + | (.[0] // empty) + | round + ' 2>/dev/null +} + +get_linux_disk_temperatures() { + local devices + local device + local output + local status + local temp + local label + local found=1 + local -a device_list + + devices="$(normalize_csv_list "$LINUX_DISK_DEVICES")" + IFS=',' read -r -a device_list <<< "$devices" + for device in "${device_list[@]}"; do + status=0 + output="$(timeout "$COMMAND_TIMEOUT" smartctl -n "$LINUX_DISK_NOCHECK" -A -j "$device" 2>/dev/null)" || status=$? + if (( status == 124 )); then + log "WARN" "Linux disk SMART query timed out for ${device}" + continue + fi + + if ! temp="$(parse_smartctl_temperature <<< "$output")" || ! is_integer "$temp"; then + log "WARN" "Linux disk SMART output contained no temperature for ${device} (smartctl status ${status})" + continue + fi + + label="${device#/dev/}" + printf 'linux_disk\t%s\t%s\n' "$label" "$temp" + found=0 + done + + return "$found" +} + +get_remote_gpu_temperatures() { + local hosts + local host + local output + local found=1 + local -a host_list + local remote_command='nvidia-smi --query-gpu=index,temperature.gpu --format=csv,noheader,nounits' + + hosts="$(normalize_csv_list "$REMOTE_GPU_HOSTS")" + IFS=',' read -r -a host_list <<< "$hosts" + for host in "${host_list[@]}"; do + if output="$(run_ssh_command "remote GPU" "$host" "$REMOTE_GPU_USERNAME" \ + "$REMOTE_GPU_PASSWORD" "$REMOTE_GPU_SSH_KEY" "$REMOTE_GPU_SSH_PORT" "$remote_command")" \ + && parse_nvidia_smi_temperatures "remote_gpu" "${host}/" <<< "$output"; then found=0 + else + log "WARN" "Remote NVIDIA temperature query failed for ${host}" fi - done <<< "$output" + done return "$found" } +apply_temperature_offset() { + local temp="$1" + local offset="$2" + local adjusted=$((temp - offset)) + (( adjusted < 0 )) && adjusted=0 + printf '%s' "$adjusted" +} + +temperature_source_esxi_validate() { + local error=0 + + require_value "ESXI_HOST" "$ESXI_HOST" || error=1 + require_value "ESXI_USERNAME" "$ESXI_USERNAME" || error=1 + require_value "DRIVE_DEVICE" "$DRIVE_DEVICE" || error=1 + if [[ -z "$ESXI_PASSWORD" && -z "$ESXI_SSH_KEY" ]]; then + log "ERROR" "Set ESXI_PASSWORD or ESXI_SSH_KEY when TEMPERATURE_SOURCES includes esxi" + error=1 + elif [[ -z "$ESXI_SSH_KEY" ]]; then + require_value "ESXI_PASSWORD" "$ESXI_PASSWORD" || error=1 + elif is_placeholder_value "$ESXI_SSH_KEY"; then + log "ERROR" "Replace placeholder value for ESXI_SSH_KEY" + error=1 + elif [[ ! -r "$ESXI_SSH_KEY" ]]; then + log "ERROR" "ESXI_SSH_KEY is not readable: ${ESXI_SSH_KEY}" + error=1 + fi + if ! is_true "$DRY_RUN"; then + require_command "ssh" || error=1 + require_command "timeout" || error=1 + if [[ -n "$ESXI_PASSWORD" && -z "$ESXI_SSH_KEY" ]]; then + require_command "sshpass" || error=1 + fi + fi + return "$error" +} + +temperature_source_esxi_collect() { get_esxi_drive_temperature; } +temperature_source_esxi_adjust() { printf '%s' "$1"; } + +temperature_source_idrac_validate() { return 0; } +temperature_source_idrac_collect() { get_idrac_temperatures; } +temperature_source_idrac_adjust() { printf '%s' "$1"; } + +temperature_source_gpu_validate() { + if ! is_true "$DRY_RUN" && ! command -v nvidia-smi >/dev/null 2>&1; then + log "WARN" "GPU source is enabled but nvidia-smi is unavailable; fail-safe will be used if no other source succeeds" + fi +} +temperature_source_gpu_collect() { get_gpu_temperatures; } +temperature_source_gpu_adjust() { apply_temperature_offset "$1" "$GPU_TEMP_OFFSET"; } + +temperature_source_linux_disk_validate() { + local devices + local device + local error=0 + local -a device_list + + require_value "LINUX_DISK_DEVICES" "$LINUX_DISK_DEVICES" || return 1 + devices="$(normalize_csv_list "$LINUX_DISK_DEVICES")" + IFS=',' read -r -a device_list <<< "$devices" + for device in "${device_list[@]}"; do + if [[ ! "$device" =~ ^/dev/[A-Za-z0-9._/+:-]+$ ]]; then + log "ERROR" "LINUX_DISK_DEVICES contains an invalid device path: ${device}" + error=1 + fi + done + case "$LINUX_DISK_NOCHECK" in + never|sleep|standby|idle) ;; + *) + log "ERROR" "LINUX_DISK_NOCHECK must be never, sleep, standby, or idle" + error=1 + ;; + esac + if ! is_true "$DRY_RUN"; then + require_command "smartctl" || error=1 + require_command "jq" || error=1 + require_command "timeout" || error=1 + fi + return "$error" +} +temperature_source_linux_disk_collect() { get_linux_disk_temperatures; } +temperature_source_linux_disk_adjust() { apply_temperature_offset "$1" "$LINUX_DISK_TEMP_OFFSET"; } + +temperature_source_remote_gpu_validate() { + local hosts + local host + local error=0 + local -a host_list + + require_value "REMOTE_GPU_HOSTS" "$REMOTE_GPU_HOSTS" || error=1 + require_value "REMOTE_GPU_USERNAME" "$REMOTE_GPU_USERNAME" || error=1 + hosts="$(normalize_csv_list "$REMOTE_GPU_HOSTS")" + IFS=',' read -r -a host_list <<< "$hosts" + for host in "${host_list[@]}"; do + if [[ ! "$host" =~ ^[A-Za-z0-9._:-]+$ ]]; then + log "ERROR" "REMOTE_GPU_HOSTS contains an invalid hostname or address: ${host}" + error=1 + fi + done + if [[ ! "$REMOTE_GPU_USERNAME" =~ ^[A-Za-z0-9._-]+$ ]]; then + log "ERROR" "REMOTE_GPU_USERNAME contains unsupported characters" + error=1 + fi + if [[ -z "$REMOTE_GPU_PASSWORD" && -z "$REMOTE_GPU_SSH_KEY" ]]; then + log "ERROR" "Set REMOTE_GPU_PASSWORD or REMOTE_GPU_SSH_KEY when TEMPERATURE_SOURCES includes remote_gpu" + error=1 + elif [[ -z "$REMOTE_GPU_SSH_KEY" ]]; then + require_value "REMOTE_GPU_PASSWORD" "$REMOTE_GPU_PASSWORD" || error=1 + elif is_placeholder_value "$REMOTE_GPU_SSH_KEY"; then + log "ERROR" "Replace placeholder value for REMOTE_GPU_SSH_KEY" + error=1 + elif [[ ! -r "$REMOTE_GPU_SSH_KEY" ]]; then + log "ERROR" "REMOTE_GPU_SSH_KEY is not readable: ${REMOTE_GPU_SSH_KEY}" + error=1 + fi + if ! is_true "$DRY_RUN"; then + require_command "ssh" || error=1 + require_command "timeout" || error=1 + if [[ -n "$REMOTE_GPU_PASSWORD" && -z "$REMOTE_GPU_SSH_KEY" ]]; then + require_command "sshpass" || error=1 + fi + fi + return "$error" +} +temperature_source_remote_gpu_collect() { get_remote_gpu_temperatures; } +temperature_source_remote_gpu_adjust() { apply_temperature_offset "$1" "$REMOTE_GPU_TEMP_OFFSET"; } + collect_temperature_readings() { local sources local source @@ -602,32 +862,12 @@ collect_temperature_readings() { for source in "${source_list[@]}"; do log "DEBUG" "Collecting temperature source: ${source}" - case "$source" in - esxi) - if output="$(get_esxi_drive_temperature)"; then - printf '%s\n' "$output" - found=0 - else - log "WARN" "ESXi drive temperature read failed" - fi - ;; - idrac) - if output="$(get_idrac_temperatures)"; then - printf '%s\n' "$output" - found=0 - else - log "WARN" "iDRAC temperature sensor read failed" - fi - ;; - gpu) - if output="$(get_gpu_temperatures)"; then - printf '%s\n' "$output" - found=0 - else - log "WARN" "GPU temperature read failed" - fi - ;; - esac + if output="$(temperature_source_method "$source" collect)"; then + printf '%s\n' "$output" + found=0 + else + log "WARN" "Temperature source '${source}' read failed" + fi done return "$found" @@ -646,10 +886,11 @@ calculate_decision_temperature() { [[ -z "${source:-}" || -z "${temp:-}" ]] && continue is_integer "$temp" || continue - adjusted="$temp" - if [[ "$source" == "gpu" ]]; then - adjusted=$(( temp - GPU_TEMP_OFFSET )) - (( adjusted < 0 )) && adjusted=0 + if ! adjusted="$(temperature_source_method "$source" adjust "$temp")" || ! is_integer "$adjusted"; then + log "WARN" "Temperature source '${source}' returned an invalid adjusted value for ${label}" + continue + fi + if [[ "$adjusted" != "$temp" ]]; then detail_text="${detail_text:+${detail_text}, }${source}:${label}=${temp}C(adjusted=${adjusted}C)" else detail_text="${detail_text:+${detail_text}, }${source}:${label}=${temp}C" @@ -895,8 +1136,10 @@ config_row() { print_effective_config() { local esxi_auth="password" + local remote_gpu_auth="password" [[ -n "$ESXI_SSH_KEY" ]] && esxi_auth="ssh-key" + [[ -n "$REMOTE_GPU_SSH_KEY" ]] && remote_gpu_auth="ssh-key" printf 'Effective configuration (credentials are never printed)\n' printf '%s\n' '------------------------------------------------------------' @@ -908,6 +1151,11 @@ print_effective_config() { config_row "ESXi authentication" "$esxi_auth" config_row "ESXi password" "$(credential_state "$ESXI_PASSWORD")" config_row "ESXi drive" "${DRIVE_DEVICE:-not set}" + config_row "Linux disks" "${LINUX_DISK_DEVICES:-not set} (offset=${LINUX_DISK_TEMP_OFFSET} C, nocheck=${LINUX_DISK_NOCHECK})" + config_row "Remote GPU targets" "${REMOTE_GPU_USERNAME}@${REMOTE_GPU_HOSTS:-not set}:${REMOTE_GPU_SSH_PORT}" + config_row "Remote GPU authentication" "$remote_gpu_auth" + config_row "Remote GPU password" "$(credential_state "$REMOTE_GPU_PASSWORD")" + config_row "GPU offsets" "local=${GPU_TEMP_OFFSET} C, remote=${REMOTE_GPU_TEMP_OFFSET} C" config_row "Fan thresholds" "${TEMP_LOW}/${TEMP_MEDIUM}/${TEMP_HIGH}/${TEMP_CRITICAL} C" config_row "Fan speeds" "${FAN_SPEED_IDLE}/${FAN_SPEED_LOW}/${FAN_SPEED_MEDIUM}/${FAN_SPEED_HIGH}/${FAN_SPEED_CRITICAL}%" config_row "Hysteresis" "${HYSTERESIS} C" @@ -978,11 +1226,7 @@ diagnose_mode() { IFS=',' read -r -a source_list <<< "$sources" for source in "${source_list[@]}"; do output="" - case "$source" in - esxi) output="$(get_esxi_drive_temperature)" || true ;; - idrac) output="$(get_idrac_temperatures)" || true ;; - gpu) output="$(get_gpu_temperatures)" || true ;; - esac + output="$(temperature_source_method "$source" collect)" || true if [[ -n "$output" ]]; then diagnostic_row "PASS" "source:${source}" "$(reading_summary "$output")" diff --git a/src/fan-control-tui.sh b/src/fan-control-tui.sh index 05e46d9..1e35769 100755 --- a/src/fan-control-tui.sh +++ b/src/fan-control-tui.sh @@ -15,6 +15,9 @@ CONFIG_KEYS=( IDRAC_SENSOR_INCLUDE_REGEX IDRAC_SENSOR_EXCLUDE_REGEX ESXI_HOST ESXI_USERNAME ESXI_PASSWORD ESXI_SSH_KEY ESXI_SSH_PORT SSH_CONNECT_TIMEOUT SSH_STRICT_HOST_KEY_CHECKING DRIVE_DEVICE + LINUX_DISK_DEVICES LINUX_DISK_TEMP_OFFSET LINUX_DISK_NOCHECK + REMOTE_GPU_HOSTS REMOTE_GPU_USERNAME REMOTE_GPU_PASSWORD REMOTE_GPU_SSH_KEY + REMOTE_GPU_SSH_PORT REMOTE_GPU_TEMP_OFFSET TEMP_LOW TEMP_MEDIUM TEMP_HIGH TEMP_CRITICAL FAN_SPEED_IDLE FAN_SPEED_LOW FAN_SPEED_MEDIUM FAN_SPEED_HIGH FAN_SPEED_CRITICAL HYSTERESIS FAILSAFE_ON_ERROR FAILSAFE_FAN_SPEED MANUAL_FAN_SPEED @@ -301,30 +304,38 @@ select_sources() { printf '%sTemperature source preset%s\n\n' "$COLOR_BLUE" "$COLOR_RESET" printf ' 1) iDRAC sensors only recommended first setup\n' printf ' 2) ESXi NVMe SMART only\n' - printf ' 3) iDRAC + ESXi NVMe SMART\n' - printf ' 4) iDRAC + NVIDIA GPU\n' - printf ' 5) ESXi NVMe SMART + NVIDIA GPU\n' - printf ' 6) iDRAC + ESXi + NVIDIA GPU\n\n' + printf ' 3) Local Linux disk SMART only\n' + printf ' 4) iDRAC + local Linux disk SMART\n' + printf ' 5) iDRAC + local NVIDIA GPU\n' + printf ' 6) Local Linux disk + remote NVIDIA VMs\n' + printf ' 7) Custom source list\n\n' while true; do - printf 'Choose [1-6]: ' + printf 'Choose [1-7]: ' IFS= read -r choice || return 1 case "$choice" in 1) sources="idrac" ;; 2) sources="esxi" ;; - 3) sources="idrac,esxi" ;; - 4) sources="idrac,gpu" ;; - 5) sources="esxi,gpu" ;; - 6) sources="idrac,esxi,gpu" ;; - *) warning "Choose a number from 1 to 6."; continue ;; + 3) sources="linux_disk" ;; + 4) sources="idrac,linux_disk" ;; + 5) sources="idrac,gpu" ;; + 6) sources="linux_disk,remote_gpu" ;; + 7) + prompt_value TEMPERATURE_SOURCES "Sources (esxi,idrac,gpu,linux_disk,remote_gpu)" true + sources="$(config_get TEMPERATURE_SOURCES)" + ;; + *) warning "Choose a number from 1 to 7."; continue ;; esac break done config_set TEMPERATURE_SOURCES "$sources" config_set WITH_GPU_TEMP "false" - if [[ "$sources" == *gpu* ]]; then + if [[ ",$sources," == *,gpu,* ]]; then prompt_integer GPU_TEMP_OFFSET "GPU temperature offset (C)" 0 120 fi + if [[ ",$sources," == *,remote_gpu,* ]]; then + prompt_integer REMOTE_GPU_TEMP_OFFSET "Remote GPU temperature offset (C)" 0 120 + fi notice "Temperature sources saved: ${sources}" } @@ -344,6 +355,31 @@ configure_esxi() { notice "ESXi settings saved." } +configure_linux_disk() { + header + printf '%sLocal Linux disk SMART source%s\n\n' "$COLOR_BLUE" "$COLOR_RESET" + prompt_value LINUX_DISK_DEVICES "Device paths, comma-separated" true + prompt_integer LINUX_DISK_TEMP_OFFSET "Disk temperature offset (C)" 0 120 + prompt_value LINUX_DISK_NOCHECK "Power mode check (never/sleep/standby/idle)" true + notice "Linux disk settings saved. Map each device into the container before diagnostics." +} + +configure_remote_gpu() { + header + printf '%sRemote NVIDIA GPU source%s\n\n' "$COLOR_BLUE" "$COLOR_RESET" + prompt_value REMOTE_GPU_HOSTS "GPU VM hostnames or IPs, comma-separated" true + prompt_value REMOTE_GPU_USERNAME "SSH username shared by GPU VMs" true + prompt_integer REMOTE_GPU_SSH_PORT "SSH port" 1 65535 + prompt_value REMOTE_GPU_SSH_KEY "SSH private key path (- clears it; blank keeps current)" false + if [[ -z "$(config_get REMOTE_GPU_SSH_KEY 2>/dev/null || true)" ]]; then + prompt_secret REMOTE_GPU_PASSWORD "SSH password shared by GPU VMs" true + else + warning "SSH key authentication selected; the password will be ignored." + fi + prompt_integer REMOTE_GPU_TEMP_OFFSET "Remote GPU temperature offset (C)" 0 120 + notice "Remote GPU settings saved." +} + configure_curve() { header printf '%sFan curve%s\n' "$COLOR_BLUE" "$COLOR_RESET" @@ -392,6 +428,12 @@ quick_setup() { if [[ "$(config_get TEMPERATURE_SOURCES)" == *esxi* ]]; then configure_esxi fi + if [[ "$(config_get TEMPERATURE_SOURCES)" == *linux_disk* ]]; then + configure_linux_disk + fi + if [[ "$(config_get TEMPERATURE_SOURCES)" == *remote_gpu* ]]; then + configure_remote_gpu + fi header notice "Quick setup is complete. Run Validate, then Diagnostics before starting auto mode." pause_screen @@ -409,10 +451,12 @@ masked_state() { show_config() { local idrac_password local esxi_password + local remote_gpu_password header idrac_password="$(config_get IDRAC_PASSWORD 2>/dev/null || true)" esxi_password="$(config_get ESXI_PASSWORD 2>/dev/null || true)" + remote_gpu_password="$(config_get REMOTE_GPU_PASSWORD 2>/dev/null || true)" printf '%sEffective setup (secrets redacted)%s\n\n' "$COLOR_BLUE" "$COLOR_RESET" printf ' %-28s %s\n' "Mode" "$(config_get OPERATION_MODE 2>/dev/null || true)" printf ' %-28s %s\n' "Sources" "$(config_get TEMPERATURE_SOURCES 2>/dev/null || true)" @@ -421,6 +465,9 @@ show_config() { printf ' %-28s %s@%s:%s\n' "ESXi" "$(config_get ESXI_USERNAME 2>/dev/null || true)" "$(config_get ESXI_HOST 2>/dev/null || true)" "$(config_get ESXI_SSH_PORT 2>/dev/null || true)" printf ' %-28s %s\n' "ESXi password" "$(masked_state "$esxi_password")" printf ' %-28s %s\n' "Drive device" "$(config_get DRIVE_DEVICE 2>/dev/null || true)" + printf ' %-28s %s\n' "Linux disks" "$(config_get LINUX_DISK_DEVICES 2>/dev/null || true)" + printf ' %-28s %s@%s:%s\n' "Remote GPU VMs" "$(config_get REMOTE_GPU_USERNAME 2>/dev/null || true)" "$(config_get REMOTE_GPU_HOSTS 2>/dev/null || true)" "$(config_get REMOTE_GPU_SSH_PORT 2>/dev/null || true)" + printf ' %-28s %s\n' "Remote GPU password" "$(masked_state "$remote_gpu_password")" printf ' %-28s %s/%s/%s/%s C\n' "Thresholds" "$(config_get TEMP_LOW)" "$(config_get TEMP_MEDIUM)" "$(config_get TEMP_HIGH)" "$(config_get TEMP_CRITICAL)" printf ' %-28s %s/%s/%s/%s/%s %%\n' "Fan speeds" "$(config_get FAN_SPEED_IDLE)" "$(config_get FAN_SPEED_LOW)" "$(config_get FAN_SPEED_MEDIUM)" "$(config_get FAN_SPEED_HIGH)" "$(config_get FAN_SPEED_CRITICAL)" printf ' %-28s enabled=%s, %s%%\n' "Fail-safe" "$(config_get FAILSAFE_ON_ERROR)" "$(config_get FAILSAFE_FAN_SPEED)" @@ -483,31 +530,35 @@ main_menu() { printf ' 2) iDRAC / IPMI settings\n' printf ' 3) Temperature source preset\n' printf ' 4) ESXi NVMe source settings\n' - printf ' 5) Fan curve\n' - printf ' 6) Safety, timing, and logging\n' - printf ' 7) Review redacted configuration\n' - printf ' 8) Validate configuration\n' - printf ' 9) Run read-only diagnostics\n' + printf ' 5) Local Linux disk settings\n' + printf ' 6) Remote NVIDIA GPU settings\n' + printf ' 7) Fan curve\n' + printf ' 8) Safety, timing, and logging\n' + printf ' 9) Review redacted configuration\n' + printf ' 10) Validate configuration\n' + printf ' 11) Run read-only diagnostics\n' printf ' 0) Save and exit\n\n' - printf 'Choose [0-9]: ' + printf 'Choose [0-11]: ' IFS= read -r choice || return 0 case "$choice" in 1) quick_setup ;; 2) configure_idrac; pause_screen ;; 3) select_sources; pause_screen ;; 4) configure_esxi; pause_screen ;; - 5) configure_curve; pause_screen ;; - 6) configure_safety; pause_screen ;; - 7) show_config ;; - 8) run_and_pause validate ;; - 9) run_and_pause diagnose ;; + 5) configure_linux_disk; pause_screen ;; + 6) configure_remote_gpu; pause_screen ;; + 7) configure_curve; pause_screen ;; + 8) configure_safety; pause_screen ;; + 9) show_config ;; + 10) run_and_pause validate ;; + 11) run_and_pause diagnose ;; 0) header notice "Saved ${CONFIG_FILE} with mode 600." printf 'Next: make validate && docker compose up -d\n' return 0 ;; - *) warning "Choose a number from 0 to 9."; pause_screen ;; + *) warning "Choose a number from 0 to 11."; pause_screen ;; esac done } diff --git a/tests/fan-control.test.sh b/tests/fan-control.test.sh index b581074..9551aa5 100755 --- a/tests/fan-control.test.sh +++ b/tests/fan-control.test.sh @@ -14,6 +14,10 @@ export ESXI_HOST=10.0.0.20 export ESXI_USERNAME=root export ESXI_PASSWORD=test-password export DRIVE_DEVICE=test-drive +export LINUX_DISK_DEVICES=/dev/nvme1 +export REMOTE_GPU_HOSTS=gpu-vm-1 +export REMOTE_GPU_USERNAME=root +export REMOTE_GPU_PASSWORD=test-password # shellcheck source=../src/FanControlWithEsxiSmart.sh source "${ROOT_DIR}/src/FanControlWithEsxiSmart.sh" @@ -62,6 +66,15 @@ test_normalize_sources() { assert_eq "idrac,gpu" "$(normalize_sources)" "WITH_GPU_TEMP appends gpu source" } +test_temperature_sources_share_complete_interface() { + local source + + for source in $TEMPERATURE_SOURCE_IDS; do + temperature_source_interface_complete "$source" || fail "${source} should implement the temperature source interface" + pass + done +} + test_hex_formatting() { assert_eq "1e" "$(fan_speed_to_hex 30)" "formats 30 percent as IPMI hex" assert_eq "64" "$(fan_speed_to_hex 100)" "formats 100 percent as IPMI hex" @@ -150,6 +163,117 @@ test_decision_temperature_uses_max_adjusted_source() { assert_contains "gpu:gpu0=90C(adjusted=75C)" "$decision" "records adjusted GPU detail" } +test_linux_disk_source_collects_multiple_devices_and_accepts_smart_warning_status() { + local output + + output="$( + LINUX_DISK_DEVICES='/dev/nvme1, /dev/sdb' + timeout() { + shift + "$@" + } + smartctl() { + case "${*: -1}" in + /dev/nvme1) printf '{"temperature":73}\n'; return 8 ;; + /dev/sdb) printf '{"temperature":41}\n' ;; + esac + } + parse_smartctl_temperature() { + sed -nE 's/.*"temperature":([0-9]+).*/\1/p' + } + get_linux_disk_temperatures + )" + + assert_contains $'linux_disk\tnvme1\t73' "$output" "keeps a valid reading when smartctl reports health status bits" + assert_contains $'linux_disk\tsdb\t41' "$output" "collects every configured Linux disk" +} + +test_smartctl_json_parser_supports_nvme_and_generic_temperature() { + if ! command -v jq >/dev/null 2>&1; then + pass + return + fi + + assert_eq "73" "$(parse_smartctl_temperature <<< '{"temperature":{"current":73},"nvme_smart_health_information_log":{"temperature":73}}')" "parses generic smartctl temperature.current" + assert_eq "61" "$(parse_smartctl_temperature <<< '{"nvme_smart_health_information_log":{"temperature":61}}')" "parses the NVMe health log fallback" +} + +test_remote_gpu_source_collects_multiple_hosts() { + local output + + output="$( + REMOTE_GPU_HOSTS='gpu-vm-1,gpu-vm-2' + run_ssh_command() { + local host="$2" + if [[ "$host" == "gpu-vm-1" ]]; then + printf '0, 81\n' + else + printf '0, 76\n1, 70\n' + fi + } + get_remote_gpu_temperatures + )" + + assert_contains $'remote_gpu\tgpu-vm-1/gpu0\t81' "$output" "labels a remote GPU with its VM" + assert_contains $'remote_gpu\tgpu-vm-2/gpu1\t70' "$output" "collects every configured remote GPU VM" +} + +test_source_specific_adjustment_is_dispatched_through_interface() { + local decision + + REMOTE_GPU_TEMP_OFFSET=15 + LINUX_DISK_TEMP_OFFSET=0 + decision="$(calculate_decision_temperature $'linux_disk\tnvme1\t73\nremote_gpu\tgpu-vm/gpu0\t90')" + assert_contains $'75\t' "$decision" "uses the highest source-adjusted temperature" + assert_contains "remote_gpu:gpu-vm/gpu0=90C(adjusted=75C)" "$decision" "records remote GPU adjustment details" +} + +test_linux_and_remote_sources_validate_configuration() { + ( + OPERATION_MODE=auto + TEMPERATURE_SOURCES=linux_disk,remote_gpu + WITH_GPU_TEMP=false + DRY_RUN=true + IDRAC_IP=10.0.0.10 + IDRAC_ID=root + IDRAC_PASSWORD=test-password + LINUX_DISK_DEVICES=/dev/nvme1 + REMOTE_GPU_HOSTS=gpu-vm-1,gpu-vm-2 + REMOTE_GPU_USERNAME=monitor + REMOTE_GPU_PASSWORD=test-password + REMOTE_GPU_SSH_KEY="" + validate_config auto >/dev/null 2>&1 + ) || fail "validate_config should accept Linux disk and remote GPU sources" + pass + + ( + OPERATION_MODE=auto + TEMPERATURE_SOURCES=linux_disk + WITH_GPU_TEMP=false + DRY_RUN=true + IDRAC_IP=10.0.0.10 + IDRAC_ID=root + IDRAC_PASSWORD=test-password + LINUX_DISK_DEVICES='nvme1; touch /tmp/unsafe' + validate_config auto >/dev/null 2>&1 + ) && fail "validate_config should reject unsafe Linux device paths" + pass + + ( + OPERATION_MODE=auto + TEMPERATURE_SOURCES=linux_disk + WITH_GPU_TEMP=false + DRY_RUN=true + IDRAC_IP=10.0.0.10 + IDRAC_ID=root + IDRAC_PASSWORD=test-password + LINUX_DISK_DEVICES=/dev/sdb + LINUX_DISK_NOCHECK=invalid + validate_config auto >/dev/null 2>&1 + ) && fail "validate_config should reject an invalid smartctl power mode" + pass +} + test_fail_safe_when_all_sources_fail() { FAILSAFE_ON_ERROR=true FAILSAFE_FAN_SPEED=70 @@ -308,9 +432,10 @@ test_config_output_redacts_credentials() { IDRAC_IP=10.0.0.10 IDRAC_PASSWORD=top-secret-password ESXI_PASSWORD=another-secret + REMOTE_GPU_PASSWORD=remote-secret output="$(print_effective_config)" assert_contains "set (19 characters)" "$output" "shows credential state without values" - if [[ "$output" == *top-secret-password* || "$output" == *another-secret* ]]; then + if [[ "$output" == *top-secret-password* || "$output" == *another-secret* || "$output" == *remote-secret* ]]; then fail "print_effective_config must never print credentials" fi pass @@ -370,6 +495,7 @@ test_diagnose_reports_read_only_source_and_preview() { } test_normalize_sources +test_temperature_sources_share_complete_interface test_hex_formatting test_remote_quote_handles_single_quotes test_esxi_password_is_not_in_process_arguments @@ -377,6 +503,11 @@ test_temperature_levels test_hysteresis_only_delays_downshift test_idrac_sensor_parsing test_decision_temperature_uses_max_adjusted_source +test_linux_disk_source_collects_multiple_devices_and_accepts_smart_warning_status +test_smartctl_json_parser_supports_nvme_and_generic_temperature +test_remote_gpu_source_collects_multiple_hosts +test_source_specific_adjustment_is_dispatched_through_interface +test_linux_and_remote_sources_validate_configuration test_fail_safe_when_all_sources_fail test_validate_accepts_idrac_only_auto_mode test_validate_rejects_placeholder_values diff --git a/tests/tui.test.sh b/tests/tui.test.sh index af816b9..0d2b47d 100644 --- a/tests/tui.test.sh +++ b/tests/tui.test.sh @@ -90,10 +90,22 @@ test_load_config_environment_is_allowlisted() { assert_eq "$before" "$PATH" "does not overwrite unrelated process environment" } +test_new_source_configuration_round_trips() { + config_set LINUX_DISK_DEVICES '/dev/nvme1,/dev/sdb' + config_set REMOTE_GPU_HOSTS 'gpu-vm-1,gpu-vm-2' + config_set REMOTE_GPU_PASSWORD 'remote secret' + + assert_eq '/dev/nvme1,/dev/sdb' "$(config_get LINUX_DISK_DEVICES)" "round-trips Linux disk devices" + assert_eq 'gpu-vm-1,gpu-vm-2' "$(config_get REMOTE_GPU_HOSTS)" "round-trips remote GPU hosts" + assert_eq 'remote secret' "$(config_get REMOTE_GPU_PASSWORD)" "round-trips the remote GPU credential" + assert_eq 'set (13 characters)' "$(masked_state "$(config_get REMOTE_GPU_PASSWORD)")" "masks the remote GPU credential" +} + test_config_round_trip_and_preserves_comments test_config_set_does_not_duplicate_keys test_config_rejects_unknown_key_and_newline test_sensitive_values_are_only_summarized test_load_config_environment_is_allowlisted +test_new_source_configuration_round_trips printf 'ok - %s assertions passed\n' "$PASS_COUNT" From 407775dde418e253edf7d8076f3ed95c55af2bfa Mon Sep 17 00:00:00 2001 From: df Date: Sun, 16 Aug 2026 01:58:27 +0800 Subject: [PATCH 2/2] fix(temperature): reject Linux device path traversal Reject dot and empty path components before invoking smartctl while preserving nested /dev paths such as disk/by-id. Add regression coverage and document the accepted path shape. --- docs/TEMPERATURE_SOURCES.md | 2 +- docs/TROUBLESHOOTING.md | 2 +- src/FanControlWithEsxiSmart.sh | 18 +++++++++++++++++- tests/fan-control.test.sh | 26 ++++++++++++++++++++++++++ 4 files changed, 45 insertions(+), 3 deletions(-) diff --git a/docs/TEMPERATURE_SOURCES.md b/docs/TEMPERATURE_SOURCES.md index ada5d9d..f6834e9 100644 --- a/docs/TEMPERATURE_SOURCES.md +++ b/docs/TEMPERATURE_SOURCES.md @@ -38,7 +38,7 @@ Labels must not contain tabs or newlines. A source must reject missing, malforme ## Linux disk behavior -Set `LINUX_DISK_DEVICES` to comma-separated `/dev` paths. For NVMe, prefer the controller node such as `/dev/nvme1`; namespace paths such as `/dev/nvme1n1` also work when smartctl supports them. +Set `LINUX_DISK_DEVICES` to comma-separated `/dev` paths. Each path must contain only regular device-name components; the controller rejects empty, `.` and `..` components to prevent traversal outside `/dev`. Nested paths such as `/dev/disk/by-id/...` remain valid. For NVMe, prefer the controller node such as `/dev/nvme1`; namespace paths such as `/dev/nvme1n1` also work when smartctl supports them. The parser first reads smartctl's generic `temperature.current`, then checks the NVMe health log and ATA temperature attributes. It accepts a valid temperature even when smartctl's exit status reports disk health bits. A timeout, invalid JSON, inaccessible device, or absent temperature fails that device only. Set `LINUX_DISK_NOCHECK=standby` to avoid waking sleeping SATA/SAS disks; keep the default `never` for always-on devices such as the CD6. diff --git a/docs/TROUBLESHOOTING.md b/docs/TROUBLESHOOTING.md index 8c8e85d..a330107 100644 --- a/docs/TROUBLESHOOTING.md +++ b/docs/TROUBLESHOOTING.md @@ -147,7 +147,7 @@ docker compose run --rm --entrypoint sh idrac-fan-control -c \ 'ls -l /dev/nvme1 && smartctl -A -j /dev/nvme1 | jq .temperature' ``` -Host 成功、容器失敗通常代表 `docker-compose.yml` 未加入 `/dev/nvme1:/dev/nvme1`。`LINUX_DISK_DEVICES` 必須是 `/dev/...` 逗號清單;控制器會拒絕空白、shell metacharacter 與相對路徑。smartctl 的非零狀態可能是 SMART health bitmask,因此只要 JSON 仍含有效溫度,控制器會保留該讀值;timeout、無效 JSON 或沒有溫度才使該 device 失敗。 +Host 成功、容器失敗通常代表 `docker-compose.yml` 未加入 `/dev/nvme1:/dev/nvme1`。`LINUX_DISK_DEVICES` 必須是 `/dev/...` 逗號清單;控制器會拒絕空白、shell metacharacter、相對路徑,以及含 `.`、`..` 或空元件的路徑。smartctl 的非零狀態可能是 SMART health bitmask,因此只要 JSON 仍含有效溫度,控制器會保留該讀值;timeout、無效 JSON 或沒有溫度才使該 device 失敗。 ## NVIDIA GPU 深入檢查 diff --git a/src/FanControlWithEsxiSmart.sh b/src/FanControlWithEsxiSmart.sh index 3cd1474..c4e3f1a 100755 --- a/src/FanControlWithEsxiSmart.sh +++ b/src/FanControlWithEsxiSmart.sh @@ -774,6 +774,22 @@ temperature_source_gpu_validate() { temperature_source_gpu_collect() { get_gpu_temperatures; } temperature_source_gpu_adjust() { apply_temperature_offset "$1" "$GPU_TEMP_OFFSET"; } +is_linux_device_path() { + local path="$1" + local relative + local component + local -a components + + [[ "$path" =~ ^/dev/[A-Za-z0-9._+:/-]+$ ]] || return 1 + relative="${path#/dev/}" + [[ -n "$relative" && "$relative" != */ && "$relative" != *//* ]] || return 1 + + IFS='/' read -r -a components <<< "$relative" + for component in "${components[@]}"; do + [[ "$component" != "." && "$component" != ".." ]] || return 1 + done +} + temperature_source_linux_disk_validate() { local devices local device @@ -784,7 +800,7 @@ temperature_source_linux_disk_validate() { devices="$(normalize_csv_list "$LINUX_DISK_DEVICES")" IFS=',' read -r -a device_list <<< "$devices" for device in "${device_list[@]}"; do - if [[ ! "$device" =~ ^/dev/[A-Za-z0-9._/+:-]+$ ]]; then + if ! is_linux_device_path "$device"; then log "ERROR" "LINUX_DISK_DEVICES contains an invalid device path: ${device}" error=1 fi diff --git a/tests/fan-control.test.sh b/tests/fan-control.test.sh index 9551aa5..5c5c17d 100755 --- a/tests/fan-control.test.sh +++ b/tests/fan-control.test.sh @@ -259,6 +259,32 @@ test_linux_and_remote_sources_validate_configuration() { ) && fail "validate_config should reject unsafe Linux device paths" pass + ( + OPERATION_MODE=auto + TEMPERATURE_SOURCES=linux_disk + WITH_GPU_TEMP=false + DRY_RUN=true + IDRAC_IP=10.0.0.10 + IDRAC_ID=root + IDRAC_PASSWORD=test-password + LINUX_DISK_DEVICES=/dev/../etc/passwd + validate_config auto >/dev/null 2>&1 + ) && fail "validate_config should reject parent traversal in Linux device paths" + pass + + ( + OPERATION_MODE=auto + TEMPERATURE_SOURCES=linux_disk + WITH_GPU_TEMP=false + DRY_RUN=true + IDRAC_IP=10.0.0.10 + IDRAC_ID=root + IDRAC_PASSWORD=test-password + LINUX_DISK_DEVICES=/dev/disk/by-id/nvme-KIOXIA_CD6 + validate_config auto >/dev/null 2>&1 + ) || fail "validate_config should accept nested Linux device paths" + pass + ( OPERATION_MODE=auto TEMPERATURE_SOURCES=linux_disk