diff --git a/.env.example b/.env.example index 7c030f7..3118b0a 100644 --- a/.env.example +++ b/.env.example @@ -32,10 +32,13 @@ RESTORE_AUTO_ON_EXIT=true # ----------------------------------------------------------------------------- # Temperature sources # ----------------------------------------------------------------------------- -# Supported values: esxi, idrac, gpu. Use comma-separated values for hybrid mode. +# Supported values: esxi, idrac, gpu, linux_disk, remote_gpu. +# Use comma-separated values for hybrid mode. # esxi reads one NVMe SMART temperature through ESXi SSH. # idrac reads all iDRAC temperature sensors through ipmitool. -# gpu reads NVIDIA GPU temperatures through nvidia-smi. +# gpu reads NVIDIA GPU temperatures on the controller host. +# linux_disk reads one or more local Linux disks through smartctl JSON output. +# remote_gpu reads NVIDIA GPU temperatures from one or more Linux VMs over SSH. TEMPERATURE_SOURCES=esxi # Backward-compatible GPU switch. If true, gpu is appended to TEMPERATURE_SOURCES. @@ -56,10 +59,34 @@ ESXI_USERNAME=root ESXI_PASSWORD=change-me ESXI_SSH_KEY= ESXI_SSH_PORT=22 +# Shared by the ESXi and remote GPU SSH transports. SSH_CONNECT_TIMEOUT=10 SSH_STRICT_HOST_KEY_CHECKING=accept-new DRIVE_DEVICE=t10.NVMe____replace_with_your_device_identifier +# ----------------------------------------------------------------------------- +# Local Linux disk source settings +# Required only when TEMPERATURE_SOURCES includes linux_disk. +# ----------------------------------------------------------------------------- +# Use controller device nodes (for NVMe, /dev/nvmeN is preferred). In Docker, +# map every listed host device into the container with Compose `devices`. +LINUX_DISK_DEVICES=/dev/nvme1 +LINUX_DISK_TEMP_OFFSET=0 +# Set standby to avoid waking sleeping SATA/SAS disks. Use never for NVMe. +LINUX_DISK_NOCHECK=never + +# ----------------------------------------------------------------------------- +# Remote NVIDIA GPU source settings +# Required only when TEMPERATURE_SOURCES includes remote_gpu. +# ----------------------------------------------------------------------------- +# Every host must provide nvidia-smi. One SSH identity is shared by all hosts. +REMOTE_GPU_HOSTS=192.0.2.31,192.0.2.32 +REMOTE_GPU_USERNAME=root +REMOTE_GPU_PASSWORD= +REMOTE_GPU_SSH_KEY=/run/secrets/gpu_vms_ed25519 +REMOTE_GPU_SSH_PORT=22 +REMOTE_GPU_TEMP_OFFSET=15 + # ----------------------------------------------------------------------------- # Fan curve # ----------------------------------------------------------------------------- diff --git a/.gitattributes b/.gitattributes new file mode 100644 index 0000000..3981d63 --- /dev/null +++ b/.gitattributes @@ -0,0 +1,4 @@ +*.sh text eol=lf +Dockerfile text eol=lf +*.yml text eol=lf +*.yaml text eol=lf diff --git a/Dockerfile b/Dockerfile index e0fab70..2f95ecc 100644 --- a/Dockerfile +++ b/Dockerfile @@ -2,7 +2,7 @@ ARG BASE_IMAGE=nvidia/cuda:12.9.0-runtime-ubuntu24.04 FROM ${BASE_IMAGE} LABEL org.opencontainers.image.title="iDRAC Fan Speed Control" \ - org.opencontainers.image.description="Automatic Dell iDRAC fan control using IPMI with ESXi, iDRAC, and optional NVIDIA GPU temperature sources" \ + org.opencontainers.image.description="Automatic Dell iDRAC fan control with pluggable disk, iDRAC, and NVIDIA GPU temperature sources" \ org.opencontainers.image.source="https://github.com/DF-wu/iDRACFanSpeedControl" ENV DEBIAN_FRONTEND=noninteractive \ @@ -13,8 +13,10 @@ RUN apt-get update && apt-get install -y --no-install-recommends \ ca-certificates \ coreutils \ ipmitool \ + jq \ openssh-client \ procps \ + smartmontools \ sshpass \ tini \ && rm -rf /var/lib/apt/lists/* diff --git a/README.md b/README.md index 8d02a0c..e314cc0 100644 --- a/README.md +++ b/README.md @@ -6,7 +6,7 @@ [![Container](https://img.shields.io/badge/container-GHCR-blue)](https://github.com/DF-wu/iDRACFanSpeedControl/pkgs/container/idrac-fan-control) [![License](https://img.shields.io/badge/license-MIT-green)](LICENSE) -A fan controller for Dell PowerEdge servers. It sets the fan duty cycle through iDRAC IPMI OEM raw commands, then selects a fan-curve level using one or more temperature sources: ESXi NVMe SMART, iDRAC temperature sensors, or NVIDIA GPUs. +A fan controller for Dell PowerEdge servers. It sets the fan duty cycle through iDRAC IPMI OEM raw commands, then selects a fan-curve level from pluggable temperature sources: ESXi NVMe SMART, iDRAC sensors, local Linux disks, and local or remote NVIDIA GPUs. > [!CAUTION] > This program temporarily overrides Dell's automatic fan control. During initial setup, keep the iDRAC Web UI or a physical console available. Verify `manual`, `restore`, and `diagnose` before running `auto` for an extended period. If the server overheats, readings become unreliable, or behavior is abnormal, run `restore` immediately and stop the container. @@ -21,11 +21,12 @@ flowchart LR CTRL --> IPMI[iDRAC / IPMI\nfan raw command] CTRL --> ESXI[ESXi\nesxcli SMART] CTRL --> SDR[iDRAC\nTemperature SDR] - CTRL --> GPU[NVIDIA\nnvidia-smi] + CTRL --> DISK[Linux disks\nsmartctl JSON] + CTRL --> GPU[Local/remote NVIDIA\nnvidia-smi] CTRL --> LOG[logs/fan_control.log\nhealthcheck] ``` -The controller sends fan commands only to iDRAC; all temperature sources are read-only. When multiple sources are enabled, it uses the highest valid decision temperature. The GPU reading is adjusted by `GPU_TEMP_OFFSET` first, preventing the GPU package temperature from driving chassis fans unnecessarily fast. +The controller sends fan commands only to iDRAC; all temperature sources are read-only. Every provider implements the same `validate`, `collect`, and `adjust` interface. When multiple sources are enabled, the controller uses the highest valid adjusted temperature. GPU and disk offsets let unlike sensors share one fan curve without source-specific decision code. ![iDRAC IPMI over LAN settings](images/image.png) @@ -50,12 +51,14 @@ The main menu looks like this. Each submenu writes changes back to `.env`, and t 1) Quick setup wizard 6) Safety, timing, and logging 2) iDRAC / IPMI settings 7) Review redacted configuration - 3) Temperature source 8) Validate configuration - 4) ESXi NVMe source 9) Run read-only diagnostics - 5) Fan curve 0) Save and exit + 3) Temperature source 8) Safety, timing, and logging + 4) ESXi NVMe source 9) Review redacted configuration + 5) Local Linux disks 10) Validate configuration + 6) Remote NVIDIA GPUs 11) Run read-only diagnostics + 7) Fan curve 0) Save and exit ``` -For your first setup, select `iDRAC sensors only` and leave ESXi and GPU sources disabled. After completing the TUI, run: +For your first setup, select `iDRAC sensors only` and leave external sources disabled. After completing the TUI, run: ```bash make validate @@ -125,7 +128,7 @@ src/FanControlWithEsxiSmart.sh diagnose src/FanControlWithEsxiSmart.sh config ``` -Local execution requires `bash`, `coreutils`, `ipmitool`, and `timeout`. ESXi password authentication also requires `sshpass`. The Docker image includes these dependencies. +Local execution requires `bash`, `coreutils`, `ipmitool`, and `timeout`. Linux disk mode also requires `smartctl` and `jq`; SSH password authentication requires `sshpass`. The Docker image includes these dependencies. ## Safe startup sequence @@ -157,13 +160,15 @@ Before unattended operation, confirm each item: ## Temperature sources and control decisions -`TEMPERATURE_SOURCES` is a comma-separated list containing `esxi`, `idrac`, and/or `gpu`. Sources can be combined, and the controller retains each source label for logging and diagnostics. +`TEMPERATURE_SOURCES` is a comma-separated list containing `esxi`, `idrac`, `gpu`, `linux_disk`, and/or `remote_gpu`. Sources can be combined, and the controller retains each source label for logging and diagnostics. See [Temperature source interface](docs/TEMPERATURE_SOURCES.md) for the extension contract and deployment examples. | Source | Reading | Requirements | Behavior on failure | | --- | --- | --- | --- | | `idrac` | Readable sensors from `ipmitool sdr type Temperature` | iDRAC IPMI | Marks this source as failed | | `esxi` | `esxcli storage core device smart get` for the configured NVMe device | SSH and `DRIVE_DEVICE` | Marks this source as failed | -| `gpu` | Temperature of each GPU reported by `nvidia-smi` | NVIDIA Container Toolkit and driver | Marks this source as failed | +| `gpu` | Temperature of each local GPU reported by `nvidia-smi` | NVIDIA Container Toolkit and driver | Marks this source as failed | +| `linux_disk` | SMART temperature for each configured Linux device | `smartctl`, `jq`, and device access | Fails unreadable devices; keeps other valid disks | +| `remote_gpu` | Every GPU reported by `nvidia-smi` on each configured VM | SSH credentials and remote NVIDIA driver | Fails unreachable hosts; keeps other valid VMs | ```mermaid flowchart TD @@ -191,7 +196,7 @@ The default curve is shown below. `validate` ensures that thresholds are strictl | `critical` | `>=80°C` | 60% | | `failsafe` | All sources fail | 70% | -Moving to a lower fan level requires the temperature to fall by `HYSTERESIS` degrees below the relevant threshold, preventing repeated speed changes near a boundary. Moving to a higher level is immediate. The adjusted GPU value is `GPU_TEMP - GPU_TEMP_OFFSET`, with a minimum of 0°C. +Moving to a lower fan level requires the temperature to fall by `HYSTERESIS` degrees below the relevant threshold, preventing repeated speed changes near a boundary. Moving to a higher level is immediate. Each source owns its adjustment: local and remote GPUs subtract their configured offset, Linux disks subtract `LINUX_DISK_TEMP_OFFSET`, and all results have a minimum of 0°C. ## Configuration reference @@ -214,7 +219,7 @@ Moving to a lower fan level requires the temperature to fall by `HYSTERESIS` deg | Variable | Default | Description | | --- | --- | --- | -| `TEMPERATURE_SOURCES` | `esxi` | Comma-separated list of `esxi,idrac,gpu` | +| `TEMPERATURE_SOURCES` | `esxi` | Comma-separated list of registered source IDs | | `WITH_GPU_TEMP` | `false` | Legacy compatibility switch; `true` appends `gpu` | | `GPU_TEMP_OFFSET` | `15` | Offset subtracted from GPU temperature | | `ESXI_HOST` / `ESXI_USERNAME` | Empty / `root` | ESXi SSH target | @@ -224,6 +229,13 @@ Moving to a lower fan level requires the temperature to fall by `HYSTERESIS` deg | `SSH_CONNECT_TIMEOUT` | `10` | SSH connection timeout in seconds | | `SSH_STRICT_HOST_KEY_CHECKING` | `accept-new` | `yes`, `no`, `ask`, or `accept-new` | | `DRIVE_DEVICE` | Empty | Full ID returned by `esxcli storage core device list` | +| `LINUX_DISK_DEVICES` | Empty | Comma-separated Linux device paths, such as `/dev/nvme1,/dev/sdb` | +| `LINUX_DISK_TEMP_OFFSET` | `0` | Offset subtracted from local disk temperatures | +| `LINUX_DISK_NOCHECK` | `never` | smartctl power-mode check; `standby` avoids waking sleeping disks | +| `REMOTE_GPU_HOSTS` | Empty | Comma-separated Linux VM hostnames or addresses | +| `REMOTE_GPU_USERNAME` / `REMOTE_GPU_SSH_PORT` | `root` / `22` | SSH identity shared by remote GPU hosts | +| `REMOTE_GPU_PASSWORD` / `REMOTE_GPU_SSH_KEY` | Empty / Empty | Remote GPU SSH authentication; keys are preferred | +| `REMOTE_GPU_TEMP_OFFSET` | `15` | Offset subtracted from remote GPU temperatures | | `IDRAC_SENSOR_INCLUDE_REGEX` | Empty | awk regex used to keep matching sensor names only | | `IDRAC_SENSOR_EXCLUDE_REGEX` | `no reading\|disabled\|not readable` | Excludes invalid SDR entries | @@ -241,7 +253,7 @@ Moving to a lower fan level requires the temperature to fall by `HYSTERESIS` deg | `LOG_LEVEL` | `INFO` | `DEBUG` adds command and source details but never logs passwords | | `HEALTHCHECK_MAX_AGE` | `0` | `0` means `CHECK_INTERVAL*3 + COMMAND_TIMEOUT` | -## Docker deployment and GPU support +## Docker deployment, Linux disks, and GPUs The Compose configuration uses the GHCR image, host networking, and a `./logs` volume by default: @@ -274,6 +286,17 @@ If GPU support is unnecessary, build locally without CUDA to reduce the image si docker build --build-arg BASE_IMAGE=ubuntu:24.04 -t idrac-fan-control:local . ``` +Linux disk mode needs a device mapping for every entry in `LINUX_DISK_DEVICES`. For example: + +```yaml +services: + idrac-fan-control: + devices: + - /dev/nvme1:/dev/nvme1 +``` + +For a TrueNAS CD6 plus NVIDIA GPUs in other VMs, use `TEMPERATURE_SOURCES=linux_disk,remote_gpu`, map the CD6 controller device, and mount a read-only SSH key for the GPU VMs. The complete example is in [docs/TEMPERATURE_SOURCES.md](docs/TEMPERATURE_SOURCES.md). + ## Debugging, logs, and troubleshooting Example normal log entry: @@ -312,7 +335,7 @@ docker compose run --rm idrac-fan-control diagnose ## Security and backups - Never commit `.env`, `logs/`, private keys, or incident dumps. The TUI sets configuration file permissions to `600`. -- Compose mounts the gitignored `./secrets` directory read-only at `/run/secrets`. Use the in-container path for ESXi keys, such as `/run/secrets/esxi_ed25519`. +- Compose mounts the gitignored `./secrets` directory read-only at `/run/secrets`. Use in-container paths such as `/run/secrets/esxi_ed25519` or `/run/secrets/gpu_vms_ed25519`. - Prefer `ESXI_SSH_KEY`. When password authentication is required, the controller uses the `SSHPASS` environment variable with `sshpass -e`, keeping the password out of argv. - The iDRAC password is supplied through the `IPMI_PASSWORD` environment variable and `ipmitool -E`. The `config` command and TUI review show only whether it is set and its character count. - Keep iDRAC and ESXi on an isolated management network. Never expose IPMI over LAN to the public internet. @@ -331,13 +354,13 @@ Copy the complete identifier into `DRIVE_DEVICE`; do not shorten it. If the SMAR ## Testing and quality gates ```bash -make test # bash -n + 37 core assertions + 10 TUI assertions +make test # bash -n + 52 core assertions + 14 TUI assertions make validate # Check the current .env with Docker dependencies; does not change fans make validate-example # DRY_RUN smoke test that does not require .env make docker-build ``` -Tests cover source normalization, SDR parsing, GPU offset, hysteresis, fail-safe behavior, curve/port/regex validation, credential redaction, health freshness, diagnostic previews, and safe round-tripping of TUI configuration files. Real hardware, ESXi SSH, and GPUs must still be verified with `diagnose` on your management network. +Tests cover the source interface, Linux SMART collection, multi-host remote GPUs, source offsets, SDR parsing, hysteresis, fail-safe behavior, configuration validation, credential redaction, health freshness, diagnostics, and safe TUI configuration round-trips. Real hardware, SSH targets, disks, iDRAC, and GPUs must still be verified with `diagnose` on your management network. ## Project structure @@ -351,6 +374,7 @@ Tests cover source normalization, SDR parsing, GPU offset, hysteresis, fail-safe │ ├── fan-control.test.sh # Core logic and safety tests │ └── tui.test.sh # .env parser/writer tests ├── docs/ +│ ├── TEMPERATURE_SOURCES.md # Source interface and deployment examples │ └── TROUBLESHOOTING.md # Symptom-based troubleshooting guide ├── images/image.png # iDRAC IPMI settings screenshot ├── .env.example # Fully commented configuration template @@ -364,10 +388,10 @@ Tests cover source normalization, SDR parsing, GPU offset, hysteresis, fail-safe ## Compatibility and limitations -The project primarily targets Dell PowerEdge R730/R730xd-class systems with iDRAC 8 OEM fan raw commands. Other generations may be compatible, but identical behavior must not be assumed. Complete `manual`, `restore`, and `diagnose` checks before unattended operation. Actual iDRAC, ESXi, and GPU sensor names and permissions vary by firmware and driver, so rely on diagnostic output from your own environment. +The project primarily targets Dell PowerEdge R730/R730xd-class systems with iDRAC 8 OEM fan raw commands. Other generations may be compatible, but identical behavior must not be assumed. Complete `manual`, `restore`, and `diagnose` checks before unattended operation. Sensor names, disk permissions, SSH access, and driver behavior vary by environment, so rely on diagnostic output from your own deployment. ## License MIT. See [LICENSE](LICENSE). -Documentation last reviewed: July 18, 2026. +Documentation last reviewed: August 15, 2026. diff --git a/README.zh-TW.md b/README.zh-TW.md index 585f537..f7870f5 100644 --- a/README.zh-TW.md +++ b/README.zh-TW.md @@ -6,7 +6,7 @@ [![Container](https://img.shields.io/badge/container-GHCR-blue)](https://github.com/DF-wu/iDRACFanSpeedControl/pkgs/container/idrac-fan-control) [![License](https://img.shields.io/badge/license-MIT-green)](LICENSE) -這是一個給 Dell PowerEdge 使用的風扇控制器。它透過 iDRAC 的 IPMI OEM raw command 設定 fan duty cycle,再用 ESXi NVMe SMART、iDRAC temperature sensor、NVIDIA GPU 中任一個或多個來源決定風扇曲線。 +這是一個給 Dell PowerEdge 使用的風扇控制器。它透過 iDRAC 的 IPMI OEM raw command 設定 fan duty cycle,再以可插拔溫度來源決定風扇曲線:ESXi NVMe SMART、iDRAC sensor、本機 Linux 磁碟,以及本機或遠端 NVIDIA GPU。 > [!CAUTION] > 這個程式會暫時覆寫 Dell 原廠風扇控制。第一次設定請保留 iDRAC Web UI 或實體主控台,先用 `manual`、`restore`、`diagnose` 驗證,再讓 `auto` 長時間執行。任何過熱、讀值不可信或行為異常時,立即執行 `restore` 並停止容器。 @@ -21,11 +21,12 @@ flowchart LR CTRL --> IPMI[iDRAC / IPMI\nfan raw command] CTRL --> ESXI[ESXi\nesxcli SMART] CTRL --> SDR[iDRAC\nTemperature SDR] - CTRL --> GPU[NVIDIA\nnvidia-smi] + CTRL --> DISK[Linux 磁碟\nsmartctl JSON] + CTRL --> GPU[本機/遠端 NVIDIA\nnvidia-smi] CTRL --> LOG[logs/fan_control.log\nhealthcheck] ``` -控制器只會對 iDRAC 發送風扇命令;溫度來源全部是讀取。多來源時採用最高的有效決策溫度,GPU 會先減去 `GPU_TEMP_OFFSET`,避免 GPU 的封裝溫度直接把機箱風扇推到不必要的高轉速。 +控制器只會對 iDRAC 發送風扇命令;溫度來源全部是讀取。每個 provider 都實作同一組 `validate`、`collect`、`adjust` 介面。多來源時採用最高的有效 adjusted temperature;GPU 與磁碟可各自套用 offset,不需在決策流程加入來源特例。 ![iDRAC IPMI over LAN 設定畫面](images/image.png) @@ -50,12 +51,14 @@ make tui 1) Quick setup wizard 6) Safety, timing, and logging 2) iDRAC / IPMI settings 7) Review redacted configuration - 3) Temperature source 8) Validate configuration - 4) ESXi NVMe source 9) Run read-only diagnostics - 5) Fan curve 0) Save and exit + 3) Temperature source 8) Safety, timing, and logging + 4) ESXi NVMe source 9) Review redacted configuration + 5) Local Linux disks 10) Validate configuration + 6) Remote NVIDIA GPUs 11) Run read-only diagnostics + 7) Fan curve 0) Save and exit ``` -第一次建議選 `iDRAC sensors only`,先不要把 ESXi 或 GPU 加入決策。TUI 完成後,依序執行: +第一次建議選 `iDRAC sensors only`,先不要把外部來源加入決策。TUI 完成後,依序執行: ```bash make validate @@ -125,7 +128,7 @@ src/FanControlWithEsxiSmart.sh diagnose src/FanControlWithEsxiSmart.sh config ``` -本機執行需要 `bash`、`coreutils`、`ipmitool`、`timeout`,ESXi password 模式另需 `sshpass`;Docker image 已包含這些依賴。 +本機執行需要 `bash`、`coreutils`、`ipmitool`、`timeout`。Linux 磁碟模式另需 `smartctl` 與 `jq`;SSH password 模式另需 `sshpass`。Docker image 已包含這些依賴。 ## 安全啟動順序 @@ -157,13 +160,15 @@ sequenceDiagram ## 溫度來源與控制決策 -`TEMPERATURE_SOURCES` 使用逗號分隔,可填 `esxi`、`idrac`、`gpu`。來源可以混合;控制器會保留每個來源的 label 供 log 與診斷追蹤。 +`TEMPERATURE_SOURCES` 使用逗號分隔,可填 `esxi`、`idrac`、`gpu`、`linux_disk`、`remote_gpu`。來源可以混合;控制器會保留每個來源的 label 供 log 與診斷追蹤。擴充介面與部署範例請見 [Temperature source interface](docs/TEMPERATURE_SOURCES.md)。 | 來源 | 讀取內容 | 必要條件 | 失敗時的行為 | | --- | --- | --- | --- | | `idrac` | `ipmitool sdr type Temperature` 中可讀的 sensor | iDRAC IPMI | 該來源標記失敗 | | `esxi` | 指定 NVMe 的 `esxcli storage core device smart get` | SSH、`DRIVE_DEVICE` | 該來源標記失敗 | -| `gpu` | `nvidia-smi` 的每張 GPU 溫度 | NVIDIA Container Toolkit/驅動 | 該來源標記失敗 | +| `gpu` | 本機 `nvidia-smi` 的每張 GPU 溫度 | NVIDIA Container Toolkit/驅動 | 該來源標記失敗 | +| `linux_disk` | 每個指定 Linux device 的 SMART 溫度 | `smartctl`、`jq`、device 權限 | 單一磁碟失敗,保留其他有效讀值 | +| `remote_gpu` | 每台指定 VM 上 `nvidia-smi` 回報的所有 GPU | SSH 與遠端 NVIDIA driver | 單一 VM 失敗,保留其他有效讀值 | ```mermaid flowchart TD @@ -191,7 +196,7 @@ flowchart TD | `critical` | `>=80°C` | 60% | | `failsafe` | 所有來源失敗 | 70% | -降檔會等待 `HYSTERESIS` 度,避免溫度在閾值附近來回跳速;升檔不延遲。GPU 的 adjusted 值為 `GPU_TEMP - GPU_TEMP_OFFSET`,最低不會小於 0°C。 +降檔會等待 `HYSTERESIS` 度,避免溫度在閾值附近來回跳速;升檔不延遲。每個來源自行調整讀值:本機與遠端 GPU 分別扣除各自 offset,Linux 磁碟扣除 `LINUX_DISK_TEMP_OFFSET`,最低都不會小於 0°C。 ## 設定參考 @@ -214,7 +219,7 @@ flowchart TD | 變數 | 預設 | 說明 | | --- | --- | --- | -| `TEMPERATURE_SOURCES` | `esxi` | `esxi,idrac,gpu` 的逗號清單 | +| `TEMPERATURE_SOURCES` | `esxi` | 已註冊 source ID 的逗號清單 | | `WITH_GPU_TEMP` | `false` | 舊版相容開關;true 會附加 `gpu` | | `GPU_TEMP_OFFSET` | `15` | GPU 溫度的補償度數 | | `ESXI_HOST` / `ESXI_USERNAME` | 空 / `root` | ESXi SSH 目標 | @@ -224,6 +229,13 @@ flowchart TD | `SSH_CONNECT_TIMEOUT` | `10` | SSH 建立連線上限(秒) | | `SSH_STRICT_HOST_KEY_CHECKING` | `accept-new` | `yes`、`no`、`ask` 或 `accept-new` | | `DRIVE_DEVICE` | 空 | `esxcli storage core device list` 找到的完整 ID | +| `LINUX_DISK_DEVICES` | 空 | Linux device path 逗號清單,例如 `/dev/nvme1,/dev/sdb` | +| `LINUX_DISK_TEMP_OFFSET` | `0` | 從本機磁碟溫度扣除的 offset | +| `LINUX_DISK_NOCHECK` | `never` | smartctl power mode check;`standby` 可避免喚醒休眠磁碟 | +| `REMOTE_GPU_HOSTS` | 空 | Linux GPU VM hostname 或 IP 逗號清單 | +| `REMOTE_GPU_USERNAME` / `REMOTE_GPU_SSH_PORT` | `root` / `22` | 遠端 GPU 主機共用的 SSH identity | +| `REMOTE_GPU_PASSWORD` / `REMOTE_GPU_SSH_KEY` | 空 / 空 | 遠端 GPU SSH 認證;建議使用 key | +| `REMOTE_GPU_TEMP_OFFSET` | `15` | 從遠端 GPU 溫度扣除的 offset | | `IDRAC_SENSOR_INCLUDE_REGEX` | 空 | 只保留符合 sensor 名稱的 awk regex | | `IDRAC_SENSOR_EXCLUDE_REGEX` | `no reading\|disabled\|not readable` | 排除無效 SDR | @@ -241,7 +253,7 @@ flowchart TD | `LOG_LEVEL` | `INFO` | `DEBUG` 會增加命令與來源細節,絕不列密碼 | | `HEALTHCHECK_MAX_AGE` | `0` | 0 代表 `CHECK_INTERVAL*3 + COMMAND_TIMEOUT` | -## Docker 部署與 GPU +## Docker 部署、Linux 磁碟與 GPU Compose 預設使用 GHCR image、host network 與 `./logs` volume: @@ -274,6 +286,17 @@ docker compose run --rm idrac-fan-control diagnose docker build --build-arg BASE_IMAGE=ubuntu:24.04 -t idrac-fan-control:local . ``` +Linux 磁碟模式必須將 `LINUX_DISK_DEVICES` 的每個 device 映射進容器,例如: + +```yaml +services: + idrac-fan-control: + devices: + - /dev/nvme1:/dev/nvme1 +``` + +若要在 TrueNAS 將 CD6 與其他 VM 的 NVIDIA GPU 納入同一決策鏈,設定 `TEMPERATURE_SOURCES=linux_disk,remote_gpu`、映射 CD6 controller device,並以唯讀 volume 掛載 GPU VM SSH key。完整範例請見 [docs/TEMPERATURE_SOURCES.md](docs/TEMPERATURE_SOURCES.md)。 + ## Debug、log 與故障排除 正常 log 範例: @@ -312,7 +335,7 @@ docker compose run --rm idrac-fan-control diagnose ## 安全與備份 - `.env`、`logs/`、私鑰與 incident dump 都不應提交到 Git;TUI 會將設定檔權限設為 `600`。 -- Compose 會以 read-only 掛載 gitignored 的 `./secrets` 到 `/run/secrets`;ESXi key 請使用容器內路徑(例如 `/run/secrets/esxi_ed25519`)。 +- Compose 會以 read-only 掛載 gitignored 的 `./secrets` 到 `/run/secrets`;請使用 `/run/secrets/esxi_ed25519` 或 `/run/secrets/gpu_vms_ed25519` 這類容器內路徑。 - 優先使用 `ESXI_SSH_KEY`;若必須用 password,控制器會以 `SSHPASS` environment 搭配 `sshpass -e`,不把密碼放進 argv。 - iDRAC 密碼透過 `IPMI_PASSWORD` environment 搭配 `ipmitool -E`;`config`/TUI review 只顯示 set 與字元數。 - iDRAC/ESXi 應位於隔離管理網路,避免把 IPMI over LAN 暴露到公網。 @@ -331,13 +354,13 @@ esxcli storage core device smart get -d 't10.NVMe____完整識別字串' ## 測試與品質門檻 ```bash -make test # bash -n + 37 個核心 assertions + 10 個 TUI assertions +make test # bash -n + 52 個核心 assertions + 14 個 TUI assertions make validate # 以 Docker 依賴檢查目前的 .env(不改變風扇) make validate-example # 不需要 .env 的 DRY_RUN smoke test make docker-build ``` -測試涵蓋 source normalization、SDR parsing、GPU offset、hysteresis、fail-safe、曲線/port/regex 驗證、credential redaction、health freshness、診斷 preview,以及 TUI 設定檔的安全 round-trip。真實硬體、ESXi SSH 和 GPU 仍須在你的管理網路上以 `diagnose` 驗證。 +測試涵蓋統一 source 介面、Linux SMART 收集、多主機 remote GPU、source offset、SDR parsing、hysteresis、fail-safe、設定驗證、credential redaction、health freshness、診斷 preview,以及 TUI 設定檔安全 round-trip。真實硬體、SSH target、磁碟、iDRAC 與 GPU 仍須在管理網路上以 `diagnose` 驗證。 ## 專案結構 @@ -351,6 +374,7 @@ make docker-build │ ├── fan-control.test.sh # 核心邏輯與安全測試 │ └── tui.test.sh # .env parser / writer 測試 ├── docs/ +│ ├── TEMPERATURE_SOURCES.md # Source 介面與部署範例 │ └── TROUBLESHOOTING.md # 症狀導向除錯手冊 ├── images/image.png # iDRAC IPMI 設定畫面 ├── .env.example # 帶完整註解的設定模板 @@ -364,10 +388,10 @@ make docker-build ## 相容性與限制 -專案以 Dell PowerEdge R730/R730xd、iDRAC 8 類機型的 OEM fan raw command 為主要驗證目標。其他世代可能相容,但不能假設相同;請在無人值守前完成 `manual`/`restore`/`diagnose`。iDRAC、ESXi、GPU 的實際 sensor 名稱與權限會依韌體和驅動改變,應以你自己的 diagnostic output 為準。 +專案以 Dell PowerEdge R730/R730xd、iDRAC 8 類機型的 OEM fan raw command 為主要驗證目標。其他世代可能相容,但不能假設相同;請在無人值守前完成 `manual`/`restore`/`diagnose`。Sensor 名稱、磁碟權限、SSH access 與 driver 行為會依環境改變,應以實際 diagnostic output 為準。 ## 授權 MIT,詳見 [LICENSE](LICENSE)。 -文件最後檢視:2026-07-18。 +文件最後檢視:2026-08-15。 diff --git a/USAGE_GUIDE.md b/USAGE_GUIDE.md index 1f8d5e0..0cd754c 100644 --- a/USAGE_GUIDE.md +++ b/USAGE_GUIDE.md @@ -124,6 +124,38 @@ cp ~/.ssh/esxi_ed25519 secrets/esxi_ed25519 chmod 600 secrets/esxi_ed25519 ``` +### TrueNAS CD6 與其他 VM 的 GPU + +目前 TrueNAS 25.10 主機上的 Kioxia CD6 controller device 是 `/dev/nvme1`。先在 host 唯讀確認: + +```bash +sudo smartctl -A -j /dev/nvme1 | jq '.temperature.current' +``` + +設定 CD6 與多台 NVIDIA VM: + +```dotenv +TEMPERATURE_SOURCES=linux_disk,remote_gpu +LINUX_DISK_DEVICES=/dev/nvme1 +LINUX_DISK_TEMP_OFFSET=0 + +REMOTE_GPU_HOSTS=gpu-vm-1,gpu-vm-2 +REMOTE_GPU_USERNAME=monitor +REMOTE_GPU_SSH_KEY=/run/secrets/gpu_vms_ed25519 +REMOTE_GPU_PASSWORD= +REMOTE_GPU_SSH_PORT=22 +REMOTE_GPU_TEMP_OFFSET=15 +``` + +在 `docker-compose.yml` 的 service 加入精確 device mapping: + +```yaml +devices: + - /dev/nvme1:/dev/nvme1 +``` + +每台 GPU VM 必須讓該 SSH 帳號可執行唯讀的 `nvidia-smi --query-gpu=index,temperature.gpu --format=csv,noheader,nounits`。設定後先跑 `validate` 與 `diagnose`;預期同時看到 `source:linux_disk`、`source:remote_gpu` 和 decision preview。詳見 [Temperature source interface](docs/TEMPERATURE_SOURCES.md)。 + ### 啟用 GPU ```dotenv @@ -178,4 +210,4 @@ make docker-build 長時間運行請確認 `logs/` 有輪替策略;專案只負責寫入單一 log,不會替主機設定 logrotate。建議以主機的 logrotate 或 journald retention 管理容量。 -文件最後檢視:2026-07-18。 +文件最後檢視:2026-08-15。 diff --git a/docker-compose.yml b/docker-compose.yml index 374b065..b17b84a 100644 --- a/docker-compose.yml +++ b/docker-compose.yml @@ -8,6 +8,10 @@ services: - ./logs:/var/log/fan-control # Put an ESXi private key in ./secrets and use /run/secrets/ in .env. - ./secrets:/run/secrets:ro + # Map each device listed in LINUX_DISK_DEVICES. Device mappings grant raw + # hardware access, so expose only the disks used by the controller. + # devices: + # - /dev/nvme1:/dev/nvme1 network_mode: host restart: unless-stopped init: true diff --git a/docs/TEMPERATURE_SOURCES.md b/docs/TEMPERATURE_SOURCES.md new file mode 100644 index 0000000..f6834e9 --- /dev/null +++ b/docs/TEMPERATURE_SOURCES.md @@ -0,0 +1,88 @@ +# Temperature source interface + +The controller treats every temperature provider through the same interface. `collect_temperature_readings` and `diagnose_mode` never branch on a source ID; they call the registered source methods instead. + +## Interface contract + +Register a source ID in `TEMPERATURE_SOURCE_IDS`, then implement these Bash functions in `src/FanControlWithEsxiSmart.sh`: + +```bash +temperature_source__validate() { ...; } +temperature_source__collect() { ...; } +temperature_source__adjust() { ...; } +``` + +- `validate` checks required configuration and runtime commands. It returns nonzero on an invalid setup. +- `collect` prints one or more tab-separated records and returns zero when at least one reading is valid. +- `adjust` receives one integer Celsius value and prints the integer used by the fan curve. + +Each collection record has this format: + +```text +source_idsensor_labelraw_temperature_celsius +``` + +Labels must not contain tabs or newlines. A source must reject missing, malformed, or unsupported readings instead of emitting `0`. The controller applies every source's `adjust` method, logs both raw and adjusted values when they differ, and selects the highest adjusted temperature. + +`temperature_source_interface_complete` checks the contract during validation. The supported source list is an allowlist, so an environment value cannot invoke an arbitrary shell function. + +## Built-in sources + +| ID | Provider | Adjustment | +| --- | --- | --- | +| `idrac` | All readable `ipmitool sdr type Temperature` sensors | None | +| `esxi` | One ESXi storage device queried through SSH and `esxcli` | None | +| `gpu` | All local NVIDIA GPUs queried with `nvidia-smi` | `GPU_TEMP_OFFSET` | +| `linux_disk` | One or more local Linux devices queried with `smartctl -A -j` | `LINUX_DISK_TEMP_OFFSET` | +| `remote_gpu` | All NVIDIA GPUs on one or more Linux hosts queried through SSH | `REMOTE_GPU_TEMP_OFFSET` | + +## Linux disk behavior + +Set `LINUX_DISK_DEVICES` to comma-separated `/dev` paths. Each path must contain only regular device-name components; the controller rejects empty, `.` and `..` components to prevent traversal outside `/dev`. Nested paths such as `/dev/disk/by-id/...` remain valid. For NVMe, prefer the controller node such as `/dev/nvme1`; namespace paths such as `/dev/nvme1n1` also work when smartctl supports them. + +The parser first reads smartctl's generic `temperature.current`, then checks the NVMe health log and ATA temperature attributes. It accepts a valid temperature even when smartctl's exit status reports disk health bits. A timeout, invalid JSON, inaccessible device, or absent temperature fails that device only. Set `LINUX_DISK_NOCHECK=standby` to avoid waking sleeping SATA/SAS disks; keep the default `never` for always-on devices such as the CD6. + +The source requires `smartctl`, `jq`, and `timeout`. The container image includes them. Containers also need an explicit device mapping for every configured disk: + +```yaml +services: + idrac-fan-control: + devices: + - /dev/nvme1:/dev/nvme1 +``` + +Grant access only to selected devices. Raw disk access is sensitive even though this controller invokes only the read-only `smartctl -A -j` query. + +## TrueNAS CD6 plus remote GPU VMs + +The following configuration combines a TrueNAS-hosted Kioxia CD6 with GPUs in two Linux VMs: + +```dotenv +TEMPERATURE_SOURCES=linux_disk,remote_gpu + +LINUX_DISK_DEVICES=/dev/nvme1 +LINUX_DISK_TEMP_OFFSET=0 +LINUX_DISK_NOCHECK=never + +REMOTE_GPU_HOSTS=gpu-vm-1,gpu-vm-2 +REMOTE_GPU_USERNAME=monitor +REMOTE_GPU_PASSWORD= +REMOTE_GPU_SSH_KEY=/run/secrets/gpu_vms_ed25519 +REMOTE_GPU_SSH_PORT=22 +REMOTE_GPU_TEMP_OFFSET=15 +``` + +Each remote host must have `nvidia-smi` in its non-interactive SSH `PATH`. Use a restricted monitoring account and SSH key where practical. The controller runs only this remote command: + +```bash +nvidia-smi --query-gpu=index,temperature.gpu --format=csv,noheader,nounits +``` + +Run `validate`, then `diagnose`. A healthy result includes one `source:linux_disk` row, one combined `source:remote_gpu` row, and a decision preview. A multi-device source passes when at least one configured target returns a reading and logs a warning for each failed target. `diagnose` returns nonzero when an enabled source returns no readings. Automatic control continues with remaining valid sources; if every source fails, the configured fail-safe applies. + +## Upstream references + +- [smartctl manual](https://github.com/smartmontools/smartmontools/blob/main/src/smartctl.8.in): Linux device forms, JSON output, and exit-status bitmask. +- [Docker Compose service `devices`](https://docs.docker.com/reference/compose-file/services/#devices): host-to-container device mapping syntax. +- [NVIDIA System Management Interface](https://docs.nvidia.com/deploy/nvidia-smi/index.html): selective GPU queries, CSV formatting, and Celsius temperature fields. +- [TrueNAS 25.10 disk API](https://api.truenas.com/v25.10/api_methods_disk.html): platform disk inventory and temperature methods used for deployment cross-checks. diff --git a/docs/TROUBLESHOOTING.md b/docs/TROUBLESHOOTING.md index 6e9bc94..a330107 100644 --- a/docs/TROUBLESHOOTING.md +++ b/docs/TROUBLESHOOTING.md @@ -36,7 +36,7 @@ docker logs --tail 200 idrac-fan-control tail -n 200 logs/fan_control.log ``` -需要命令階段與來源選擇細節時,在 `.env` 設 `LOG_LEVEL=DEBUG` 後重跑 `diagnose`。Debug log 不會列出 iDRAC/ESXi 密碼;提交 issue 前仍應檢查主機名稱、IP、device identifier 是否需要遮罩。 +需要命令階段與來源選擇細節時,在 `.env` 設 `LOG_LEVEL=DEBUG` 後重跑 `diagnose`。Debug log 不會列出 iDRAC、ESXi 或 remote GPU 密碼;提交 issue 前仍應檢查主機名稱、IP、device identifier 是否需要遮罩。 ## Diagnostic output 怎麼讀 @@ -67,6 +67,8 @@ Result: PASS - controller dependencies are ready. | `source:idrac ... no valid` | SDR 無 readable temperature 或 regex 過濾全部 | `status`、清空 include regex | 修正 filter;保留預設 exclude | | `source:esxi ... no valid` | SSH、device ID 或 SMART 格式錯 | ESXi 上直接跑 `esxcli ... smart get` | 修復 key/password、port、完整 ID | | `source:gpu ... no valid` | 容器無 GPU、驅動/Toolkit 未配置 | `docker ... nvidia-smi` | 啟用 `gpus: all` 與 Toolkit,或移除 gpu source | +| `source:linux_disk ... no valid` | device 未映射、權限不足、SMART 無溫度 | host 與容器內分別執行 `smartctl -A -j` | 加入精確 `devices` mapping,確認 `jq` 與路徑 | +| `source:remote_gpu ... no valid` | VM 不可達、SSH 認證或遠端 `nvidia-smi` 失敗 | 以相同 key 手動 SSH 執行 query | 修復 host/key/known_hosts/driver,或移除失敗 target | | `No valid temperature readings` | 所有 selected sources 都失敗 | `diagnose` | 修復至少一個 source;先保留 fail-safe | | `Fail-safe fan speed applied` | 上述全部失敗且 fail-safe 開啟 | 同時看前面的 source WARN | 修復來源;這不是曲線 level | | `Fan speeds must not decrease` | 高溫檔位比低溫檔位低 | TUI Review | 讓 idle ≤ low ≤ medium ≤ high ≤ critical | @@ -130,6 +132,23 @@ ssh -i /path/to/key -o BatchMode=yes root@ESXI_HOST true 容器要使用 host key file 時,還必須把 key 以 read-only volume 掛進 container,且 `.env` 的 `ESXI_SSH_KEY` 要填 container 內路徑;只填 host path 不會自動掛載。 +## Linux 磁碟深入檢查 + +先在 host 上執行與控制器相同的唯讀查詢: + +```bash +sudo smartctl -A -j /dev/nvme1 | jq '.temperature.current // .nvme_smart_health_information_log.temperature' +``` + +再確認容器看到同一個 device: + +```bash +docker compose run --rm --entrypoint sh idrac-fan-control -c \ + 'ls -l /dev/nvme1 && smartctl -A -j /dev/nvme1 | jq .temperature' +``` + +Host 成功、容器失敗通常代表 `docker-compose.yml` 未加入 `/dev/nvme1:/dev/nvme1`。`LINUX_DISK_DEVICES` 必須是 `/dev/...` 逗號清單;控制器會拒絕空白、shell metacharacter、相對路徑,以及含 `.`、`..` 或空元件的路徑。smartctl 的非零狀態可能是 SMART health bitmask,因此只要 JSON 仍含有效溫度,控制器會保留該讀值;timeout、無效 JSON 或沒有溫度才使該 device 失敗。 + ## NVIDIA GPU 深入檢查 ```bash @@ -140,6 +159,15 @@ docker compose run --rm idrac-fan-control diagnose 第一個失敗代表主機驅動問題;第一個成功、第二個失敗通常是 Container Toolkit;兩者成功但 Compose 失敗則檢查 `docker-compose.yml` 的 `gpus: all`。不用 GPU 做控制時,最安全的修復是從 `TEMPERATURE_SOURCES` 移除 `gpu`,而不是把錯誤隱藏。 +遠端 GPU 來源請從 controller host 或容器測試完全相同的命令: + +```bash +ssh -i /run/secrets/gpu_vms_ed25519 monitor@gpu-vm-1 \ + 'nvidia-smi --query-gpu=index,temperature.gpu --format=csv,noheader,nounits' +``` + +若互動式 SSH 成功但 controller 失敗,檢查 key 的容器內路徑、`REMOTE_GPU_USERNAME`、port 與 `SSH_STRICT_HOST_KEY_CHECKING`。遠端登入 shell 必須能在非互動 `PATH` 找到 `nvidia-smi`。 + ## Healthcheck 與 log auto mode 每次成功決策或套用 fail-safe 都會更新 `${LOG_DIR}/${LOG_FILE}`。預設允許的最大年齡為: @@ -173,4 +201,4 @@ docker inspect idrac-fan-control --format '{{json .Mounts}}' 請勿附上 `.env`、密碼、SSH private key、公開可達的管理 IP 或完整 service tag。 -文件最後檢視:2026-07-18。 +文件最後檢視:2026-08-15。 diff --git a/src/FanControlWithEsxiSmart.sh b/src/FanControlWithEsxiSmart.sh index e726b39..c4e3f1a 100755 --- a/src/FanControlWithEsxiSmart.sh +++ b/src/FanControlWithEsxiSmart.sh @@ -22,6 +22,17 @@ PROGRAM_NAME="$(basename "$0")" : "${SSH_STRICT_HOST_KEY_CHECKING:=accept-new}" : "${DRIVE_DEVICE:=}" +: "${LINUX_DISK_DEVICES:=}" +: "${LINUX_DISK_TEMP_OFFSET:=0}" +: "${LINUX_DISK_NOCHECK:=never}" + +: "${REMOTE_GPU_HOSTS:=}" +: "${REMOTE_GPU_USERNAME:=root}" +: "${REMOTE_GPU_PASSWORD:=}" +: "${REMOTE_GPU_SSH_KEY:=}" +: "${REMOTE_GPU_SSH_PORT:=22}" +: "${REMOTE_GPU_TEMP_OFFSET:=15}" + : "${OPERATION_MODE:=manual}" : "${TEMPERATURE_SOURCES:=esxi}" : "${WITH_GPU_TEMP:=false}" @@ -54,6 +65,9 @@ PROGRAM_NAME="$(basename "$0")" : "${HEALTHCHECK_MAX_AGE:=0}" : "${DRY_RUN:=false}" +readonly TEMPERATURE_SOURCE_INTERFACE_VERSION=1 +readonly TEMPERATURE_SOURCE_IDS="esxi idrac gpu linux_disk remote_gpu" + LAST_LEVEL="" LAST_SPEED="" @@ -234,6 +248,57 @@ source_enabled_in_list() { return 1 } +normalize_csv_list() { + local raw="${1:-}" + local normalized="" + local item + local -a items + + IFS=',' read -r -a items <<< "$raw" + for item in "${items[@]}"; do + item="$(trim "$item")" + [[ -z "$item" ]] && continue + if ! source_enabled_in_list "$item" "$normalized"; then + normalized="${normalized:+${normalized},}${item}" + fi + done + + printf '%s' "$normalized" +} + +temperature_source_supported() { + local wanted="$1" + local source + + for source in $TEMPERATURE_SOURCE_IDS; do + [[ "$source" == "$wanted" ]] && return 0 + done + return 1 +} + +temperature_source_method() { + local source="$1" + local method="$2" + shift 2 + local function_name="temperature_source_${source}_${method}" + + temperature_source_supported "$source" || return 1 + declare -F "$function_name" >/dev/null 2>&1 || { + log "ERROR" "Temperature source '${source}' does not implement ${method}() for interface v${TEMPERATURE_SOURCE_INTERFACE_VERSION}" + return 1 + } + "$function_name" "$@" +} + +temperature_source_interface_complete() { + local source="$1" + local method + + for method in validate collect adjust; do + declare -F "temperature_source_${source}_${method}" >/dev/null 2>&1 || return 1 + done +} + command_needs_ipmi() { case "$1" in auto|once|manual|restore|status|diagnose) return 0 ;; @@ -267,19 +332,19 @@ validate_sources() { sources="$(normalize_sources)" if [[ -z "$sources" ]]; then - log "ERROR" "TEMPERATURE_SOURCES must contain at least one source: esxi, idrac, gpu" + log "ERROR" "TEMPERATURE_SOURCES must contain at least one source: ${TEMPERATURE_SOURCE_IDS// /, }" return 1 fi IFS=',' read -r -a source_list <<< "$sources" for source in "${source_list[@]}"; do - case "$source" in - esxi|idrac|gpu) ;; - *) - log "ERROR" "Unsupported temperature source '${source}'. Use esxi, idrac, or gpu." - error=1 - ;; - esac + if ! temperature_source_supported "$source"; then + log "ERROR" "Unsupported temperature source '${source}'. Use ${TEMPERATURE_SOURCE_IDS// /, }." + error=1 + elif ! temperature_source_interface_complete "$source"; then + log "ERROR" "Temperature source '${source}' has an incomplete interface" + error=1 + fi done return "$error" @@ -332,7 +397,10 @@ validate_config() { validate_integer_range "CHECK_INTERVAL" "$CHECK_INTERVAL" 1 86400 || error=1 validate_integer_range "SSH_CONNECT_TIMEOUT" "$SSH_CONNECT_TIMEOUT" 1 300 || error=1 validate_integer_range "ESXI_SSH_PORT" "$ESXI_SSH_PORT" 1 65535 || error=1 + validate_integer_range "REMOTE_GPU_SSH_PORT" "$REMOTE_GPU_SSH_PORT" 1 65535 || error=1 validate_integer_range "GPU_TEMP_OFFSET" "$GPU_TEMP_OFFSET" 0 120 || error=1 + validate_integer_range "LINUX_DISK_TEMP_OFFSET" "$LINUX_DISK_TEMP_OFFSET" 0 120 || error=1 + validate_integer_range "REMOTE_GPU_TEMP_OFFSET" "$REMOTE_GPU_TEMP_OFFSET" 0 120 || error=1 validate_integer_range "HYSTERESIS" "$HYSTERESIS" 0 30 || error=1 validate_integer_range "HEALTHCHECK_MAX_AGE" "$HEALTHCHECK_MAX_AGE" 0 86400 || error=1 @@ -389,39 +457,12 @@ validate_config() { if command_needs_temperature "$command"; then validate_sources || error=1 sources="$(normalize_sources)" - - if source_enabled_in_list "esxi" "$sources"; then - require_value "ESXI_HOST" "$ESXI_HOST" || error=1 - require_value "ESXI_USERNAME" "$ESXI_USERNAME" || error=1 - require_value "DRIVE_DEVICE" "$DRIVE_DEVICE" || error=1 - if [[ -z "$ESXI_PASSWORD" && -z "$ESXI_SSH_KEY" ]]; then - log "ERROR" "Set ESXI_PASSWORD or ESXI_SSH_KEY when TEMPERATURE_SOURCES includes esxi" - error=1 - fi - if [[ -z "$ESXI_SSH_KEY" ]]; then - require_value "ESXI_PASSWORD" "$ESXI_PASSWORD" || error=1 - elif is_placeholder_value "$ESXI_SSH_KEY"; then - log "ERROR" "Replace placeholder value for ESXI_SSH_KEY" - error=1 - fi - if [[ -n "$ESXI_SSH_KEY" && ! -r "$ESXI_SSH_KEY" ]]; then - log "ERROR" "ESXI_SSH_KEY is not readable: ${ESXI_SSH_KEY}" - error=1 - fi - if ! is_true "$DRY_RUN"; then - require_command "ssh" || error=1 - require_command "timeout" || error=1 - if [[ -n "$ESXI_PASSWORD" && -z "$ESXI_SSH_KEY" ]]; then - require_command "sshpass" || error=1 - fi - fi - fi - - if source_enabled_in_list "gpu" "$sources" && ! is_true "$DRY_RUN"; then - if ! command -v nvidia-smi >/dev/null 2>&1; then - log "WARN" "GPU source is enabled but nvidia-smi is not available; fail-safe will be used if no other source succeeds" - fi - fi + local source + local -a source_list + IFS=',' read -r -a source_list <<< "$sources" + for source in "${source_list[@]}"; do + temperature_source_method "$source" validate || error=1 + done fi return "$error" @@ -468,24 +509,35 @@ remote_quote() { printf "'%s'" "${value//\'/\'\\\'\'}" } -run_esxi_command() { - local remote_command="$1" - local ssh_command=(ssh -p "$ESXI_SSH_PORT" -o "StrictHostKeyChecking=${SSH_STRICT_HOST_KEY_CHECKING}" -o "ConnectTimeout=${SSH_CONNECT_TIMEOUT}") - - if [[ -n "$ESXI_SSH_KEY" ]]; then - ssh_command+=(-i "$ESXI_SSH_KEY" -o BatchMode=yes) +run_ssh_command() { + local context="$1" + local host="$2" + local username="$3" + local password="$4" + local key="$5" + local port="$6" + local remote_command="$7" + local ssh_command=(ssh -p "$port" -o "StrictHostKeyChecking=${SSH_STRICT_HOST_KEY_CHECKING}" -o "ConnectTimeout=${SSH_CONNECT_TIMEOUT}") + + if [[ -n "$key" ]]; then + ssh_command+=(-i "$key" -o BatchMode=yes) fi - ssh_command+=("${ESXI_USERNAME}@${ESXI_HOST}" "$remote_command") - log "DEBUG" "Running ESXi command against ${ESXI_HOST}:${ESXI_SSH_PORT} as ${ESXI_USERNAME}" + ssh_command+=("${username}@${host}" "$remote_command") + log "DEBUG" "Running ${context} SSH query against ${host}:${port} as ${username}" - if [[ -n "$ESXI_PASSWORD" && -z "$ESXI_SSH_KEY" ]]; then - SSHPASS="$ESXI_PASSWORD" timeout "$COMMAND_TIMEOUT" sshpass -e "${ssh_command[@]}" + if [[ -n "$password" && -z "$key" ]]; then + SSHPASS="$password" timeout "$COMMAND_TIMEOUT" sshpass -e "${ssh_command[@]}" else timeout "$COMMAND_TIMEOUT" "${ssh_command[@]}" fi } +run_esxi_command() { + run_ssh_command "ESXi" "$ESXI_HOST" "$ESXI_USERNAME" "$ESXI_PASSWORD" \ + "$ESXI_SSH_KEY" "$ESXI_SSH_PORT" "$1" +} + get_esxi_drive_temperature() { local remote_device local output @@ -567,9 +619,6 @@ get_idrac_temperatures() { get_gpu_temperatures() { local output - local index - local temp - local found=1 command -v nvidia-smi >/dev/null 2>&1 || return 1 @@ -578,18 +627,245 @@ get_gpu_temperatures() { return 1 fi + parse_nvidia_smi_temperatures "gpu" "" <<< "$output" +} + +parse_nvidia_smi_temperatures() { + local source="$1" + local label_prefix="$2" + local index + local temp + local found=1 + while IFS=',' read -r index temp; do index="$(trim "$index")" temp="$(trim "$temp")" if is_integer "$temp"; then - printf 'gpu\tgpu%s\t%s\n' "$index" "$temp" + printf '%s\t%sgpu%s\t%s\n' "$source" "$label_prefix" "$index" "$temp" + found=0 + fi + done + + return "$found" +} + +parse_smartctl_temperature() { + jq -er ' + [ + .temperature.current?, + .nvme_smart_health_information_log.temperature?, + ( + .ata_smart_attributes.table[]? + | select((.name // "" | ascii_downcase) | contains("temperature")) + | .raw.value? + ) + ] + | map(if type == "string" then (tonumber? // empty) else . end) + | map(select(type == "number")) + | (.[0] // empty) + | round + ' 2>/dev/null +} + +get_linux_disk_temperatures() { + local devices + local device + local output + local status + local temp + local label + local found=1 + local -a device_list + + devices="$(normalize_csv_list "$LINUX_DISK_DEVICES")" + IFS=',' read -r -a device_list <<< "$devices" + for device in "${device_list[@]}"; do + status=0 + output="$(timeout "$COMMAND_TIMEOUT" smartctl -n "$LINUX_DISK_NOCHECK" -A -j "$device" 2>/dev/null)" || status=$? + if (( status == 124 )); then + log "WARN" "Linux disk SMART query timed out for ${device}" + continue + fi + + if ! temp="$(parse_smartctl_temperature <<< "$output")" || ! is_integer "$temp"; then + log "WARN" "Linux disk SMART output contained no temperature for ${device} (smartctl status ${status})" + continue + fi + + label="${device#/dev/}" + printf 'linux_disk\t%s\t%s\n' "$label" "$temp" + found=0 + done + + return "$found" +} + +get_remote_gpu_temperatures() { + local hosts + local host + local output + local found=1 + local -a host_list + local remote_command='nvidia-smi --query-gpu=index,temperature.gpu --format=csv,noheader,nounits' + + hosts="$(normalize_csv_list "$REMOTE_GPU_HOSTS")" + IFS=',' read -r -a host_list <<< "$hosts" + for host in "${host_list[@]}"; do + if output="$(run_ssh_command "remote GPU" "$host" "$REMOTE_GPU_USERNAME" \ + "$REMOTE_GPU_PASSWORD" "$REMOTE_GPU_SSH_KEY" "$REMOTE_GPU_SSH_PORT" "$remote_command")" \ + && parse_nvidia_smi_temperatures "remote_gpu" "${host}/" <<< "$output"; then found=0 + else + log "WARN" "Remote NVIDIA temperature query failed for ${host}" fi - done <<< "$output" + done return "$found" } +apply_temperature_offset() { + local temp="$1" + local offset="$2" + local adjusted=$((temp - offset)) + (( adjusted < 0 )) && adjusted=0 + printf '%s' "$adjusted" +} + +temperature_source_esxi_validate() { + local error=0 + + require_value "ESXI_HOST" "$ESXI_HOST" || error=1 + require_value "ESXI_USERNAME" "$ESXI_USERNAME" || error=1 + require_value "DRIVE_DEVICE" "$DRIVE_DEVICE" || error=1 + if [[ -z "$ESXI_PASSWORD" && -z "$ESXI_SSH_KEY" ]]; then + log "ERROR" "Set ESXI_PASSWORD or ESXI_SSH_KEY when TEMPERATURE_SOURCES includes esxi" + error=1 + elif [[ -z "$ESXI_SSH_KEY" ]]; then + require_value "ESXI_PASSWORD" "$ESXI_PASSWORD" || error=1 + elif is_placeholder_value "$ESXI_SSH_KEY"; then + log "ERROR" "Replace placeholder value for ESXI_SSH_KEY" + error=1 + elif [[ ! -r "$ESXI_SSH_KEY" ]]; then + log "ERROR" "ESXI_SSH_KEY is not readable: ${ESXI_SSH_KEY}" + error=1 + fi + if ! is_true "$DRY_RUN"; then + require_command "ssh" || error=1 + require_command "timeout" || error=1 + if [[ -n "$ESXI_PASSWORD" && -z "$ESXI_SSH_KEY" ]]; then + require_command "sshpass" || error=1 + fi + fi + return "$error" +} + +temperature_source_esxi_collect() { get_esxi_drive_temperature; } +temperature_source_esxi_adjust() { printf '%s' "$1"; } + +temperature_source_idrac_validate() { return 0; } +temperature_source_idrac_collect() { get_idrac_temperatures; } +temperature_source_idrac_adjust() { printf '%s' "$1"; } + +temperature_source_gpu_validate() { + if ! is_true "$DRY_RUN" && ! command -v nvidia-smi >/dev/null 2>&1; then + log "WARN" "GPU source is enabled but nvidia-smi is unavailable; fail-safe will be used if no other source succeeds" + fi +} +temperature_source_gpu_collect() { get_gpu_temperatures; } +temperature_source_gpu_adjust() { apply_temperature_offset "$1" "$GPU_TEMP_OFFSET"; } + +is_linux_device_path() { + local path="$1" + local relative + local component + local -a components + + [[ "$path" =~ ^/dev/[A-Za-z0-9._+:/-]+$ ]] || return 1 + relative="${path#/dev/}" + [[ -n "$relative" && "$relative" != */ && "$relative" != *//* ]] || return 1 + + IFS='/' read -r -a components <<< "$relative" + for component in "${components[@]}"; do + [[ "$component" != "." && "$component" != ".." ]] || return 1 + done +} + +temperature_source_linux_disk_validate() { + local devices + local device + local error=0 + local -a device_list + + require_value "LINUX_DISK_DEVICES" "$LINUX_DISK_DEVICES" || return 1 + devices="$(normalize_csv_list "$LINUX_DISK_DEVICES")" + IFS=',' read -r -a device_list <<< "$devices" + for device in "${device_list[@]}"; do + if ! is_linux_device_path "$device"; then + log "ERROR" "LINUX_DISK_DEVICES contains an invalid device path: ${device}" + error=1 + fi + done + case "$LINUX_DISK_NOCHECK" in + never|sleep|standby|idle) ;; + *) + log "ERROR" "LINUX_DISK_NOCHECK must be never, sleep, standby, or idle" + error=1 + ;; + esac + if ! is_true "$DRY_RUN"; then + require_command "smartctl" || error=1 + require_command "jq" || error=1 + require_command "timeout" || error=1 + fi + return "$error" +} +temperature_source_linux_disk_collect() { get_linux_disk_temperatures; } +temperature_source_linux_disk_adjust() { apply_temperature_offset "$1" "$LINUX_DISK_TEMP_OFFSET"; } + +temperature_source_remote_gpu_validate() { + local hosts + local host + local error=0 + local -a host_list + + require_value "REMOTE_GPU_HOSTS" "$REMOTE_GPU_HOSTS" || error=1 + require_value "REMOTE_GPU_USERNAME" "$REMOTE_GPU_USERNAME" || error=1 + hosts="$(normalize_csv_list "$REMOTE_GPU_HOSTS")" + IFS=',' read -r -a host_list <<< "$hosts" + for host in "${host_list[@]}"; do + if [[ ! "$host" =~ ^[A-Za-z0-9._:-]+$ ]]; then + log "ERROR" "REMOTE_GPU_HOSTS contains an invalid hostname or address: ${host}" + error=1 + fi + done + if [[ ! "$REMOTE_GPU_USERNAME" =~ ^[A-Za-z0-9._-]+$ ]]; then + log "ERROR" "REMOTE_GPU_USERNAME contains unsupported characters" + error=1 + fi + if [[ -z "$REMOTE_GPU_PASSWORD" && -z "$REMOTE_GPU_SSH_KEY" ]]; then + log "ERROR" "Set REMOTE_GPU_PASSWORD or REMOTE_GPU_SSH_KEY when TEMPERATURE_SOURCES includes remote_gpu" + error=1 + elif [[ -z "$REMOTE_GPU_SSH_KEY" ]]; then + require_value "REMOTE_GPU_PASSWORD" "$REMOTE_GPU_PASSWORD" || error=1 + elif is_placeholder_value "$REMOTE_GPU_SSH_KEY"; then + log "ERROR" "Replace placeholder value for REMOTE_GPU_SSH_KEY" + error=1 + elif [[ ! -r "$REMOTE_GPU_SSH_KEY" ]]; then + log "ERROR" "REMOTE_GPU_SSH_KEY is not readable: ${REMOTE_GPU_SSH_KEY}" + error=1 + fi + if ! is_true "$DRY_RUN"; then + require_command "ssh" || error=1 + require_command "timeout" || error=1 + if [[ -n "$REMOTE_GPU_PASSWORD" && -z "$REMOTE_GPU_SSH_KEY" ]]; then + require_command "sshpass" || error=1 + fi + fi + return "$error" +} +temperature_source_remote_gpu_collect() { get_remote_gpu_temperatures; } +temperature_source_remote_gpu_adjust() { apply_temperature_offset "$1" "$REMOTE_GPU_TEMP_OFFSET"; } + collect_temperature_readings() { local sources local source @@ -602,32 +878,12 @@ collect_temperature_readings() { for source in "${source_list[@]}"; do log "DEBUG" "Collecting temperature source: ${source}" - case "$source" in - esxi) - if output="$(get_esxi_drive_temperature)"; then - printf '%s\n' "$output" - found=0 - else - log "WARN" "ESXi drive temperature read failed" - fi - ;; - idrac) - if output="$(get_idrac_temperatures)"; then - printf '%s\n' "$output" - found=0 - else - log "WARN" "iDRAC temperature sensor read failed" - fi - ;; - gpu) - if output="$(get_gpu_temperatures)"; then - printf '%s\n' "$output" - found=0 - else - log "WARN" "GPU temperature read failed" - fi - ;; - esac + if output="$(temperature_source_method "$source" collect)"; then + printf '%s\n' "$output" + found=0 + else + log "WARN" "Temperature source '${source}' read failed" + fi done return "$found" @@ -646,10 +902,11 @@ calculate_decision_temperature() { [[ -z "${source:-}" || -z "${temp:-}" ]] && continue is_integer "$temp" || continue - adjusted="$temp" - if [[ "$source" == "gpu" ]]; then - adjusted=$(( temp - GPU_TEMP_OFFSET )) - (( adjusted < 0 )) && adjusted=0 + if ! adjusted="$(temperature_source_method "$source" adjust "$temp")" || ! is_integer "$adjusted"; then + log "WARN" "Temperature source '${source}' returned an invalid adjusted value for ${label}" + continue + fi + if [[ "$adjusted" != "$temp" ]]; then detail_text="${detail_text:+${detail_text}, }${source}:${label}=${temp}C(adjusted=${adjusted}C)" else detail_text="${detail_text:+${detail_text}, }${source}:${label}=${temp}C" @@ -895,8 +1152,10 @@ config_row() { print_effective_config() { local esxi_auth="password" + local remote_gpu_auth="password" [[ -n "$ESXI_SSH_KEY" ]] && esxi_auth="ssh-key" + [[ -n "$REMOTE_GPU_SSH_KEY" ]] && remote_gpu_auth="ssh-key" printf 'Effective configuration (credentials are never printed)\n' printf '%s\n' '------------------------------------------------------------' @@ -908,6 +1167,11 @@ print_effective_config() { config_row "ESXi authentication" "$esxi_auth" config_row "ESXi password" "$(credential_state "$ESXI_PASSWORD")" config_row "ESXi drive" "${DRIVE_DEVICE:-not set}" + config_row "Linux disks" "${LINUX_DISK_DEVICES:-not set} (offset=${LINUX_DISK_TEMP_OFFSET} C, nocheck=${LINUX_DISK_NOCHECK})" + config_row "Remote GPU targets" "${REMOTE_GPU_USERNAME}@${REMOTE_GPU_HOSTS:-not set}:${REMOTE_GPU_SSH_PORT}" + config_row "Remote GPU authentication" "$remote_gpu_auth" + config_row "Remote GPU password" "$(credential_state "$REMOTE_GPU_PASSWORD")" + config_row "GPU offsets" "local=${GPU_TEMP_OFFSET} C, remote=${REMOTE_GPU_TEMP_OFFSET} C" config_row "Fan thresholds" "${TEMP_LOW}/${TEMP_MEDIUM}/${TEMP_HIGH}/${TEMP_CRITICAL} C" config_row "Fan speeds" "${FAN_SPEED_IDLE}/${FAN_SPEED_LOW}/${FAN_SPEED_MEDIUM}/${FAN_SPEED_HIGH}/${FAN_SPEED_CRITICAL}%" config_row "Hysteresis" "${HYSTERESIS} C" @@ -978,11 +1242,7 @@ diagnose_mode() { IFS=',' read -r -a source_list <<< "$sources" for source in "${source_list[@]}"; do output="" - case "$source" in - esxi) output="$(get_esxi_drive_temperature)" || true ;; - idrac) output="$(get_idrac_temperatures)" || true ;; - gpu) output="$(get_gpu_temperatures)" || true ;; - esac + output="$(temperature_source_method "$source" collect)" || true if [[ -n "$output" ]]; then diagnostic_row "PASS" "source:${source}" "$(reading_summary "$output")" diff --git a/src/fan-control-tui.sh b/src/fan-control-tui.sh index 05e46d9..1e35769 100755 --- a/src/fan-control-tui.sh +++ b/src/fan-control-tui.sh @@ -15,6 +15,9 @@ CONFIG_KEYS=( IDRAC_SENSOR_INCLUDE_REGEX IDRAC_SENSOR_EXCLUDE_REGEX ESXI_HOST ESXI_USERNAME ESXI_PASSWORD ESXI_SSH_KEY ESXI_SSH_PORT SSH_CONNECT_TIMEOUT SSH_STRICT_HOST_KEY_CHECKING DRIVE_DEVICE + LINUX_DISK_DEVICES LINUX_DISK_TEMP_OFFSET LINUX_DISK_NOCHECK + REMOTE_GPU_HOSTS REMOTE_GPU_USERNAME REMOTE_GPU_PASSWORD REMOTE_GPU_SSH_KEY + REMOTE_GPU_SSH_PORT REMOTE_GPU_TEMP_OFFSET TEMP_LOW TEMP_MEDIUM TEMP_HIGH TEMP_CRITICAL FAN_SPEED_IDLE FAN_SPEED_LOW FAN_SPEED_MEDIUM FAN_SPEED_HIGH FAN_SPEED_CRITICAL HYSTERESIS FAILSAFE_ON_ERROR FAILSAFE_FAN_SPEED MANUAL_FAN_SPEED @@ -301,30 +304,38 @@ select_sources() { printf '%sTemperature source preset%s\n\n' "$COLOR_BLUE" "$COLOR_RESET" printf ' 1) iDRAC sensors only recommended first setup\n' printf ' 2) ESXi NVMe SMART only\n' - printf ' 3) iDRAC + ESXi NVMe SMART\n' - printf ' 4) iDRAC + NVIDIA GPU\n' - printf ' 5) ESXi NVMe SMART + NVIDIA GPU\n' - printf ' 6) iDRAC + ESXi + NVIDIA GPU\n\n' + printf ' 3) Local Linux disk SMART only\n' + printf ' 4) iDRAC + local Linux disk SMART\n' + printf ' 5) iDRAC + local NVIDIA GPU\n' + printf ' 6) Local Linux disk + remote NVIDIA VMs\n' + printf ' 7) Custom source list\n\n' while true; do - printf 'Choose [1-6]: ' + printf 'Choose [1-7]: ' IFS= read -r choice || return 1 case "$choice" in 1) sources="idrac" ;; 2) sources="esxi" ;; - 3) sources="idrac,esxi" ;; - 4) sources="idrac,gpu" ;; - 5) sources="esxi,gpu" ;; - 6) sources="idrac,esxi,gpu" ;; - *) warning "Choose a number from 1 to 6."; continue ;; + 3) sources="linux_disk" ;; + 4) sources="idrac,linux_disk" ;; + 5) sources="idrac,gpu" ;; + 6) sources="linux_disk,remote_gpu" ;; + 7) + prompt_value TEMPERATURE_SOURCES "Sources (esxi,idrac,gpu,linux_disk,remote_gpu)" true + sources="$(config_get TEMPERATURE_SOURCES)" + ;; + *) warning "Choose a number from 1 to 7."; continue ;; esac break done config_set TEMPERATURE_SOURCES "$sources" config_set WITH_GPU_TEMP "false" - if [[ "$sources" == *gpu* ]]; then + if [[ ",$sources," == *,gpu,* ]]; then prompt_integer GPU_TEMP_OFFSET "GPU temperature offset (C)" 0 120 fi + if [[ ",$sources," == *,remote_gpu,* ]]; then + prompt_integer REMOTE_GPU_TEMP_OFFSET "Remote GPU temperature offset (C)" 0 120 + fi notice "Temperature sources saved: ${sources}" } @@ -344,6 +355,31 @@ configure_esxi() { notice "ESXi settings saved." } +configure_linux_disk() { + header + printf '%sLocal Linux disk SMART source%s\n\n' "$COLOR_BLUE" "$COLOR_RESET" + prompt_value LINUX_DISK_DEVICES "Device paths, comma-separated" true + prompt_integer LINUX_DISK_TEMP_OFFSET "Disk temperature offset (C)" 0 120 + prompt_value LINUX_DISK_NOCHECK "Power mode check (never/sleep/standby/idle)" true + notice "Linux disk settings saved. Map each device into the container before diagnostics." +} + +configure_remote_gpu() { + header + printf '%sRemote NVIDIA GPU source%s\n\n' "$COLOR_BLUE" "$COLOR_RESET" + prompt_value REMOTE_GPU_HOSTS "GPU VM hostnames or IPs, comma-separated" true + prompt_value REMOTE_GPU_USERNAME "SSH username shared by GPU VMs" true + prompt_integer REMOTE_GPU_SSH_PORT "SSH port" 1 65535 + prompt_value REMOTE_GPU_SSH_KEY "SSH private key path (- clears it; blank keeps current)" false + if [[ -z "$(config_get REMOTE_GPU_SSH_KEY 2>/dev/null || true)" ]]; then + prompt_secret REMOTE_GPU_PASSWORD "SSH password shared by GPU VMs" true + else + warning "SSH key authentication selected; the password will be ignored." + fi + prompt_integer REMOTE_GPU_TEMP_OFFSET "Remote GPU temperature offset (C)" 0 120 + notice "Remote GPU settings saved." +} + configure_curve() { header printf '%sFan curve%s\n' "$COLOR_BLUE" "$COLOR_RESET" @@ -392,6 +428,12 @@ quick_setup() { if [[ "$(config_get TEMPERATURE_SOURCES)" == *esxi* ]]; then configure_esxi fi + if [[ "$(config_get TEMPERATURE_SOURCES)" == *linux_disk* ]]; then + configure_linux_disk + fi + if [[ "$(config_get TEMPERATURE_SOURCES)" == *remote_gpu* ]]; then + configure_remote_gpu + fi header notice "Quick setup is complete. Run Validate, then Diagnostics before starting auto mode." pause_screen @@ -409,10 +451,12 @@ masked_state() { show_config() { local idrac_password local esxi_password + local remote_gpu_password header idrac_password="$(config_get IDRAC_PASSWORD 2>/dev/null || true)" esxi_password="$(config_get ESXI_PASSWORD 2>/dev/null || true)" + remote_gpu_password="$(config_get REMOTE_GPU_PASSWORD 2>/dev/null || true)" printf '%sEffective setup (secrets redacted)%s\n\n' "$COLOR_BLUE" "$COLOR_RESET" printf ' %-28s %s\n' "Mode" "$(config_get OPERATION_MODE 2>/dev/null || true)" printf ' %-28s %s\n' "Sources" "$(config_get TEMPERATURE_SOURCES 2>/dev/null || true)" @@ -421,6 +465,9 @@ show_config() { printf ' %-28s %s@%s:%s\n' "ESXi" "$(config_get ESXI_USERNAME 2>/dev/null || true)" "$(config_get ESXI_HOST 2>/dev/null || true)" "$(config_get ESXI_SSH_PORT 2>/dev/null || true)" printf ' %-28s %s\n' "ESXi password" "$(masked_state "$esxi_password")" printf ' %-28s %s\n' "Drive device" "$(config_get DRIVE_DEVICE 2>/dev/null || true)" + printf ' %-28s %s\n' "Linux disks" "$(config_get LINUX_DISK_DEVICES 2>/dev/null || true)" + printf ' %-28s %s@%s:%s\n' "Remote GPU VMs" "$(config_get REMOTE_GPU_USERNAME 2>/dev/null || true)" "$(config_get REMOTE_GPU_HOSTS 2>/dev/null || true)" "$(config_get REMOTE_GPU_SSH_PORT 2>/dev/null || true)" + printf ' %-28s %s\n' "Remote GPU password" "$(masked_state "$remote_gpu_password")" printf ' %-28s %s/%s/%s/%s C\n' "Thresholds" "$(config_get TEMP_LOW)" "$(config_get TEMP_MEDIUM)" "$(config_get TEMP_HIGH)" "$(config_get TEMP_CRITICAL)" printf ' %-28s %s/%s/%s/%s/%s %%\n' "Fan speeds" "$(config_get FAN_SPEED_IDLE)" "$(config_get FAN_SPEED_LOW)" "$(config_get FAN_SPEED_MEDIUM)" "$(config_get FAN_SPEED_HIGH)" "$(config_get FAN_SPEED_CRITICAL)" printf ' %-28s enabled=%s, %s%%\n' "Fail-safe" "$(config_get FAILSAFE_ON_ERROR)" "$(config_get FAILSAFE_FAN_SPEED)" @@ -483,31 +530,35 @@ main_menu() { printf ' 2) iDRAC / IPMI settings\n' printf ' 3) Temperature source preset\n' printf ' 4) ESXi NVMe source settings\n' - printf ' 5) Fan curve\n' - printf ' 6) Safety, timing, and logging\n' - printf ' 7) Review redacted configuration\n' - printf ' 8) Validate configuration\n' - printf ' 9) Run read-only diagnostics\n' + printf ' 5) Local Linux disk settings\n' + printf ' 6) Remote NVIDIA GPU settings\n' + printf ' 7) Fan curve\n' + printf ' 8) Safety, timing, and logging\n' + printf ' 9) Review redacted configuration\n' + printf ' 10) Validate configuration\n' + printf ' 11) Run read-only diagnostics\n' printf ' 0) Save and exit\n\n' - printf 'Choose [0-9]: ' + printf 'Choose [0-11]: ' IFS= read -r choice || return 0 case "$choice" in 1) quick_setup ;; 2) configure_idrac; pause_screen ;; 3) select_sources; pause_screen ;; 4) configure_esxi; pause_screen ;; - 5) configure_curve; pause_screen ;; - 6) configure_safety; pause_screen ;; - 7) show_config ;; - 8) run_and_pause validate ;; - 9) run_and_pause diagnose ;; + 5) configure_linux_disk; pause_screen ;; + 6) configure_remote_gpu; pause_screen ;; + 7) configure_curve; pause_screen ;; + 8) configure_safety; pause_screen ;; + 9) show_config ;; + 10) run_and_pause validate ;; + 11) run_and_pause diagnose ;; 0) header notice "Saved ${CONFIG_FILE} with mode 600." printf 'Next: make validate && docker compose up -d\n' return 0 ;; - *) warning "Choose a number from 0 to 9."; pause_screen ;; + *) warning "Choose a number from 0 to 11."; pause_screen ;; esac done } diff --git a/tests/fan-control.test.sh b/tests/fan-control.test.sh index b581074..5c5c17d 100755 --- a/tests/fan-control.test.sh +++ b/tests/fan-control.test.sh @@ -14,6 +14,10 @@ export ESXI_HOST=10.0.0.20 export ESXI_USERNAME=root export ESXI_PASSWORD=test-password export DRIVE_DEVICE=test-drive +export LINUX_DISK_DEVICES=/dev/nvme1 +export REMOTE_GPU_HOSTS=gpu-vm-1 +export REMOTE_GPU_USERNAME=root +export REMOTE_GPU_PASSWORD=test-password # shellcheck source=../src/FanControlWithEsxiSmart.sh source "${ROOT_DIR}/src/FanControlWithEsxiSmart.sh" @@ -62,6 +66,15 @@ test_normalize_sources() { assert_eq "idrac,gpu" "$(normalize_sources)" "WITH_GPU_TEMP appends gpu source" } +test_temperature_sources_share_complete_interface() { + local source + + for source in $TEMPERATURE_SOURCE_IDS; do + temperature_source_interface_complete "$source" || fail "${source} should implement the temperature source interface" + pass + done +} + test_hex_formatting() { assert_eq "1e" "$(fan_speed_to_hex 30)" "formats 30 percent as IPMI hex" assert_eq "64" "$(fan_speed_to_hex 100)" "formats 100 percent as IPMI hex" @@ -150,6 +163,143 @@ test_decision_temperature_uses_max_adjusted_source() { assert_contains "gpu:gpu0=90C(adjusted=75C)" "$decision" "records adjusted GPU detail" } +test_linux_disk_source_collects_multiple_devices_and_accepts_smart_warning_status() { + local output + + output="$( + LINUX_DISK_DEVICES='/dev/nvme1, /dev/sdb' + timeout() { + shift + "$@" + } + smartctl() { + case "${*: -1}" in + /dev/nvme1) printf '{"temperature":73}\n'; return 8 ;; + /dev/sdb) printf '{"temperature":41}\n' ;; + esac + } + parse_smartctl_temperature() { + sed -nE 's/.*"temperature":([0-9]+).*/\1/p' + } + get_linux_disk_temperatures + )" + + assert_contains $'linux_disk\tnvme1\t73' "$output" "keeps a valid reading when smartctl reports health status bits" + assert_contains $'linux_disk\tsdb\t41' "$output" "collects every configured Linux disk" +} + +test_smartctl_json_parser_supports_nvme_and_generic_temperature() { + if ! command -v jq >/dev/null 2>&1; then + pass + return + fi + + assert_eq "73" "$(parse_smartctl_temperature <<< '{"temperature":{"current":73},"nvme_smart_health_information_log":{"temperature":73}}')" "parses generic smartctl temperature.current" + assert_eq "61" "$(parse_smartctl_temperature <<< '{"nvme_smart_health_information_log":{"temperature":61}}')" "parses the NVMe health log fallback" +} + +test_remote_gpu_source_collects_multiple_hosts() { + local output + + output="$( + REMOTE_GPU_HOSTS='gpu-vm-1,gpu-vm-2' + run_ssh_command() { + local host="$2" + if [[ "$host" == "gpu-vm-1" ]]; then + printf '0, 81\n' + else + printf '0, 76\n1, 70\n' + fi + } + get_remote_gpu_temperatures + )" + + assert_contains $'remote_gpu\tgpu-vm-1/gpu0\t81' "$output" "labels a remote GPU with its VM" + assert_contains $'remote_gpu\tgpu-vm-2/gpu1\t70' "$output" "collects every configured remote GPU VM" +} + +test_source_specific_adjustment_is_dispatched_through_interface() { + local decision + + REMOTE_GPU_TEMP_OFFSET=15 + LINUX_DISK_TEMP_OFFSET=0 + decision="$(calculate_decision_temperature $'linux_disk\tnvme1\t73\nremote_gpu\tgpu-vm/gpu0\t90')" + assert_contains $'75\t' "$decision" "uses the highest source-adjusted temperature" + assert_contains "remote_gpu:gpu-vm/gpu0=90C(adjusted=75C)" "$decision" "records remote GPU adjustment details" +} + +test_linux_and_remote_sources_validate_configuration() { + ( + OPERATION_MODE=auto + TEMPERATURE_SOURCES=linux_disk,remote_gpu + WITH_GPU_TEMP=false + DRY_RUN=true + IDRAC_IP=10.0.0.10 + IDRAC_ID=root + IDRAC_PASSWORD=test-password + LINUX_DISK_DEVICES=/dev/nvme1 + REMOTE_GPU_HOSTS=gpu-vm-1,gpu-vm-2 + REMOTE_GPU_USERNAME=monitor + REMOTE_GPU_PASSWORD=test-password + REMOTE_GPU_SSH_KEY="" + validate_config auto >/dev/null 2>&1 + ) || fail "validate_config should accept Linux disk and remote GPU sources" + pass + + ( + OPERATION_MODE=auto + TEMPERATURE_SOURCES=linux_disk + WITH_GPU_TEMP=false + DRY_RUN=true + IDRAC_IP=10.0.0.10 + IDRAC_ID=root + IDRAC_PASSWORD=test-password + LINUX_DISK_DEVICES='nvme1; touch /tmp/unsafe' + validate_config auto >/dev/null 2>&1 + ) && fail "validate_config should reject unsafe Linux device paths" + pass + + ( + OPERATION_MODE=auto + TEMPERATURE_SOURCES=linux_disk + WITH_GPU_TEMP=false + DRY_RUN=true + IDRAC_IP=10.0.0.10 + IDRAC_ID=root + IDRAC_PASSWORD=test-password + LINUX_DISK_DEVICES=/dev/../etc/passwd + validate_config auto >/dev/null 2>&1 + ) && fail "validate_config should reject parent traversal in Linux device paths" + pass + + ( + OPERATION_MODE=auto + TEMPERATURE_SOURCES=linux_disk + WITH_GPU_TEMP=false + DRY_RUN=true + IDRAC_IP=10.0.0.10 + IDRAC_ID=root + IDRAC_PASSWORD=test-password + LINUX_DISK_DEVICES=/dev/disk/by-id/nvme-KIOXIA_CD6 + validate_config auto >/dev/null 2>&1 + ) || fail "validate_config should accept nested Linux device paths" + pass + + ( + OPERATION_MODE=auto + TEMPERATURE_SOURCES=linux_disk + WITH_GPU_TEMP=false + DRY_RUN=true + IDRAC_IP=10.0.0.10 + IDRAC_ID=root + IDRAC_PASSWORD=test-password + LINUX_DISK_DEVICES=/dev/sdb + LINUX_DISK_NOCHECK=invalid + validate_config auto >/dev/null 2>&1 + ) && fail "validate_config should reject an invalid smartctl power mode" + pass +} + test_fail_safe_when_all_sources_fail() { FAILSAFE_ON_ERROR=true FAILSAFE_FAN_SPEED=70 @@ -308,9 +458,10 @@ test_config_output_redacts_credentials() { IDRAC_IP=10.0.0.10 IDRAC_PASSWORD=top-secret-password ESXI_PASSWORD=another-secret + REMOTE_GPU_PASSWORD=remote-secret output="$(print_effective_config)" assert_contains "set (19 characters)" "$output" "shows credential state without values" - if [[ "$output" == *top-secret-password* || "$output" == *another-secret* ]]; then + if [[ "$output" == *top-secret-password* || "$output" == *another-secret* || "$output" == *remote-secret* ]]; then fail "print_effective_config must never print credentials" fi pass @@ -370,6 +521,7 @@ test_diagnose_reports_read_only_source_and_preview() { } test_normalize_sources +test_temperature_sources_share_complete_interface test_hex_formatting test_remote_quote_handles_single_quotes test_esxi_password_is_not_in_process_arguments @@ -377,6 +529,11 @@ test_temperature_levels test_hysteresis_only_delays_downshift test_idrac_sensor_parsing test_decision_temperature_uses_max_adjusted_source +test_linux_disk_source_collects_multiple_devices_and_accepts_smart_warning_status +test_smartctl_json_parser_supports_nvme_and_generic_temperature +test_remote_gpu_source_collects_multiple_hosts +test_source_specific_adjustment_is_dispatched_through_interface +test_linux_and_remote_sources_validate_configuration test_fail_safe_when_all_sources_fail test_validate_accepts_idrac_only_auto_mode test_validate_rejects_placeholder_values diff --git a/tests/tui.test.sh b/tests/tui.test.sh index af816b9..0d2b47d 100644 --- a/tests/tui.test.sh +++ b/tests/tui.test.sh @@ -90,10 +90,22 @@ test_load_config_environment_is_allowlisted() { assert_eq "$before" "$PATH" "does not overwrite unrelated process environment" } +test_new_source_configuration_round_trips() { + config_set LINUX_DISK_DEVICES '/dev/nvme1,/dev/sdb' + config_set REMOTE_GPU_HOSTS 'gpu-vm-1,gpu-vm-2' + config_set REMOTE_GPU_PASSWORD 'remote secret' + + assert_eq '/dev/nvme1,/dev/sdb' "$(config_get LINUX_DISK_DEVICES)" "round-trips Linux disk devices" + assert_eq 'gpu-vm-1,gpu-vm-2' "$(config_get REMOTE_GPU_HOSTS)" "round-trips remote GPU hosts" + assert_eq 'remote secret' "$(config_get REMOTE_GPU_PASSWORD)" "round-trips the remote GPU credential" + assert_eq 'set (13 characters)' "$(masked_state "$(config_get REMOTE_GPU_PASSWORD)")" "masks the remote GPU credential" +} + test_config_round_trip_and_preserves_comments test_config_set_does_not_duplicate_keys test_config_rejects_unknown_key_and_newline test_sensitive_values_are_only_summarized test_load_config_environment_is_allowlisted +test_new_source_configuration_round_trips printf 'ok - %s assertions passed\n' "$PASS_COUNT"