diff --git a/.dockerignore b/.dockerignore index df7066d..bfd1096 100644 --- a/.dockerignore +++ b/.dockerignore @@ -6,3 +6,5 @@ __pycache__ .git .env *.md +tmp/ +.venv/ diff --git a/Dockerfile b/Dockerfile index e702b6d..ef2c990 100644 --- a/Dockerfile +++ b/Dockerfile @@ -1,4 +1,4 @@ -# ---- Stage 1: Build frontend ---- +# ---- Stage: build frontend ---- FROM node:22-alpine AS frontend-build WORKDIR /app/frontend COPY frontend/package.json frontend/package-lock.json* ./ @@ -7,8 +7,10 @@ COPY frontend/ ./ RUN npm run build # Output: /app/static/ -# ---- Stage 2: Production runtime ---- -FROM python:3.13-slim +# ---- Target: poller (hardware producer) ---- +# Privileged at runtime; the only image with SAS/SMART tooling. Slim Python +# deps (just Redis) — no web stack. +FROM python:3.13-slim AS poller RUN apt-get update && apt-get install -y --no-install-recommends \ smartmontools \ @@ -20,15 +22,48 @@ RUN apt-get update && apt-get install -y --no-install-recommends \ WORKDIR /app -COPY requirements.txt . -RUN pip install --no-cache-dir -r requirements.txt +COPY requirements-poller.txt . +RUN pip install --no-cache-dir -r requirements-poller.txt + +COPY poller.py . +COPY services/ services/ + +CMD ["python", "-u", "poller.py"] + +# ---- Target: host-poller (chassis IPMI/iDRAC + CPU + on-metal drives) ---- +# Privileged at runtime (needs /dev/ipmi0 + raw disks). Carries ipmitool + +# smartmontools; no SAS-shelf tooling. +FROM python:3.13-slim AS host-poller + +RUN apt-get update && apt-get install -y --no-install-recommends \ + ipmitool \ + smartmontools \ + util-linux \ + && rm -rf /var/lib/apt/lists/* + +WORKDIR /app + +COPY requirements-poller.txt . +RUN pip install --no-cache-dir -r requirements-poller.txt + +COPY host_poller.py . +COPY services/ services/ + +CMD ["python", "-u", "host_poller.py"] + +# ---- Target: web (read-only consumer) ---- +# Unprivileged at runtime; no hardware tools. Serves REST/UI + MQTT, reads Redis. +FROM python:3.13-slim AS web + +WORKDIR /app + +COPY requirements-web.txt . +RUN pip install --no-cache-dir -r requirements-web.txt COPY main.py . COPY routers/ routers/ COPY services/ services/ COPY models/ models/ - -# Copy built frontend from stage 1 COPY --from=frontend-build /app/static/ static/ EXPOSE 8000 @@ -36,4 +71,7 @@ EXPOSE 8000 HEALTHCHECK --interval=30s --timeout=5s --retries=3 \ CMD python -c "import urllib.request; urllib.request.urlopen('http://localhost:8000/api/health')" || exit 1 -CMD ["uvicorn", "main:app", "--host", "0.0.0.0", "--port", "8000", "--workers", "2"] +# Single worker: the MQTT publisher is a per-process singleton (multiple +# workers would connect with the same client_id and flap the broker). The +# API is a low-traffic internal monitor, so one worker is plenty. +CMD ["uvicorn", "main:app", "--host", "0.0.0.0", "--port", "8000", "--workers", "1"] diff --git a/README.md b/README.md index 7162376..b848881 100644 --- a/README.md +++ b/README.md @@ -1,107 +1,127 @@ # JBOD Monitor -REST API for monitoring drive health in JBOD enclosures on Linux. +Drive-health monitoring for JBOD enclosures on Linux, with a REST API, web UI, +and optional Home Assistant MQTT publishing. -Auto-discovers SES enclosures via sysfs, maps drives to physical slots, and exposes SMART health data. +Auto-discovers SES enclosures via sysfs, maps drives to physical slots, reads +SMART, and aggregates per-enclosure temperatures (named SES sensors + hotspot). + +## Architecture + +Two processes, with **Redis as the only interface between them**: + +``` +poller (privileged, pid:host) ──smartctl/sg_ses/zpool/ledctl──> hardware + │ serialized + paced + ▼ + Redis ◀── reads ── app (REST API + web UI + MQTT) [unprivileged] + ▲ + └── locate-LED command queue (app enqueues, poller runs) +``` + +- **`poller`** is the *only* process that touches the SAS bus. All hardware + commands run behind a single gate (`POLL_CONCURRENCY`, default 1 = fully + serialized) with a gap between drives, so it never storms fragile SAS + expanders. It sweeps on an interval and writes results to Redis. +- **`app`** (and the MQTT publisher) are pure Redis readers — unprivileged, no + `/dev`. They can restart freely without issuing a single hardware command. +- **Locate LEDs**: the API enqueues a request in Redis; the poller executes it + serialized with everything else. + +This split exists for a reason: concurrent/bursty SMART+SES traffic — especially +repeated on every container restart — can drop SAS expanders and suspend pools. +One paced owner makes that structurally impossible. ## Prerequisites - Linux with SAS/SATA JBODs connected via HBA -- `smartmontools` — for `smartctl` (SMART data) -- `sg3-utils` — for `sg_ses` (SES enclosure data) +- `smartmontools` (`smartctl`), `sg3-utils` (`sg_ses`), `ledmon` (`ledctl`) +- Redis - Python 3.11+ ```bash -# Debian/Ubuntu -apt install smartmontools sg3-utils - -# RHEL/Fedora -dnf install smartmontools sg3_utils +apt install smartmontools sg3-utils ledmon # Debian/Ubuntu ``` -## Install +## Run (docker compose) + +Two **independent** stacks so web iterations never recreate the privileged +poller (which touches the SAS bus). They're built as two images — +`jbod-poller` (privileged, has the SAS/SMART tooling) and `jbod-web` (slim, +unprivileged) — and share Redis over host networking. ```bash -pip install -r requirements.txt +# 1. Infra: poller + redis — the stable substrate. Deploy once; touch deliberately. +docker compose -f compose.infra.yml up -d + +# 2. Web: REST/UI/MQTT — iterate freely; this only ever recreates `app`. +docker compose -f compose.web.yml up -d --build ``` -## Run - -The API needs root access for `smartctl` to query drives: +Pin both images to a known-good SHA in deploy (`./build.sh` tags `:` and +`:latest` for both). To run the two processes by hand instead: ```bash -sudo uvicorn main:app --host 0.0.0.0 --port 8000 +REDIS_HOST=localhost sudo -E python poller.py # producer (needs root) +REDIS_HOST=localhost uvicorn main:app --port 8000 # consumer ``` ## API Endpoints | Endpoint | Description | |---|---| -| `GET /api/health` | Service health + tool availability | -| `GET /api/enclosures` | List all discovered SES enclosures | -| `GET /api/enclosures/{id}/drives` | List drive slots for an enclosure | +| `GET /api/health` | Service health + `redis`/`mqtt`/`poller_fresh` status | +| `GET /api/enclosures` | List discovered SES enclosures | +| `GET /api/enclosures/{id}/drives` | Drive slots for an enclosure | | `GET /api/drives/{device}` | SMART detail for a block device | | `GET /api/overview` | Aggregate enclosure + drive health | | `GET /api/temps` | Per-enclosure temperatures: named SES sensors + hotspot | -| `GET /docs` | Interactive API docs (Swagger UI) | +| `POST /api/drives/{device}/led` | Queue a locate-LED change (`state`: `locate`/`off`) | +| `GET /docs` | Swagger UI | + +Read endpoints serve the poller's last sweep; `X-Poll-Age` (seconds) headers and +the `poller_fresh` health flag surface staleness. + +## Poller tuning (env) + +| Variable | Default | Description | +|---|---|---| +| `POLL_SWEEP_INTERVAL` | `300` | Seconds between full sweeps | +| `POLL_DRIVE_GAP` | `0.5` | Seconds between each drive's SMART query | +| `POLL_CONCURRENCY` | `1` | Max concurrent hardware ops (1 = serialized) | +| `POLL_STARTUP_JITTER` | `10` | Random startup delay to avoid restart storms | +| `POLL_DATA_TTL` | `900` | Redis TTL for poller data (must exceed sweep interval) | +| `ZFS_USE_NSENTER` | — | `true` to run `zpool` in the host namespace (containers) | ## Home Assistant (MQTT) -The monitor can publish per-enclosure temperatures to Home Assistant over MQTT -using [HA discovery](https://www.home-assistant.io/integrations/mqtt/#mqtt-discovery) — -no HA YAML required. Each enclosure becomes a device with: - -- a **Hotspot** sensor — the hottest reading across all SES temperature - sensors *and* the drives housed in that enclosure (drive temps are usually - the real hotspot); `drive_max_c` and `drive_temp_count` are exposed as - attributes. -- one sensor per named SES temperature element (e.g. *Temp Inlet*, - *Temp Outlet*), labelled from the SES element descriptor page. - -Publishing is **opt-in**: set `MQTT_HOST` to enable it. +The `app` publishes per-enclosure temperatures via +[MQTT discovery](https://www.home-assistant.io/integrations/mqtt/#mqtt-discovery) +— no HA YAML. Each enclosure becomes a device with a **Hotspot** sensor (max +across SES sensors + housed drive temps; `drive_max_c`/`drive_temp_count` as +attributes) and one sensor per named SES temperature element. Opt-in via +`MQTT_HOST`; availability is tracked via an MQTT LWT. | Variable | Default | Description | |---|---|---| | `MQTT_HOST` | _(unset)_ | Broker host. Unset → MQTT disabled. | | `MQTT_PORT` | `1883` | Broker port | -| `MQTT_USERNAME` / `MQTT_PASSWORD` | _(unset)_ | Broker credentials (fallback if OpenBao unused/unavailable) | +| `MQTT_USERNAME` / `MQTT_PASSWORD` | _(unset)_ | Broker credentials | | `MQTT_DISCOVERY_PREFIX` | `homeassistant` | HA discovery topic prefix | | `MQTT_BASE_TOPIC` | `jbod-monitor` | State/availability topic root | | `MQTT_PUBLISH_INTERVAL` | `60` | Seconds between publishes | -| `MQTT_NODE_ID` | _(hostname)_ | Stable id for this monitor (disambiguates multiple hosts) | -| `MQTT_CLIENT_ID` | `jbod-monitor-` | MQTT client id | +| `MQTT_NODE_ID` | _(hostname)_ | Stable id (disambiguates multiple hosts) | -### Credentials via OpenBao - -Broker credentials are sourced from OpenBao at startup rather than stored in -plaintext (matching the homelab runtime-secret pattern). Set `MQTT_SECRET_PATH` -to a KV v2 path holding `username`/`password`; the app reads -`/v1//data/` with the token from -`OPENBAO_TOKEN_FILE` (or `OPENBAO_TOKEN`). If the read fails, it falls back to -the `MQTT_USERNAME`/`MQTT_PASSWORD` env vars. - -| Variable | Default | Description | -|---|---|---| -| `MQTT_SECRET_PATH` | _(unset)_ | KV v2 path holding the broker creds (e.g. `home_assistant`). Unset → use env creds. | -| `MQTT_SECRET_USER_KEY` | `username` | Key within the secret for the broker username | -| `MQTT_SECRET_PASS_KEY` | `password` | Key within the secret for the broker password | -| `OPENBAO_ADDR` | `https://vault.adamksmith.xyz` | OpenBao API base URL | -| `OPENBAO_MOUNT` | `secret` | KV v2 mount point | -| `OPENBAO_TOKEN_FILE` | _(unset)_ | Path to a file containing the OpenBao token (preferred; mount read-only) | -| `OPENBAO_TOKEN` | _(unset)_ | OpenBao token (used if no token file) | - -The deployed compose reads the broker user/pass from `secret/home_assistant` -(keys `mqtt_user`/`mqtt_pass`) using the read-only `claude-read` token mounted -at `/run/secrets/openbao-token`. - -Topics (defaults): +**Credentials**: simplest is a root-owned `/etc/jbod-monitor/mqtt.env` with +`MQTT_USERNAME=`/`MQTT_PASSWORD=`, loaded via compose `env_file`. Optionally the +app can read them from OpenBao at runtime (`MQTT_SECRET_PATH` + `OPENBAO_*`, +keys via `MQTT_SECRET_USER_KEY`/`MQTT_SECRET_PASS_KEY`) — see +`services/secrets.py`; it falls back to the env vars if the read fails. +Topics: ``` homeassistant/sensor/_enc/hotspot/config # retained discovery homeassistant/sensor/_enc/temp/config # retained discovery jbod-monitor//status # online | offline (LWT) -jbod-monitor//enclosure//state # {"hotspot_c":42.0, "sensor_0":28.5, ...} +jbod-monitor//enclosure//state # {"hotspot_c":42.0, ...} ``` - -Availability is tracked via an MQTT LWT, so entities show *unavailable* in HA -if the monitor stops. diff --git a/build.sh b/build.sh index 2abb9cc..0e00e2a 100755 --- a/build.sh +++ b/build.sh @@ -1,25 +1,46 @@ #!/usr/bin/env bash set -euo pipefail -IMAGE="docker.adamksmith.xyz/jbod-monitor" +# Usage: ./build.sh [poller|web|all] [commit message] +# web — build/push ONLY the web image (use this for web iterations so the +# poller image is never rebuilt/repushed) +# poller— build/push ONLY the poller image (touch deliberately) +# all — both (default) +# +# Deploy pins images by SHA (compose.*.yml ship :latest as a convenience only). + +REG="docker.adamksmith.xyz" cd "$(dirname "$0")" +# First arg is the target if it's a known one; otherwise it's treated as the +# commit message (backward-compatible with `./build.sh "message"`). +case "${1:-}" in + poller) TARGETS="poller"; MSG="${2:-Build and push images}" ;; + host-poller) TARGETS="host-poller"; MSG="${2:-Build and push images}" ;; + web) TARGETS="web"; MSG="${2:-Build and push images}" ;; + all|"") TARGETS="poller host-poller web"; MSG="${2:-Build and push images}" ;; + *) TARGETS="poller host-poller web"; MSG="${1}" ;; +esac +TARGET="${TARGETS// /+}" + # Stage and commit all changes git add -A if git diff --cached --quiet; then echo "No changes to commit, using HEAD" else - git commit -m "${1:-Build and push image}" + git commit -m "${MSG}" fi SHA=$(git rev-parse --short HEAD) -echo "Building ${IMAGE}:${SHA}" -docker build -t "${IMAGE}:${SHA}" -t "${IMAGE}:latest" . +for target in ${TARGETS}; do + img="${REG}/jbod-${target}" + echo "Building ${img}:${SHA}" + docker build --target "${target}" -t "${img}:${SHA}" -t "${img}:latest" . + echo "Pushing ${img}:${SHA}" + docker push "${img}:${SHA}" + docker push "${img}:latest" +done -echo "Pushing ${IMAGE}:${SHA}" -docker push "${IMAGE}:${SHA}" -docker push "${IMAGE}:latest" - -echo "Done: ${IMAGE}:${SHA}" +echo "Done (${TARGET}): ${REG}/jbod-{${TARGETS// /,}}:${SHA}" diff --git a/compose.infra.yml b/compose.infra.yml new file mode 100644 index 0000000..f9edefc --- /dev/null +++ b/compose.infra.yml @@ -0,0 +1,74 @@ +# Infra stack: the hardware poller + Redis. This is the STABLE substrate — +# deploy once, pin to a known-good image SHA, and touch it deliberately +# (never as part of a web iteration). Web redeploys use compose.web.yml and +# can never name these services. +# +# docker compose -f compose.infra.yml up -d # bring up / update poller+redis +# +name: jbod-infra + +services: + poller: + image: docker.adamksmith.xyz/jbod-poller:latest # pin to a SHA in deploy + container_name: jbod-poller + restart: unless-stopped + privileged: true + pid: host + network_mode: host + volumes: + - /dev:/dev + - /sys:/sys:ro + - /run/udev:/run/udev:ro + environment: + - TZ=America/Denver + - ZFS_USE_NSENTER=true + - REDIS_HOST=127.0.0.1 + - REDIS_PORT=6379 + - REDIS_DB=0 + # Pacing — keep the SAS bus quiet. POLL_CONCURRENCY=1 fully serializes. + - POLL_SWEEP_INTERVAL=300 + - POLL_DRIVE_GAP=0.5 + - POLL_CONCURRENCY=1 + - POLL_STARTUP_JITTER=10 + - POLL_DATA_TTL=900 + depends_on: + - redis + + # Chassis producer: IPMI/iDRAC + CPU coretemp + on-metal drives → Redis. + # Separate from the SAS poller (different subsystem); writes host_sensors + + # host_drives. The web consumer publishes these on the hub device. + host-poller: + image: docker.adamksmith.xyz/jbod-host-poller:latest # pin to a SHA in deploy + container_name: jbod-host-poller + restart: unless-stopped + privileged: true + network_mode: host + volumes: + - /dev:/dev + - /sys:/sys:ro + - /run/udev:/run/udev:ro + environment: + - TZ=America/Denver + - REDIS_HOST=127.0.0.1 + - REDIS_PORT=6379 + - REDIS_DB=0 + - HOST_ENV_INTERVAL=60 # IPMI/iDRAC + CPU — responsive (inlet temp 1/min) + - HOST_DRIVE_INTERVAL=300 # on-metal SMART — gentle, isolated from env + - POLL_DRIVE_GAP=0.5 + - POLL_CONCURRENCY=1 + - POLL_STARTUP_JITTER=10 + - POLL_DATA_TTL=900 + depends_on: + - redis + + redis: + image: redis:7-alpine + container_name: jbod-redis + restart: unless-stopped + network_mode: host + volumes: + - redis-data:/data + command: redis-server --save 60 1 --loglevel warning + +volumes: + redis-data: diff --git a/compose.web.yml b/compose.web.yml new file mode 100644 index 0000000..dd58d64 --- /dev/null +++ b/compose.web.yml @@ -0,0 +1,33 @@ +# Web stack: REST API + UI + Home Assistant MQTT. Read-only consumer of the +# Redis filled by compose.infra.yml. Iterate freely here — this command only +# ever touches the `app` container, never the poller: +# +# docker compose -f compose.web.yml up -d --build +# +# Requires the infra stack (Redis) to be running. Reaches Redis at +# 127.0.0.1:6379 via host networking (separate compose project, shared host net). +name: jbod-web + +services: + app: + image: docker.adamksmith.xyz/jbod-web:latest # pin to a SHA in deploy + container_name: jbod-monitor + restart: unless-stopped + network_mode: host + environment: + - TZ=America/Denver + - UVICORN_LOG_LEVEL=info + - REDIS_HOST=127.0.0.1 + - REDIS_PORT=6379 + - REDIS_DB=0 + # Home Assistant MQTT publishing (EMQX, bridged to Mosquitto HA add-on). + - MQTT_HOST=10.5.30.3 + - MQTT_PORT=1883 + - MQTT_DISCOVERY_PREFIX=homeassistant + - MQTT_BASE_TOPIC=jbod-monitor + - MQTT_PUBLISH_INTERVAL=60 + # Broker credentials (MQTT_USERNAME/MQTT_PASSWORD). Optional: container + # starts without it (MQTT just won't authenticate). + env_file: + - path: /etc/jbod-monitor/mqtt.env + required: false diff --git a/docker-compose.yml b/docker-compose.yml deleted file mode 100644 index 6ab1226..0000000 --- a/docker-compose.yml +++ /dev/null @@ -1,54 +0,0 @@ -services: - jbod-monitor: - build: . - image: docker.adamksmith.xyz/jbod-monitor:latest - container_name: jbod-monitor - restart: unless-stopped - privileged: true - pid: host - network_mode: host - volumes: - - /dev:/dev - - /sys:/sys:ro - - /run/udev:/run/udev:ro - # OpenBao RO token (claude-read) mounted read-only; never bake into image. - - /etc/jbod-monitor/openbao-token:/run/secrets/openbao-token:ro - environment: - - TZ=America/Denver - - UVICORN_LOG_LEVEL=info - - ZFS_USE_NSENTER=true - - REDIS_HOST=127.0.0.1 - - REDIS_PORT=6379 - - REDIS_DB=0 - - SMART_CACHE_TTL=120 - - SMART_POLL_INTERVAL=90 - # Home Assistant MQTT publishing (opt-in: leave MQTT_HOST unset to disable). - # EMQX at 10.5.30.3 is reachable by IP and bridged to the Mosquitto HA - # add-on, so discovery reaches Home Assistant either way. - - MQTT_HOST=10.5.30.3 - - MQTT_PORT=1883 - - MQTT_DISCOVERY_PREFIX=homeassistant - - MQTT_BASE_TOPIC=jbod-monitor - - MQTT_PUBLISH_INTERVAL=60 - # Broker credentials sourced from OpenBao (secret/home_assistant, - # keys: mqtt_user/mqtt_pass). Falls back to MQTT_USERNAME/MQTT_PASSWORD - # env if the read fails. Token is the read-only claude-read token. - - OPENBAO_ADDR=https://vault.adamksmith.xyz - - OPENBAO_TOKEN_FILE=/run/secrets/openbao-token - - MQTT_SECRET_PATH=home_assistant - - MQTT_SECRET_USER_KEY=mqtt_user - - MQTT_SECRET_PASS_KEY=mqtt_pass - depends_on: - - redis - - redis: - image: redis:7-alpine - container_name: jbod-redis - restart: unless-stopped - network_mode: host - volumes: - - redis-data:/data - command: redis-server --save 60 1 --loglevel warning - -volumes: - redis-data: diff --git a/host_poller.py b/host_poller.py new file mode 100644 index 0000000..f8c32d5 --- /dev/null +++ b/host_poller.py @@ -0,0 +1,168 @@ +"""Host-poller — server chassis temps, split into two independent loops. + +Separate process/container from the SAS poller. Two concurrent loops in one +process, sharing the single-instance lock AND the per-process hardware gate +(so IPMI and SMART never hit hardware at the same instant), but on separate +schedules: + + - env loop (fast): IPMI/iDRAC env + CPU coretemp -> jbod:host_sensors + - drive loop (slow): on-metal drive SMART -> jbod:host_drives + +Keeping SMART on a gentler cadence than the responsive inlet/CPU reads. Each +loop has its own heartbeat + restart-storm guard. Host-drive ZFS annotation +reuses jbod:zfs_map written by the SAS poller. +""" +import asyncio +import logging +import os +import random +import socket +import time + +from services import store +from services.cache import close_cache, init_cache, redis_available +from services.host import get_host_drives +from services.host_sensors import read_coretemp, read_hwmon_env, read_ipmi_temps + +logging.basicConfig( + level=logging.INFO, + format="%(asctime)s [%(levelname)s] %(name)s: %(message)s", +) +logger = logging.getLogger("host-poller") + +ENV_INTERVAL = int(os.environ.get("HOST_ENV_INTERVAL", "60")) +DRIVE_INTERVAL = int(os.environ.get("HOST_DRIVE_INTERVAL", "300")) +STARTUP_JITTER = float(os.environ.get("POLL_STARTUP_JITTER", "10")) +LOCK_TTL = 30 + +TOKEN = f"{socket.gethostname()}-hostpoll-{os.getpid()}" + + +async def sweep_env() -> None: + """IPMI/iDRAC environmental + CPU temps (the responsive loop).""" + t0 = time.monotonic() + errors: list[str] = [] + try: + ipmi = await read_ipmi_temps() + except Exception as e: + errors.append(f"ipmi:{e}") + ipmi = [] + env = [s for s in ipmi if s.get("kind") == "env"] + read_hwmon_env() + cpu = read_coretemp() or [s for s in ipmi if s.get("kind") == "cpu"] + + await store.set_host_sensors({ + "node": socket.gethostname(), + "env": env, + "cpu": cpu, + "ts": store.now(), + }) + await store.set_meta({ + "last_sweep_ts": store.now(), + "last_sweep_duration": round(time.monotonic() - t0, 1), + "env_count": len(env), + "cpu_count": len(cpu), + "errors": errors[:20], + "interval": ENV_INTERVAL, + }, key=store.K_HOST_ENV_META) + logger.info("env sweep: %d env, %d cpu, %d error(s)", len(env), len(cpu), len(errors)) + + +async def sweep_drives() -> None: + """On-metal (non-enclosure) drive SMART (the gentle loop).""" + t0 = time.monotonic() + errors: list[str] = [] + drives: list = [] + try: + pool_map = await store.get_zfs_map() + drives = await get_host_drives(pool_map) + await store.set_host_drives(drives) + except Exception as e: + errors.append(f"hostdrives:{e}") + await store.set_meta({ + "last_sweep_ts": store.now(), + "last_sweep_duration": round(time.monotonic() - t0, 1), + "host_drive_count": len(drives), + "errors": errors[:20], + "interval": DRIVE_INTERVAL, + }, key=store.K_HOST_DRIVE_META) + logger.info("drive sweep: %d drives, %.1fs, %d error(s)", + len(drives), round(time.monotonic() - t0, 1), len(errors)) + + +async def run_loop(name: str, fn, interval: int, meta_key: str) -> None: + """Run `fn` every `interval`s, with a restart-storm guard from its meta.""" + meta = await store.get_meta(key=meta_key) + if meta and meta.get("last_sweep_ts"): + since = store.now() - meta["last_sweep_ts"] + if 0 <= since < interval: + wait = interval - since + logger.info("%s: last sweep %.0fs ago — deferring first %.0fs", name, since, wait) + await asyncio.sleep(wait) + while True: + t0 = time.monotonic() + try: + await fn() + except asyncio.CancelledError: + raise + except Exception as e: + logger.error("%s loop error: %s", name, e) + await asyncio.sleep(max(5, interval - (time.monotonic() - t0))) + + +async def lock_refresher() -> None: + while True: + await asyncio.sleep(LOCK_TTL / 3) + try: + ok = await store.refresh_lock(TOKEN, LOCK_TTL, key=store.K_HOST_LOCK) + except Exception as e: + logger.error("lock refresh error (%s) — exiting", e) + os._exit(1) + if not ok: + logger.error("lost host-poller lock — exiting") + os._exit(1) + + +def _critical_task_done(task: asyncio.Task) -> None: + if task.cancelled(): + return + logger.error("critical task ended unexpectedly — exiting") + os._exit(1) + + +async def main() -> None: + await init_cache() + while not redis_available(): + logger.warning("Redis unavailable — host-poller needs Redis; retrying in 5s") + await asyncio.sleep(5) + await init_cache() + + while not await store.acquire_lock(TOKEN, LOCK_TTL, key=store.K_HOST_LOCK): + logger.warning("another host-poller holds the lock; waiting %ds", LOCK_TTL // 2) + await asyncio.sleep(LOCK_TTL // 2) + logger.info("host-poller lock acquired (%s)", TOKEN) + + refresher = asyncio.create_task(lock_refresher()) + refresher.add_done_callback(_critical_task_done) + + if os.geteuid() != 0: + logger.warning("not running as root — ipmitool/smartctl may fail") + + await asyncio.sleep(random.uniform(0, STARTUP_JITTER)) + + env_task = asyncio.create_task( + run_loop("env", sweep_env, ENV_INTERVAL, store.K_HOST_ENV_META)) + drive_task = asyncio.create_task( + run_loop("drive", sweep_drives, DRIVE_INTERVAL, store.K_HOST_DRIVE_META)) + try: + await asyncio.gather(env_task, drive_task) + finally: + for t in (refresher, env_task, drive_task): + t.cancel() + await close_cache() + + +if __name__ == "__main__": + try: + asyncio.run(main()) + except KeyboardInterrupt: + pass diff --git a/k8s/README.md b/k8s/README.md new file mode 100644 index 0000000..08ba23b --- /dev/null +++ b/k8s/README.md @@ -0,0 +1,66 @@ +# jbod-monitor on RKE2 (DaemonSet) + +Per-node self-contained deployment for Kubernetes worker nodes that can't run +the docker-compose stack. Each opted-in node runs one pod (`host-poller` + +`redis` + `web`) and publishes its own Home Assistant device +(`JBOD Monitor ()`) over MQTT. + +This mirrors the docker-compose stack with **no code changes** — each node is +independent, exactly like the standalone hosts. + +## What it collects + +- **host-poller** (privileged, hostPath `/dev`+`/sys`): IPMI/iDRAC env, CPU + coretemp, on-metal drive SMART → node-local redis. +- **web**: reads redis, publishes per-node MQTT discovery to the broker. +- No SAS poller (no SES enclosures on these nodes). Consequence: host-drive + entities have no ZFS pool label (no zfs_map producer). + +## Prereqs — verified 2026-06-16 + +1. **Registry pull**: registry requires auth. `registry-cred-vaultstaticsecret.yaml` + syncs `secret/registry/nexus` → `docker-registry-cred` (dockerconfigjson), + referenced as `imagePullSecrets` in the DaemonSet. (Mirrors the cluster's + existing `*-registry-cred` pattern.) +2. **VSO**: v1.3.0; `openbao-kubernetes` VaultAuth healthy; `transformation.templates` + supported. No existing VSS in the cluster uses transformation, so watch the + `jbod-mqtt` VSS status on first apply for template errors. +3. **MQTT egress**: confirmed — a cluster pod reached `10.5.30.3:1883`. +4. **Nodes**: pascal/currie/fermi are untainted → no tolerations needed. + +## Apply order + +```bash +# 1. Label ONLY the intended nodes (do not blanket-label all workers): +kubectl label node usco-dc-pascal usco-dc-currie usco-dc-fermi jbod-monitor=enabled +# (canary: label just usco-dc-fermi first) + +# 2. Namespace + DaemonSet: +kubectl apply -f k8s/daemonset.yaml + +# 3. Secrets (VSO → docker-registry-cred + jbod-mqtt). Apply, then confirm sync: +kubectl apply -f k8s/registry-cred-vaultstaticsecret.yaml +kubectl apply -f k8s/mqtt-vaultstaticsecret.yaml +kubectl -n jbod-monitor get vaultstaticsecret # SYNCED=True for both +kubectl -n jbod-monitor get secret jbod-mqtt -o jsonpath='{.data.MQTT_USERNAME}' | base64 -d + +# 4. Roll the DaemonSet so pods pick up the secrets if they raced ahead: +kubectl -n jbod-monitor rollout restart daemonset/jbod-monitor +``` + +## Verify + +```bash +kubectl -n jbod-monitor get pods -o wide # one per labeled node +kubectl -n jbod-monitor logs -c host-poller | grep sweep +kubectl -n jbod-monitor logs -c web | grep -i mqtt # "MQTT connected" +``` + +Then check Home Assistant for `JBOD Monitor (usco-dc-pascal|currie|fermi)`. + +## Notes + +- pascal (R620) already has an iDRAC "System Board Inlet Temp" in HA via another + integration — its env temp will partially duplicate; drives + CPU are net-new. +- Pin image tags by SHA (already done: host-poller `514502e`, web `1839e9d`). +- Removal: `kubectl delete -f k8s/daemonset.yaml` and unlabel the nodes. diff --git a/k8s/daemonset.yaml b/k8s/daemonset.yaml new file mode 100644 index 0000000..a64c46d --- /dev/null +++ b/k8s/daemonset.yaml @@ -0,0 +1,103 @@ +# jbod-monitor on RKE2 nodes — per-node self-contained DaemonSet (Model A). +# +# Each scheduled node runs one pod = host-poller + redis + web, sharing the +# pod's localhost (no hostNetwork needed). Only host-poller is privileged and +# mounts /dev + /sys to read IPMI / SMART / coretemp. web egresses to the MQTT +# broker and publishes a per-node HA device (MQTT_NODE_ID = spec.nodeName). +# +# No SAS poller (these nodes have no SES enclosures). NOTE: without it there is +# no zfs_map, so host-drive entities won't carry a ZFS pool label. +# +# Target nodes are chosen by label — see k8s/README.md (label only +# pascal/currie/fermi). Apply order: namespace -> VaultStaticSecret -> this. +--- +apiVersion: v1 +kind: Namespace +metadata: + name: jbod-monitor +--- +apiVersion: apps/v1 +kind: DaemonSet +metadata: + name: jbod-monitor + namespace: jbod-monitor + labels: + app: jbod-monitor +spec: + selector: + matchLabels: + app: jbod-monitor + template: + metadata: + labels: + app: jbod-monitor + spec: + # Only nodes explicitly opted in (see README: kubectl label node ...). + nodeSelector: + jbod-monitor: "enabled" + imagePullSecrets: + - name: docker-registry-cred # synced by registry-cred-vaultstaticsecret.yaml + containers: + - name: redis + image: redis:7-alpine + args: ["redis-server", "--save", "60", "1", "--loglevel", "warning"] + volumeMounts: + - name: redis-data + mountPath: /data + resources: + requests: {cpu: 25m, memory: 32Mi} + limits: {memory: 128Mi} + + - name: host-poller + image: docker.adamksmith.xyz/jbod-host-poller:514502e + securityContext: + privileged: true # raw /dev access for ipmitool + smartctl + env: + - {name: TZ, value: "America/Denver"} + - {name: REDIS_HOST, value: "127.0.0.1"} + - {name: REDIS_PORT, value: "6379"} + - {name: HOST_ENV_INTERVAL, value: "60"} + - {name: HOST_DRIVE_INTERVAL, value: "300"} + - {name: POLL_CONCURRENCY, value: "1"} + - {name: POLL_DRIVE_GAP, value: "0.5"} + - {name: POLL_STARTUP_JITTER, value: "10"} + - {name: POLL_DATA_TTL, value: "900"} + volumeMounts: + - {name: dev, mountPath: /dev} + - {name: sys, mountPath: /sys, readOnly: true} + - {name: udev, mountPath: /run/udev, readOnly: true} + resources: + requests: {cpu: 25m, memory: 64Mi} + limits: {memory: 256Mi} + + - name: web + image: docker.adamksmith.xyz/jbod-web:1839e9d + env: + - {name: TZ, value: "America/Denver"} + - {name: REDIS_HOST, value: "127.0.0.1"} + - {name: REDIS_PORT, value: "6379"} + - {name: MQTT_HOST, value: "10.5.30.3"} + - {name: MQTT_PORT, value: "1883"} + - {name: MQTT_DISCOVERY_PREFIX, value: "homeassistant"} + - {name: MQTT_BASE_TOPIC, value: "jbod-monitor"} + - {name: MQTT_PUBLISH_INTERVAL, value: "60"} + - name: MQTT_NODE_ID + valueFrom: + fieldRef: + fieldPath: spec.nodeName + envFrom: + - secretRef: + name: jbod-mqtt # MQTT_USERNAME / MQTT_PASSWORD (see VaultStaticSecret) + resources: + requests: {cpu: 25m, memory: 64Mi} + limits: {memory: 256Mi} + + volumes: + - name: redis-data + emptyDir: {} + - name: dev + hostPath: {path: /dev} + - name: sys + hostPath: {path: /sys} + - name: udev + hostPath: {path: /run/udev} diff --git a/k8s/mqtt-vaultstaticsecret.yaml b/k8s/mqtt-vaultstaticsecret.yaml new file mode 100644 index 0000000..1741804 --- /dev/null +++ b/k8s/mqtt-vaultstaticsecret.yaml @@ -0,0 +1,35 @@ +# MQTT broker creds for the web container, synced from OpenBao by VSO. +# +# Reads secret/home_assistant (keys mqtt_user / mqtt_pass) and renders a k8s +# Secret `jbod-mqtt` with MQTT_USERNAME / MQTT_PASSWORD keys (consumed via +# envFrom in the DaemonSet). VSO's `vso-read` policy can read secret/data/*, +# so unlike the claude-read token it CAN read home_assistant — no creds ever +# touch these manifests. +# +# Verify the transformation syntax against the installed VSO version. +apiVersion: secrets.hashicorp.com/v1beta1 +kind: VaultStaticSecret +metadata: + name: jbod-mqtt + namespace: jbod-monitor +spec: + vaultAuthRef: vault-secrets-operator-system/openbao-kubernetes + mount: secret + type: kv-v2 + path: home_assistant + refreshAfter: 1h + destination: + name: jbod-mqtt + create: true + transformation: + excludeRaw: true + # Drop ALL passthrough source keys (api_key, vantage_*, etc.) — keep only + # the templated MQTT_USERNAME/MQTT_PASSWORD below. Without this, envFrom + # would inject the entire home_assistant secret into the web container. + excludes: + - ".*" + templates: + MQTT_USERNAME: + text: '{{ .Secrets.mqtt_user }}' + MQTT_PASSWORD: + text: '{{ .Secrets.mqtt_pass }}' diff --git a/k8s/registry-cred-vaultstaticsecret.yaml b/k8s/registry-cred-vaultstaticsecret.yaml new file mode 100644 index 0000000..3afa97b --- /dev/null +++ b/k8s/registry-cred-vaultstaticsecret.yaml @@ -0,0 +1,18 @@ +# Image pull secret for docker.adamksmith.xyz, synced by VSO from OpenBao. +# Mirrors the cluster's existing *-registry-cred pattern (secret/registry/nexus +# -> dockerconfigjson). Referenced as imagePullSecrets in the DaemonSet. +apiVersion: secrets.hashicorp.com/v1beta1 +kind: VaultStaticSecret +metadata: + name: docker-registry-cred + namespace: jbod-monitor +spec: + vaultAuthRef: vault-secrets-operator-system/openbao-kubernetes + mount: secret + path: registry/nexus + type: kv-v2 + refreshAfter: 1h + destination: + name: docker-registry-cred + create: true + type: kubernetes.io/dockerconfigjson diff --git a/main.py b/main.py index bcb0756..f1ac65b 100644 --- a/main.py +++ b/main.py @@ -1,5 +1,4 @@ import asyncio -import json import logging import os from contextlib import asynccontextmanager @@ -11,11 +10,9 @@ from fastapi.staticfiles import StaticFiles from models.schemas import HealthCheck from routers import drives, enclosures, leds, overview, temps -from services.cache import cache_set, close_cache, init_cache, redis_available -from services.enclosure import discover_enclosures, list_slots +from services import store +from services.cache import close_cache, init_cache, redis_available from services.mqtt_publisher import MqttPublisher, _PAHO_AVAILABLE, mqtt_enabled -from services.smart import SMART_CACHE_TTL, _run_smartctl, sg_ses_available, smartctl_available -from services.zfs import get_zfs_pool_map logging.basicConfig( level=logging.INFO, @@ -23,81 +20,20 @@ logging.basicConfig( ) logger = logging.getLogger(__name__) -SMART_POLL_INTERVAL = int(os.environ.get("SMART_POLL_INTERVAL", "90")) - _tool_status: dict[str, bool] = {} -_poll_task: asyncio.Task | None = None _mqtt_task: asyncio.Task | None = None _mqtt_publisher: MqttPublisher | None = None -async def smart_poll_loop(): - """Pre-warm Redis with SMART data for all drives.""" - await asyncio.sleep(2) # let app finish starting - while True: - try: - # Discover all enclosure devices - enclosures_raw = discover_enclosures() - devices: set[str] = set() - for enc in enclosures_raw: - for slot in list_slots(enc["id"]): - if slot["device"]: - devices.add(slot["device"]) - - # Discover host block devices via lsblk - try: - proc = await asyncio.create_subprocess_exec( - "lsblk", "-d", "-o", "NAME,TYPE", "-J", - stdout=asyncio.subprocess.PIPE, - stderr=asyncio.subprocess.PIPE, - ) - stdout, _ = await proc.communicate() - if stdout: - for dev in json.loads(stdout).get("blockdevices", []): - if dev.get("type") == "disk": - name = dev.get("name", "") - if name and name not in devices: - devices.add(name) - except Exception as e: - logger.warning("lsblk discovery failed in poller: %s", e) - - # Poll each drive and cache result - for device in sorted(devices): - try: - result = await _run_smartctl(device) - await cache_set(f"jbod:smart:{device}", result, SMART_CACHE_TTL) - except Exception as e: - logger.warning("Poll failed for %s: %s", device, e) - - # Pre-warm ZFS map (bypasses cache by calling directly) - await get_zfs_pool_map() - - logger.info("SMART poll complete: %d devices", len(devices)) - except Exception as e: - logger.error("SMART poll loop error: %s", e) - await asyncio.sleep(SMART_POLL_INTERVAL) - - @asynccontextmanager async def lifespan(app: FastAPI): - global _poll_task, _mqtt_task, _mqtt_publisher - # Startup - _tool_status["smartctl"] = smartctl_available() - _tool_status["sg_ses"] = sg_ses_available() - - if not _tool_status["smartctl"]: - logger.warning("smartctl not found — install smartmontools for SMART data") - if not _tool_status["sg_ses"]: - logger.warning("sg_ses not found — install sg3-utils for enclosure SES data") - if os.geteuid() != 0: - logger.warning("Not running as root — smartctl may fail on some devices") + """API/web + MQTT consumer. Reads everything from Redis; the dedicated + poller process is the only thing that touches hardware.""" + global _mqtt_task, _mqtt_publisher await init_cache() _tool_status["redis"] = redis_available() - if redis_available(): - _poll_task = asyncio.create_task(smart_poll_loop()) - # Optional MQTT publisher for Home Assistant (opt-in via MQTT_HOST). _tool_status["mqtt"] = False if mqtt_enabled(): @@ -114,7 +50,6 @@ async def lifespan(app: FastAPI): yield - # Shutdown if _mqtt_task is not None: _mqtt_task.cancel() try: @@ -123,19 +58,13 @@ async def lifespan(app: FastAPI): pass if _mqtt_publisher is not None: _mqtt_publisher.stop() - if _poll_task is not None: - _poll_task.cancel() - try: - await _poll_task - except asyncio.CancelledError: - pass await close_cache() app = FastAPI( title="JBOD Monitor", description="Drive health monitoring for JBOD enclosures", - version="0.1.0", + version="0.2.0", lifespan=lifespan, ) @@ -155,8 +84,18 @@ app.include_router(temps.router) @app.get("/api/health", response_model=HealthCheck, tags=["health"]) async def health(): - _tool_status["redis"] = redis_available() - return HealthCheck(status="ok", tools=_tool_status) + tools = dict(_tool_status) + tools["redis"] = redis_available() + + # Poller freshness: is the producer keeping the store warm? + meta = await store.get_meta() + if meta and meta.get("last_sweep_ts"): + age = store.now() - meta["last_sweep_ts"] + tools["poller_fresh"] = age < meta.get("sweep_interval", 300) * 3 + else: + tools["poller_fresh"] = False + + return HealthCheck(status="ok", tools=tools) # Serve built frontend static files — mounted last so /api routes take priority. diff --git a/poller.py b/poller.py new file mode 100644 index 0000000..ce242fc --- /dev/null +++ b/poller.py @@ -0,0 +1,212 @@ +"""Dedicated hardware poller — the sole owner of SAS/SMART/SES I/O. + +Runs as its own (privileged, pid:host) container. Everything that touches +the bus happens here, serialized behind the hardware gate and paced, then +written to Redis. The API and MQTT services are pure Redis readers and +never issue a hardware command — so they can crash-loop freely without +ever storming the backplanes. + +Design points that matter for fragile SAS expanders: + - one global semaphore (POLL_CONCURRENCY=1) serializes all hardware I/O + - a gap (POLL_DRIVE_GAP) between each drive keeps the bus mostly quiet + - startup jitter avoids a thundering herd on (re)start + - a single-instance Redis lock means two pollers can't both poll + - last-good results stay in Redis; a failed sweep never blanks them +""" +import asyncio +import logging +import os +import random +import socket +import time + +from services import store +from services.cache import close_cache, init_cache, redis_available +from services.enclosure import discover_enclosures, get_enclosure_status, list_slots +from services.leds import run_led +from services.smart import _run_smartctl, sg_ses_available, smartctl_available +from services.zfs import get_zfs_pool_map + +logging.basicConfig( + level=logging.INFO, + format="%(asctime)s [%(levelname)s] %(name)s: %(message)s", +) +logger = logging.getLogger("poller") + +SWEEP_INTERVAL = int(os.environ.get("POLL_SWEEP_INTERVAL", "300")) +STARTUP_JITTER = float(os.environ.get("POLL_STARTUP_JITTER", "10")) +LOCK_TTL = 30 +# Pacing between hardware ops (POLL_DRIVE_GAP) is enforced centrally by the +# hardware gate, so every op — SMART, SES, host, MegaRAID, ledctl — is spaced. + +TOKEN = f"{socket.gethostname()}-{os.getpid()}" + + +async def sweep() -> None: + """One full, paced pass over all hardware → Redis.""" + t0 = time.monotonic() + errors: list[str] = [] + + # 1. Discover enclosures + slots (sysfs, no bus I/O). + enclosures = discover_enclosures() + inv_enclosures = [] + drive_devices: list[str] = [] + for enc in enclosures: + slots = list_slots(enc["id"]) + inv_enclosures.append({**enc, "slots": slots}) + for s in slots: + if s["device"]: + drive_devices.append(s["device"]) + + # 2. ZFS topology (gated). + pool_map = await store.get_zfs_map() + try: + pool_map = await get_zfs_pool_map() + await store.set_zfs_map(pool_map) + except Exception as e: + errors.append(f"zfs:{e}") + + # 3. Per-drive SMART (gate serializes + paces each call). + for dev in drive_devices: + try: + data = await _run_smartctl(dev) + await store.set_smart(dev, data) + except Exception as e: + errors.append(f"smart:{dev}:{e}") + + # 4. Per-enclosure SES status (gate serializes + paces each call). + for enc in enclosures: + if not enc.get("sg_device"): + continue + try: + ses = await get_enclosure_status(enc["sg_device"]) + if ses is not None: + await store.set_ses(enc["id"], ses) + except Exception as e: + errors.append(f"ses:{enc['id']}:{e}") + + # NOTE: host/on-metal drives are owned by the separate host-poller now. + + # 6. Inventory + heartbeat. + await store.set_inventory({"enclosures": inv_enclosures, "ts": store.now()}) + duration = round(time.monotonic() - t0, 1) + await store.set_meta({ + "last_sweep_ts": store.now(), + "last_sweep_duration": duration, + "drive_count": len(drive_devices), + "enclosure_count": len(enclosures), + "errors": errors[:20], + "sweep_interval": SWEEP_INTERVAL, + }) + logger.info( + "sweep done: %d drives, %d enclosures, %.1fs, %d error(s)", + len(drive_devices), len(enclosures), duration, len(errors), + ) + + +async def led_consumer() -> None: + """Drain locate-LED requests from Redis and run them (gated).""" + while True: + try: + cmd = await store.pop_led(timeout=5) + if cmd: + try: + await run_led(cmd["device"], cmd["state"]) + logger.info("LED %s -> %s", cmd.get("device"), cmd.get("state")) + except Exception as e: + logger.warning("LED command failed %s: %s", cmd, e) + except asyncio.CancelledError: + raise + except Exception as e: + logger.warning("LED consumer error: %s", e) + await asyncio.sleep(2) + + +async def lock_refresher() -> None: + while True: + await asyncio.sleep(LOCK_TTL / 3) + # Any failure to confirm we still hold the lock is fatal — keeping the + # lock alive is what prevents a second poller from sweeping concurrently. + try: + ok = await store.refresh_lock(TOKEN, LOCK_TTL) + except Exception as e: + logger.error("lock refresh error (%s) — exiting to avoid double-polling", e) + os._exit(1) + if not ok: + logger.error("lost poller lock — exiting to avoid double-polling") + os._exit(1) + + +def _critical_task_done(task: asyncio.Task) -> None: + """Fail-safe: if a critical task ends for any reason other than a clean + cancel (shutdown), exit so we never run unguarded.""" + if task.cancelled(): + return + logger.error("critical task ended unexpectedly — exiting") + os._exit(1) + + +async def main() -> None: + await init_cache() + while not redis_available(): + logger.warning("Redis unavailable — poller needs Redis; retrying in 5s") + await asyncio.sleep(5) + await init_cache() + + # Single-instance guard. + while not await store.acquire_lock(TOKEN, LOCK_TTL): + logger.warning("another poller holds the lock; waiting %ds", LOCK_TTL // 2) + await asyncio.sleep(LOCK_TTL // 2) + logger.info("poller lock acquired (%s)", TOKEN) + + # Start the refresher IMMEDIATELY — before any pre-sweep sleep. The jitter + # and restart-storm deferral below can together exceed LOCK_TTL, so without + # an active refresher the lock would expire and a second poller could begin + # sweeping concurrently (a hardware-storm risk). + refresher = asyncio.create_task(lock_refresher()) + refresher.add_done_callback(_critical_task_done) + + if not smartctl_available(): + logger.warning("smartctl not found — install smartmontools") + if not sg_ses_available(): + logger.warning("sg_ses not found — install sg3-utils") + if os.geteuid() != 0: + logger.warning("not running as root — smartctl/sg_ses may fail") + + # Stagger startup so a restart doesn't immediately storm the bus. + jitter = random.uniform(0, STARTUP_JITTER) + logger.info("startup jitter %.1fs", jitter) + await asyncio.sleep(jitter) + + # Restart-storm guard: if a sweep ran recently (persisted in Redis across + # restarts), wait out the remainder of the interval instead of immediately + # re-sweeping. This is the key protection against deploy-cycling storms. + meta = await store.get_meta() + if meta and meta.get("last_sweep_ts"): + since = store.now() - meta["last_sweep_ts"] + if 0 <= since < SWEEP_INTERVAL: + wait = SWEEP_INTERVAL - since + logger.info("last sweep %.0fs ago — deferring first sweep %.0fs", since, wait) + await asyncio.sleep(wait) + + leds = asyncio.create_task(led_consumer()) + try: + while True: + t0 = time.monotonic() + try: + await sweep() + except Exception as e: + logger.error("sweep error: %s", e) + elapsed = time.monotonic() - t0 + await asyncio.sleep(max(5, SWEEP_INTERVAL - elapsed)) + finally: + refresher.cancel() + leds.cancel() + await close_cache() + + +if __name__ == "__main__": + try: + asyncio.run(main()) + except KeyboardInterrupt: + pass diff --git a/requirements-poller.txt b/requirements-poller.txt new file mode 100644 index 0000000..a0390dd --- /dev/null +++ b/requirements-poller.txt @@ -0,0 +1,2 @@ +# Poller image — hardware producer. Only needs Redis; no web stack. +redis>=5.0.0 diff --git a/requirements-web.txt b/requirements-web.txt new file mode 100644 index 0000000..20eeae1 --- /dev/null +++ b/requirements-web.txt @@ -0,0 +1,6 @@ +# Web/API image — read-only consumer + MQTT. No hardware tools needed. +fastapi>=0.115.0 +uvicorn>=0.34.0 +pydantic>=2.10.0 +redis>=5.0.0 +paho-mqtt>=2.1.0 diff --git a/routers/drives.py b/routers/drives.py index ff38fc7..47ff0b2 100644 --- a/routers/drives.py +++ b/routers/drives.py @@ -1,30 +1,37 @@ +import re + from fastapi import APIRouter, HTTPException, Response from models.schemas import DriveDetail -from services.smart import get_smart_data -from services.zfs import get_zfs_pool_map +from services import store router = APIRouter(prefix="/api/drives", tags=["drives"]) @router.get("/{device}", response_model=DriveDetail) async def get_drive_detail(device: str, response: Response): - """Get SMART detail for a specific block device.""" - try: - data, cache_hit = await get_smart_data(device) - except ValueError as e: - raise HTTPException(status_code=400, detail=str(e)) + """Get SMART detail for a specific block device (from the store).""" + if not re.match(r"^[a-zA-Z0-9\-]+$", device): + raise HTTPException(status_code=400, detail=f"Invalid device name: {device}") + data = await store.get_smart(device) + if data is None: + raise HTTPException( + status_code=404, + detail=f"No SMART data for '{device}' yet (poller may not have swept it)", + ) if "error" in data: raise HTTPException(status_code=502, detail=data["error"]) - pool_map = await get_zfs_pool_map() + pool_map = await store.get_zfs_map() zfs_info = pool_map.get(device) if zfs_info: data["zfs_pool"] = zfs_info["pool"] data["zfs_vdev"] = zfs_info["vdev"] data["zfs_state"] = zfs_info.get("state") - response.headers["X-Cache"] = "HIT" if cache_hit else "MISS" + meta = await store.get_meta() + if meta and meta.get("last_sweep_ts"): + response.headers["X-Poll-Age"] = str(int(store.now() - meta["last_sweep_ts"])) return DriveDetail(**data) diff --git a/routers/enclosures.py b/routers/enclosures.py index 59858f5..2030163 100644 --- a/routers/enclosures.py +++ b/routers/enclosures.py @@ -1,24 +1,23 @@ from fastapi import APIRouter, HTTPException from models.schemas import Enclosure, SlotInfo -from services.enclosure import discover_enclosures, list_slots +from services import store router = APIRouter(prefix="/api/enclosures", tags=["enclosures"]) @router.get("", response_model=list[Enclosure]) async def get_enclosures(): - """Discover all SES enclosures.""" - return discover_enclosures() + """List all discovered SES enclosures (from the poller inventory).""" + inventory = await store.get_inventory() + return [Enclosure(**e) for e in inventory.get("enclosures", [])] @router.get("/{enclosure_id}/drives", response_model=list[SlotInfo]) async def get_enclosure_drives(enclosure_id: str): - """List all drive slots for an enclosure.""" - slots = list_slots(enclosure_id) - if not slots: - # Check if the enclosure exists at all - enclosures = discover_enclosures() - if not any(e["id"] == enclosure_id for e in enclosures): - raise HTTPException(status_code=404, detail=f"Enclosure '{enclosure_id}' not found") - return slots + """List all drive slots for an enclosure (from the poller inventory).""" + inventory = await store.get_inventory() + for enc in inventory.get("enclosures", []): + if enc["id"] == enclosure_id: + return [SlotInfo(**s) for s in enc.get("slots", [])] + raise HTTPException(status_code=404, detail=f"Enclosure '{enclosure_id}' not found") diff --git a/routers/leds.py b/routers/leds.py index 4ee8ec4..e7f0e29 100644 --- a/routers/leds.py +++ b/routers/leds.py @@ -1,7 +1,9 @@ +import re + from fastapi import APIRouter, HTTPException from pydantic import BaseModel, field_validator -from services.leds import set_led +from services import store router = APIRouter(prefix="/api/drives", tags=["leds"]) @@ -19,11 +21,12 @@ class LedRequest(BaseModel): @router.post("/{device}/led") async def set_drive_led(device: str, body: LedRequest): - """Set the locate LED for a drive.""" - try: - result = await set_led(device, body.state) - except ValueError as e: - raise HTTPException(status_code=400, detail=str(e)) - except RuntimeError as e: - raise HTTPException(status_code=502, detail=str(e)) - return result + """Queue a locate-LED change. The poller (sole hardware owner) executes it.""" + if not re.fullmatch(r"[a-zA-Z0-9]+", device): + raise HTTPException(status_code=400, detail=f"Invalid device name: {device}") + + queued = await store.enqueue_led(device, body.state) + if not queued: + raise HTTPException(status_code=503, detail="LED queue unavailable (Redis down)") + + return {"device": device, "state": body.state, "queued": True} diff --git a/routers/overview.py b/routers/overview.py index ca27c8e..3a00c8e 100644 --- a/routers/overview.py +++ b/routers/overview.py @@ -1,4 +1,3 @@ -import asyncio import logging from fastapi import APIRouter, Response @@ -11,10 +10,8 @@ from models.schemas import ( Overview, SlotWithDrive, ) -from services.enclosure import discover_enclosures, get_enclosure_status, list_slots -from services.host import get_host_drives -from services.smart import get_smart_data -from services.zfs import get_zfs_pool_map +from services import store +from services.health import compute_health_status logger = logging.getLogger(__name__) @@ -23,142 +20,92 @@ router = APIRouter(prefix="/api/overview", tags=["overview"]) @router.get("", response_model=Overview) async def get_overview(response: Response): - """Aggregate view of all enclosures, slots, and drive health.""" - enclosures_raw = discover_enclosures() - pool_map = await get_zfs_pool_map() + """Aggregate view of all enclosures, slots, and drive health. - # Fetch SES health data for all enclosures concurrently - async def _get_health(enc): - if enc.get("sg_device"): - return await get_enclosure_status(enc["sg_device"]) - return None - - health_results = await asyncio.gather( - *[_get_health(enc) for enc in enclosures_raw], - return_exceptions=True, - ) + Pure read from the store — the poller is the only thing that touched + hardware to produce this data. + """ + inventory = await store.get_inventory() + pool_map = await store.get_zfs_map() enc_results: list[EnclosureWithDrives] = [] total_drives = 0 warnings = 0 errors = 0 all_healthy = True - all_cache_hits = True - any_lookups = False - - for enc_idx, enc in enumerate(enclosures_raw): - slots_raw = list_slots(enc["id"]) - - # Gather SMART data for all populated slots concurrently - populated = [(s, s["device"]) for s in slots_raw if s["populated"] and s["device"]] - smart_tasks = [get_smart_data(dev) for _, dev in populated] - smart_results = await asyncio.gather(*smart_tasks, return_exceptions=True) - - smart_map: dict[str, dict] = {} - for (slot_info, dev), result in zip(populated, smart_results): - if isinstance(result, Exception): - logger.warning("SMART query failed for %s: %s", dev, result) - smart_map[dev] = {"device": dev, "smart_supported": False} - all_cache_hits = False - else: - data, hit = result - smart_map[dev] = data - any_lookups = True - if not hit: - all_cache_hits = False + for enc in inventory.get("enclosures", []): slots_out: list[SlotWithDrive] = [] - for s in slots_raw: + for s in enc.get("slots", []): + dev = s.get("device") drive_summary = None - if s["device"] and s["device"] in smart_map: - sd = smart_map[s["device"]] + + if dev: total_drives += 1 + sd = await store.get_smart(dev) + if sd: + status = compute_health_status(sd) + if status == "error": + errors += 1 + all_healthy = False + elif status == "warning": + warnings += 1 - healthy = sd.get("smart_healthy") - if healthy is False: - errors += 1 - all_healthy = False - elif healthy is None and sd.get("smart_supported", True): - warnings += 1 - - # Check for concerning SMART values - if sd.get("reallocated_sectors") and sd["reallocated_sectors"] > 0: - warnings += 1 - if sd.get("pending_sectors") and sd["pending_sectors"] > 0: - warnings += 1 - if sd.get("uncorrectable_errors") and sd["uncorrectable_errors"] > 0: - warnings += 1 - - # Compute health_status for frontend - realloc = sd.get("reallocated_sectors") or 0 - pending = sd.get("pending_sectors") or 0 - unc = sd.get("uncorrectable_errors") or 0 - if healthy is False: - health_status = "error" - elif realloc > 0 or pending > 0 or unc > 0 or (healthy is None and sd.get("smart_supported", True)): - health_status = "warning" - else: - health_status = "healthy" - - drive_summary = DriveHealthSummary( - device=sd["device"], - model=sd.get("model"), - serial=sd.get("serial"), - wwn=sd.get("wwn"), - firmware=sd.get("firmware"), - capacity_bytes=sd.get("capacity_bytes"), - smart_healthy=healthy, - smart_supported=sd.get("smart_supported", True), - temperature_c=sd.get("temperature_c"), - power_on_hours=sd.get("power_on_hours"), - reallocated_sectors=sd.get("reallocated_sectors"), - pending_sectors=sd.get("pending_sectors"), - uncorrectable_errors=sd.get("uncorrectable_errors"), - zfs_pool=pool_map.get(sd["device"], {}).get("pool"), - zfs_vdev=pool_map.get(sd["device"], {}).get("vdev"), - zfs_state=pool_map.get(sd["device"], {}).get("state"), - health_status=health_status, - ) - elif s["populated"]: + zinfo = pool_map.get(dev, {}) + drive_summary = DriveHealthSummary( + device=sd.get("device", dev), + model=sd.get("model"), + serial=sd.get("serial"), + wwn=sd.get("wwn"), + firmware=sd.get("firmware"), + capacity_bytes=sd.get("capacity_bytes"), + smart_healthy=sd.get("smart_healthy"), + smart_supported=sd.get("smart_supported", True), + temperature_c=sd.get("temperature_c"), + power_on_hours=sd.get("power_on_hours"), + reallocated_sectors=sd.get("reallocated_sectors"), + pending_sectors=sd.get("pending_sectors"), + uncorrectable_errors=sd.get("uncorrectable_errors"), + zfs_pool=zinfo.get("pool"), + zfs_vdev=zinfo.get("vdev"), + zfs_state=zinfo.get("state"), + health_status=status, + ) + elif s.get("populated"): total_drives += 1 slots_out.append(SlotWithDrive( slot=s["slot"], populated=s["populated"], - device=s["device"], + device=dev, drive=drive_summary, )) - # Attach enclosure health from SES - health_data = health_results[enc_idx] + ses = await store.get_ses(enc["id"]) enc_health = None - if isinstance(health_data, dict): - enc_health = EnclosureHealth(**health_data) - # Count enclosure-level issues + if ses: + enc_health = EnclosureHealth(**ses) if enc_health.overall_status == "CRITICAL": errors += 1 all_healthy = False elif enc_health.overall_status == "WARNING": warnings += 1 - elif isinstance(health_data, Exception): - logger.warning("SES health failed for %s: %s", enc["id"], health_data) enc_results.append(EnclosureWithDrives( id=enc["id"], sg_device=enc.get("sg_device"), - vendor=enc["vendor"], - model=enc["model"], - revision=enc["revision"], - total_slots=enc["total_slots"], - populated_slots=enc["populated_slots"], + vendor=enc.get("vendor", ""), + model=enc.get("model", ""), + revision=enc.get("revision", ""), + total_slots=enc.get("total_slots", 0), + populated_slots=enc.get("populated_slots", 0), slots=slots_out, health=enc_health, )) - # Host drives (non-enclosure) - host_drives_raw = await get_host_drives() + # Host (non-enclosure) drives — fully precomputed by the poller. host_drives_out: list[HostDrive] = [] - for hd in host_drives_raw: + for hd in await store.get_host_drives(): total_drives += 1 hs = hd.get("health_status", "healthy") if hs == "error": @@ -166,7 +113,6 @@ async def get_overview(response: Response): all_healthy = False elif hs == "warning": warnings += 1 - # Count physical drives behind RAID controllers for pd in hd.get("physical_drives", []): total_drives += 1 pd_hs = pd.get("health_status", "healthy") @@ -177,7 +123,10 @@ async def get_overview(response: Response): warnings += 1 host_drives_out.append(HostDrive(**hd)) - response.headers["X-Cache"] = "HIT" if (any_lookups and all_cache_hits) else "MISS" + # Surface poller freshness so stale data is visible. + meta = await store.get_meta() + if meta and meta.get("last_sweep_ts"): + response.headers["X-Poll-Age"] = str(int(store.now() - meta["last_sweep_ts"])) return Overview( healthy=all_healthy and errors == 0, diff --git a/services/cache.py b/services/cache.py index 2b922b6..a7c9970 100644 --- a/services/cache.py +++ b/services/cache.py @@ -38,6 +38,11 @@ def redis_available() -> bool: return _redis is not None +def get_client() -> "redis.Redis | None": + """Return the raw Redis client for primitives not covered by get/set.""" + return _redis + + async def cache_get(key: str) -> Any | None: """GET key from Redis, return deserialized value or None on miss/error.""" if _redis is None: @@ -53,10 +58,13 @@ async def cache_get(key: str) -> Any | None: async def cache_set(key: str, value: Any, ttl: int = 120) -> None: - """SET key in Redis with expiry, silently catches errors.""" + """SET key in Redis. ttl<=0 stores with no expiry (Redis rejects EX 0).""" if _redis is None: return try: - await _redis.set(key, json.dumps(value), ex=ttl) + if ttl and ttl > 0: + await _redis.set(key, json.dumps(value), ex=ttl) + else: + await _redis.set(key, json.dumps(value)) except Exception as e: logger.warning("Redis SET %s failed: %s", key, e) diff --git a/services/enclosure.py b/services/enclosure.py index 418b5f7..ca5beb5 100644 --- a/services/enclosure.py +++ b/services/enclosure.py @@ -4,6 +4,8 @@ import os import re from pathlib import Path +from services.hwgate import gate + logger = logging.getLogger(__name__) ENCLOSURE_BASE = Path("/sys/class/enclosure") @@ -153,14 +155,16 @@ def _parse_slot_number(entry: Path) -> int | None: async def _run_sg_ses(sg_device: str, page: str) -> str | None: - """Run sg_ses for a given page, returning decoded stdout or None.""" + """Run sg_ses for a given page, returning decoded stdout or None. + Hardware-gated.""" try: - proc = await asyncio.create_subprocess_exec( - "sg_ses", f"--page={page}", sg_device, - stdout=asyncio.subprocess.PIPE, - stderr=asyncio.subprocess.PIPE, - ) - stdout, stderr = await proc.communicate() + async with gate(): + proc = await asyncio.create_subprocess_exec( + "sg_ses", f"--page={page}", sg_device, + stdout=asyncio.subprocess.PIPE, + stderr=asyncio.subprocess.PIPE, + ) + stdout, stderr = await proc.communicate() if proc.returncode != 0: logger.warning( "sg_ses page %s failed for %s: %s", @@ -177,22 +181,18 @@ async def _run_sg_ses(sg_device: str, page: str) -> str | None: async def get_enclosure_status(sg_device: str) -> dict | None: - """Run sg_ses status (0x02) + element descriptor (0x07) pages and parse. + """Run sg_ses status page (0x02) and parse enclosure health data. - The element descriptor page provides human-readable names for each - element (e.g. "Temp Inlet", "Fan 1"); these are merged into the status - elements so callers get labelled sensors. Descriptor lookup is - best-effort — if page 0x07 is unsupported, elements fall back to bare - indices. + Deliberately only the status page: the element descriptor page (0x07) + is omitted because it errored on the IOM6/Xyratex expanders here + (``couldn't read config page, res=35``) and returned empty names + anyway. Elements fall back to generic labels. If descriptor names are + ever wanted, fetch 0x07 ONCE at startup and cache — never per poll. """ - status_text, desc_text = await asyncio.gather( - _run_sg_ses(sg_device, "0x02"), - _run_sg_ses(sg_device, "0x07"), - ) + status_text = await _run_sg_ses(sg_device, "0x02") if status_text is None: return None - names = _parse_element_descriptors(desc_text) if desc_text else {} - return _parse_ses_page02(status_text, names) + return _parse_ses_page02(status_text) def _norm_type(element_type: str) -> str: diff --git a/services/health.py b/services/health.py new file mode 100644 index 0000000..0a28e3b --- /dev/null +++ b/services/health.py @@ -0,0 +1,21 @@ +"""Shared drive-health classification. + +Single source of truth for turning a parsed SMART dict into a +healthy/warning/error status, used by both the poller (host drives) and +the API (enclosure drives) so the rule never drifts between them. +""" + + +def compute_health_status(smart: dict) -> str: + """Return "healthy", "warning", or "error" for a SMART dict.""" + healthy = smart.get("smart_healthy") + realloc = smart.get("reallocated_sectors") or 0 + pending = smart.get("pending_sectors") or 0 + unc = smart.get("uncorrectable_errors") or 0 + supported = smart.get("smart_supported", True) + + if healthy is False: + return "error" + if realloc > 0 or pending > 0 or unc > 0 or (healthy is None and supported): + return "warning" + return "healthy" diff --git a/services/host.py b/services/host.py index 6938aaa..585fd68 100644 --- a/services/host.py +++ b/services/host.py @@ -3,22 +3,28 @@ import json import logging from services.enclosure import discover_enclosures, list_slots -from services.smart import get_smart_data, scan_megaraid_drives -from services.zfs import get_zfs_pool_map +from services.health import compute_health_status +from services.hwgate import gate +from services.smart import _run_smartctl, scan_megaraid_drives logger = logging.getLogger(__name__) -async def get_host_drives() -> list[dict]: - """Discover non-enclosure block devices and return SMART data for each.""" +async def get_host_drives(pool_map: dict) -> list[dict]: + """Discover non-enclosure block devices and return SMART data for each. + + Poller-side producer: serialized through the hardware gate. ``pool_map`` + is passed in (computed once per sweep) to avoid a second zpool call. + """ # Get all block devices via lsblk try: - proc = await asyncio.create_subprocess_exec( - "lsblk", "-d", "-o", "NAME,SIZE,TYPE,MODEL,ROTA,TRAN", "-J", - stdout=asyncio.subprocess.PIPE, - stderr=asyncio.subprocess.PIPE, - ) - stdout, _ = await proc.communicate() + async with gate(): + proc = await asyncio.create_subprocess_exec( + "lsblk", "-d", "-o", "NAME,SIZE,TYPE,MODEL,ROTA,TRAN", "-J", + stdout=asyncio.subprocess.PIPE, + stderr=asyncio.subprocess.PIPE, + ) + stdout, _ = await proc.communicate() lsblk_data = json.loads(stdout) except (FileNotFoundError, json.JSONDecodeError) as e: logger.warning("lsblk failed: %s", e) @@ -59,10 +65,11 @@ async def get_host_drives() -> list[dict]: host_devices.append({"name": name, "drive_type": drive_type}) - # Fetch SMART + ZFS data concurrently - pool_map = await get_zfs_pool_map() - smart_tasks = [get_smart_data(d["name"]) for d in host_devices] - smart_results = await asyncio.gather(*smart_tasks, return_exceptions=True) + # Fetch SMART for each host disk (serialized by the gate inside _run_smartctl) + smart_results = await asyncio.gather( + *[_run_smartctl(d["name"]) for d in host_devices], + return_exceptions=True, + ) results: list[dict] = [] for dev_info, smart_result in zip(host_devices, smart_results): @@ -72,21 +79,9 @@ async def get_host_drives() -> list[dict]: logger.warning("SMART query failed for host drive %s: %s", name, smart_result) smart = {"device": name, "smart_supported": False} else: - smart, _ = smart_result - - # Compute health_status (same logic as overview.py) - healthy = smart.get("smart_healthy") - realloc = smart.get("reallocated_sectors") or 0 - pending = smart.get("pending_sectors") or 0 - unc = smart.get("uncorrectable_errors") or 0 - - if healthy is False: - health_status = "error" - elif realloc > 0 or pending > 0 or unc > 0 or (healthy is None and smart.get("smart_supported", True)): - health_status = "warning" - else: - health_status = "healthy" + smart = smart_result + health_status = compute_health_status(smart) zfs_info = pool_map.get(name, {}) results.append({ @@ -97,7 +92,7 @@ async def get_host_drives() -> list[dict]: "wwn": smart.get("wwn"), "firmware": smart.get("firmware"), "capacity_bytes": smart.get("capacity_bytes"), - "smart_healthy": healthy, + "smart_healthy": smart.get("smart_healthy"), "smart_supported": smart.get("smart_supported", True), "temperature_c": smart.get("temperature_c"), "power_on_hours": smart.get("power_on_hours"), @@ -116,16 +111,7 @@ async def get_host_drives() -> list[dict]: if has_raid: megaraid_drives = await scan_megaraid_drives() for pd in megaraid_drives: - pd_healthy = pd.get("smart_healthy") - pd_realloc = pd.get("reallocated_sectors") or 0 - pd_pending = pd.get("pending_sectors") or 0 - pd_unc = pd.get("uncorrectable_errors") or 0 - if pd_healthy is False: - pd["health_status"] = "error" - elif pd_realloc > 0 or pd_pending > 0 or pd_unc > 0: - pd["health_status"] = "warning" - else: - pd["health_status"] = "healthy" + pd["health_status"] = compute_health_status(pd) pd["drive_type"] = "physical" pd["physical_drives"] = [] diff --git a/services/host_sensors.py b/services/host_sensors.py new file mode 100644 index 0000000..ce2650b --- /dev/null +++ b/services/host_sensors.py @@ -0,0 +1,147 @@ +"""Host/chassis temperature collectors: IPMI (iDRAC) + CPU coretemp. + +Used by the host-poller (separate process from the SAS poller). In-band IPMI +via ipmitool (/dev/ipmi0) gives environmental sensors (Inlet/Exhaust/…); the +kernel coretemp hwmon gives CPU package temps. IPMI is gated for serialization; +coretemp is a cheap sysfs read. +""" +import asyncio +import logging +import os +import re +from collections import Counter +from pathlib import Path + +from services.hwgate import gate + +logger = logging.getLogger(__name__) + +HWMON = Path("/sys/class/hwmon") + + +async def read_ipmi_temps() -> list[dict]: + """`ipmitool sdr type temperature` → [{name, temperature_c, status, kind}].""" + try: + async with gate(): + proc = await asyncio.create_subprocess_exec( + "ipmitool", "sdr", "type", "temperature", + stdout=asyncio.subprocess.PIPE, + stderr=asyncio.subprocess.PIPE, + ) + out, err = await proc.communicate() + except FileNotFoundError: + logger.warning("ipmitool not found") + return [] + if proc.returncode != 0: + logger.warning("ipmitool failed: %s", err.decode(errors="replace").strip()) + return [] + return _parse_ipmi(out.decode(errors="replace")) + + +def _parse_ipmi(text: str) -> list[dict]: + # e.g. "Inlet Temp | 04h | ok | 7.1 | 23 degrees C" + # "Temp | 0Eh | ok | 3.1 | 40 degrees C" (CPU1) + sensors: list[dict] = [] + for line in text.splitlines(): + parts = [p.strip() for p in line.split("|")] + if len(parts) < 5: + continue + name, _sid, status, entity, reading = parts[:5] + # Skip relative/derived sensors (e.g. "P1 Therm Margin") — they read in + # "degrees C" but are a distance-to-throttle, not an absolute temp. + if "margin" in name.lower(): + continue + m = re.search(r"(-?\d+(?:\.\d+)?)\s*degrees?\s*C", reading, re.IGNORECASE) + if not m: # "no reading", "disabled", "ns", etc. + continue + temp = float(m.group(1)) + kind = "env" + # Dell reports CPU temps as "Temp" under entity 3.x — relabel them. + if name.lower() == "temp" and entity.startswith("3."): + name = f"CPU {entity.split('.')[-1]}" + kind = "cpu" + sensors.append({ + "name": name, + "temperature_c": temp, + "status": status or "ok", + "kind": kind, + }) + return sensors + + +def read_hwmon_env() -> list[dict]: + """Read labelled temps from extra hwmon chips as env sensors. + + For boxes without IPMI (e.g. Dell workstations), the BIOS SMM interface + (`dell_smm`) exposes CPU/ambient/SODIMM/board temps. Chips are configurable + via HOST_HWMON_CHIPS (comma list; default "dell_smm"); set to "" to disable. + Names are "