commit dbc7464b588e6b612ba06e74d7d98267ab524531 Author: Johan Lundberg Date: Thu Jul 30 23:22:10 2026 +0200 Prepare project for public release diff --git a/.dockerignore b/.dockerignore new file mode 100644 index 0000000..1114078 --- /dev/null +++ b/.dockerignore @@ -0,0 +1,26 @@ +# Never let credentials into an image layer. config.ini holds the SNMP auth +# and priv passphrases; supply them at runtime via env instead. +config.ini +.env + +# Local runtime artefacts. +.git/ +.github/ +.agents/ +.codex/ +.venv/ +venv/ +tests/ +state.json +state.json.tmp +__pycache__/ +*.pyc +.coverage +htmlcov/ + +# Host-specific deployment files, irrelevant inside the container. +wlc-monitor.service +wlc-monitor.timer +README.md +CONTRIBUTING.md +SECURITY.md diff --git a/.env.example b/.env.example new file mode 100644 index 0000000..c43d7cc --- /dev/null +++ b/.env.example @@ -0,0 +1,36 @@ +# Copy to .env and fill in. Contains SNMP credentials -- chmod 600, never commit. +# cp .env.example .env && chmod 600 .env + +# --- controller --- +WLC_HOST=192.0.2.10 +WLC_PORT=161 +WLC_SNMP_USER=wlcmon +WLC_SNMP_AUTH= +WLC_SNMP_PRIV= +# sha256 is preferred when the controller supports it; use sha for older setups. +WLC_SNMP_AUTH_PROTOCOL=sha256 +WLC_TIMEOUT=5 +WLC_RETRIES=1 + +# --- mail --- +MAIL_SMTP_HOST=smtp.example.com +MAIL_SMTP_PORT=587 +MAIL_SMTP_SECURITY=starttls +MAIL_SMTP_USER= +MAIL_SMTP_PASSWORD= +MAIL_FROM=wlc-monitor@example.com +# Comma-separated for multiple recipients. +MAIL_TO=network-ops@example.com +MAIL_SUBJECT_PREFIX=[WLC] + +# --- monitor --- +TZ=UTC +# Seconds between polls. This is also your detection resolution: an AP that +# drops and recovers inside one interval is never seen. +MONITOR_INTERVAL=300 +# Consecutive failed polls before declaring the controller unreachable. +# 2 x 300s means roughly 10 minutes of silence before alerting, which rides +# out a single dropped UDP packet without crying wolf. +MONITOR_FAIL_THRESHOLD=2 +# Alert if running-config stays unsaved this many minutes. +MONITOR_UNSAVED_MINUTES=60 diff --git a/.github/ISSUE_TEMPLATE/bug_report.yml b/.github/ISSUE_TEMPLATE/bug_report.yml new file mode 100644 index 0000000..449c019 --- /dev/null +++ b/.github/ISSUE_TEMPLATE/bug_report.yml @@ -0,0 +1,67 @@ +name: Bug report +description: Report reproducible incorrect behavior +title: "[Bug]: " +labels: ["bug"] +body: + - type: markdown + attributes: + value: | + Do not include credentials, private addresses, AP names, `config.ini`, + `.env`, `state.json`, or unredacted logs. Report security issues using + the private process in SECURITY.md. + - type: input + id: version + attributes: + label: Version or commit + placeholder: v0.1.0 or commit SHA + validations: + required: true + - type: dropdown + id: installation + attributes: + label: Installation mode + options: + - Docker Compose + - systemd + - Manual Python invocation + validations: + required: true + - type: input + id: runtime + attributes: + label: Runtime + description: Python version, or Docker Engine and Compose versions + validations: + required: true + - type: input + id: controller + attributes: + label: Controller model and IOS XE version + description: Do not include hostnames, addresses, serial numbers, or AP names. + validations: + required: true + - type: textarea + id: steps + attributes: + label: Steps to reproduce + validations: + required: true + - type: textarea + id: expected + attributes: + label: Expected behavior + validations: + required: true + - type: textarea + id: actual + attributes: + label: Actual behavior and redacted logs + validations: + required: true + - type: checkboxes + id: privacy + attributes: + label: Data-safety confirmation + options: + - label: I removed credentials, private infrastructure details, and personal data. + required: true diff --git a/.github/dependabot.yml b/.github/dependabot.yml new file mode 100644 index 0000000..8ba1e84 --- /dev/null +++ b/.github/dependabot.yml @@ -0,0 +1,14 @@ +version: 2 +updates: + - package-ecosystem: pip + directory: / + schedule: + interval: monthly + - package-ecosystem: docker + directory: / + schedule: + interval: monthly + - package-ecosystem: github-actions + directory: / + schedule: + interval: monthly diff --git a/.github/pull_request_template.md b/.github/pull_request_template.md new file mode 100644 index 0000000..0a8f4f2 --- /dev/null +++ b/.github/pull_request_template.md @@ -0,0 +1,18 @@ +## Summary + +Describe the problem and the focused solution. + +## Verification + +- [ ] `python -m py_compile wlc_monitor.py` +- [ ] `python -m unittest discover -v` +- [ ] `python -m pip check` +- [ ] Docker checks run, or not applicable +- [ ] User-facing documentation updated, or not applicable +- [ ] No credentials, private infrastructure data, runtime state, or personal data + are present in the diff + +## Operational impact + +Call out compatibility, configuration, alerting, state-migration, or deployment +effects. diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml new file mode 100644 index 0000000..e2b0c26 --- /dev/null +++ b/.github/workflows/ci.yml @@ -0,0 +1,35 @@ +name: CI + +on: + push: + pull_request: + +permissions: + contents: read + +jobs: + test: + runs-on: ubuntu-latest + strategy: + fail-fast: false + matrix: + python-version: ["3.10", "3.12", "3.13"] + steps: + - uses: actions/checkout@v4 + - uses: actions/setup-python@v5 + with: + python-version: ${{ matrix.python-version }} + cache: pip + - run: python -m pip install --upgrade pip + - run: python -m pip install -r requirements.txt + - run: python -m py_compile wlc_monitor.py + - run: python -m unittest discover -v + - run: python -m pip check + + container: + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@v4 + - run: cp .env.example .env + - run: docker compose config --quiet + - run: docker build --tag wlc-monitor:test . diff --git a/.gitignore b/.gitignore new file mode 100644 index 0000000..136bfc6 --- /dev/null +++ b/.gitignore @@ -0,0 +1,27 @@ +# Local configuration and credentials +.env +.env.* +!.env.example +config.ini + +# Runtime state and logs +state.json +state.json.tmp +*.log + +# Python environments and generated files +.venv/ +venv/ +__pycache__/ +*.py[cod] +.pytest_cache/ +.ruff_cache/ +.mypy_cache/ +.coverage +htmlcov/ + +# Editor and operating-system metadata +.DS_Store +.idea/ +.vscode/ +*.swp diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md new file mode 100644 index 0000000..fa761c3 --- /dev/null +++ b/CONTRIBUTING.md @@ -0,0 +1,54 @@ +# Contributing + +Thanks for helping improve WLC Monitor. + +## Before you start + +Use GitHub Issues for reproducible bugs and focused feature proposals. Search +existing issues first. Use the private process in [SECURITY.md](SECURITY.md) +for vulnerabilities. + +## Development setup + +```bash +python3 -m venv .venv +. .venv/bin/activate +python -m pip install -r requirements.txt +cp config.ini.example config.ini +chmod 600 config.ini +``` + +Never commit real credentials or infrastructure data. Use reserved example +addresses/domains and synthetic state fixtures. + +## Making changes + +- Create a focused branch. +- Add or update tests for behavior changes. +- Update the README and configuration examples for user-visible changes. +- Keep unrelated formatting out of the patch. + +Run the local checks before opening a pull request: + +```bash +python -m py_compile wlc_monitor.py +python -m unittest discover -v +python -m pip check +``` + +If Docker is available, also run: + +```bash +cp .env.example .env +docker compose config --quiet +docker build -t wlc-monitor:test . +``` + +Remove the local `.env` afterward or keep it private; it is ignored by Git. + +## Pull requests + +Explain the problem and solution, link related issues, list the checks you ran, +and call out compatibility or operational impact. Confirm that the diff +contains no credentials, `.env`, `config.ini`, `state.json`, controller/AP +identifiers, or personal data. diff --git a/Dockerfile b/Dockerfile new file mode 100644 index 0000000..3821a53 --- /dev/null +++ b/Dockerfile @@ -0,0 +1,38 @@ +# syntax=docker/dockerfile:1 +FROM python:3.12-slim + +# Unbuffered so `docker logs` shows poll output as it happens rather than in +# 4KB bursts -- stdout is block-buffered when it is not a TTY. +ENV PYTHONUNBUFFERED=1 \ + PYTHONDONTWRITEBYTECODE=1 + +WORKDIR /app + +# Dependencies first so edits to the script do not invalidate the layer. +# cryptography is required, not optional: without it pysnmp cannot do AES +# privacy and every poll fails with "Ciphering services not available". +COPY requirements.txt . +RUN pip install --no-cache-dir -r requirements.txt + +COPY wlc_monitor.py . + +# Run unprivileged. The image needs no root at runtime: outbound UDP/161 and +# TCP/SMTP only, plus one JSON file in /data. +RUN useradd --system --uid 10001 --no-create-home --shell /usr/sbin/nologin wlcmon \ + && mkdir -p /data \ + && chown wlcmon:wlcmon /data +USER wlcmon + +# State lives here; mount a named volume so alert de-duplication and the +# "consecutive failures" counter survive a restart. Without persistence the +# container starts with a blank comparison baseline after recreation. +VOLUME ["/data"] +ENV MONITOR_STATE_FILE=/data/state.json + +# Reports on the poll loop, not on the controller: an unreachable WLC means the +# monitor is working, and restarting it would discard alert-suppression state. +HEALTHCHECK --interval=60s --timeout=10s --start-period=90s --retries=3 \ + CMD ["python", "/app/wlc_monitor.py", "--healthcheck"] + +ENTRYPOINT ["python", "/app/wlc_monitor.py"] +CMD ["--loop"] diff --git a/LICENSE b/LICENSE new file mode 100644 index 0000000..92a0c59 --- /dev/null +++ b/LICENSE @@ -0,0 +1,21 @@ +MIT License + +Copyright (c) 2026 WLC Monitor contributors + +Permission is hereby granted, free of charge, to any person obtaining a copy +of this software and associated documentation files (the "Software"), to deal +in the Software without restriction, including without limitation the rights +to use, copy, modify, merge, publish, distribute, sublicense, and/or sell +copies of the Software, and to permit persons to whom the Software is +furnished to do so, subject to the following conditions: + +The above copyright notice and this permission notice shall be included in all +copies or substantial portions of the Software. + +THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +SOFTWARE. diff --git a/README.md b/README.md new file mode 100644 index 0000000..f1949c8 --- /dev/null +++ b/README.md @@ -0,0 +1,237 @@ +# WLC Monitor + +WLC Monitor polls a Cisco Catalyst 9800 wireless LAN controller over SNMPv3 +and sends email when controller, access-point, or configuration state changes. +It can run continuously in Docker or as a one-shot systemd timer. + +Polling makes silence observable: a failed poll is itself a signal, whereas +traps and syslog stop when a controller becomes unreachable. + +## Alerts + +| Severity | Condition | Detection | +|---|---|---| +| CRITICAL | Controller unreachable | Configured number of consecutive failed polls | +| CRITICAL | Controller rebooted | `sysUpTime` reset | +| CRITICAL | AP down | AP disappears from `cLApTable` | +| WARNING | AP rebooted | AP uptime reset | +| WARNING | AP rejoined | Association uptime reset while AP uptime did not | +| NOTICE | Running configuration changed | `ccmHistoryRunningLastChanged` advanced | +| NOTICE | Configuration remains unsaved | Running change is newer than the saved change beyond the threshold | +| RECOVERY | Controller or AP returns | Transition back to the available state | + +TimeTicks wrap at about 497 days. WLC Monitor compares the apparent counter +movement with elapsed wall time so a normal wrap is not reported as a reboot. + +## Requirements and compatibility + +- Docker Engine with Compose v2; or Python 3.10 or newer. +- A Cisco Catalyst 9800 reachable over SNMPv3 authPriv. +- Access to the CISCO-LWAPP-AP-MIB and CISCO-CONFIG-MAN-MIB objects used by the + controller. +- A reachable SMTP server. + +The implementation was developed against a Catalyst 9800-CL running IOS XE +17.12. Other 9800 models and releases may expose different MIB behavior; +compatibility reports are welcome. + +SMTP supports STARTTLS, implicit TLS, optional username/password +authentication, and plaintext delivery to a trusted relay. Do not select +`none` across an untrusted network: alert bodies contain infrastructure names +and operational state. Authentication is rejected in plaintext mode so SMTP +credentials cannot be sent over an unencrypted connection. + +## Quick start with Docker + +```bash +cp .env.example .env +chmod 600 .env +# Edit .env and set the controller, SNMP credentials, mail server, +# sender, and recipients. +docker compose up -d --build +``` + +Verify the controller and mail paths: + +```bash +docker compose exec wlc-monitor python /app/wlc_monitor.py --show +docker compose exec wlc-monitor python /app/wlc_monitor.py --test-email +docker compose logs -f wlc-monitor +docker inspect --format '{{.State.Health.Status}}' wlc-monitor +``` + +The container runs unprivileged with a read-only root filesystem. Its named +`/data` volume holds comparison state, failure counts, and any email awaiting +retry. Preserve this volume across upgrades. Losing it creates a blank +baseline: the next successful poll is deliberately silent, and changes during +the gap cannot be reconstructed. + +The health check reports whether the poll loop is progressing, not whether the +controller is reachable. Controller failure is a condition the monitor is +expected to observe, not a reason to restart it. + +## Configuration + +Environment variables override values in `config.ini`. In containers, prefer +environment variables or mounted secret files. For a host install, begin with +`config.ini.example`. + +| Variable | Required/default | Meaning | +|---|---|---| +| `WLC_HOST` | required | Controller hostname or address | +| `WLC_PORT` | `161` | SNMP port | +| `WLC_SNMP_USER` | required | SNMPv3 username | +| `WLC_SNMP_AUTH` | required | Authentication passphrase | +| `WLC_SNMP_PRIV` | required | AES privacy passphrase | +| `WLC_SNMP_AUTH_PROTOCOL` | `sha` | `sha`, `sha224`, `sha256`, `sha384`, or `sha512` | +| `WLC_TIMEOUT` | `5` | Per-request timeout in seconds | +| `WLC_RETRIES` | `1` | SNMP retries | +| `MAIL_SMTP_HOST` | required | SMTP server | +| `MAIL_SMTP_PORT` | `25` | SMTP port | +| `MAIL_SMTP_SECURITY` | `none` | `none`, `starttls`, or `ssl` | +| `MAIL_SMTP_USER` | optional | SMTP username; password must also be set | +| `MAIL_SMTP_PASSWORD` | optional | SMTP password; username must also be set | +| `MAIL_FROM` | required | Envelope/header sender | +| `MAIL_TO` | required | Comma-separated recipients | +| `MAIL_SUBJECT_PREFIX` | `[WLC]` | Subject prefix | +| `MONITOR_INTERVAL` | `300` | Seconds between continuous polls | +| `MONITOR_FAIL_THRESHOLD` | `2` | Consecutive failures before alerting | +| `MONITOR_UNSAVED_MINUTES` | `60` | Time before unsaved configuration alerts | +| `MONITOR_STATE_FILE` | `/data/state.json` | Persistent state path | +| `TZ` | `UTC` in Compose | Time zone used in alert timestamps | + +For secrets, `WLC_SNMP_AUTH_FILE`, `WLC_SNMP_PRIV_FILE`, and +`MAIL_SMTP_PASSWORD_FILE` may point to mounted files instead of placing values +directly in the process environment. Set either the direct variable or its +`_FILE` form, never both. + +The SNMP authentication protocol must match the controller user. SHA-2 is +preferred where the IOS XE release supports it; `sha` remains available for +older configurations. + +## Install with systemd + +Create a dedicated account and install the files: + +```bash +sudo useradd --system --no-create-home --shell /usr/sbin/nologin wlcmon +sudo install -d -o root -g wlcmon -m 0750 /opt/wlc-monitor +sudo install -o root -g wlcmon -m 0755 wlc_monitor.py /opt/wlc-monitor/ +sudo install -o root -g wlcmon -m 0644 requirements.txt README.md /opt/wlc-monitor/ +sudo cp config.ini.example /opt/wlc-monitor/config.ini +sudo chown root:wlcmon /opt/wlc-monitor/config.ini +sudo chmod 0640 /opt/wlc-monitor/config.ini +sudo python3 -m venv /opt/wlc-monitor/venv +sudo /opt/wlc-monitor/venv/bin/pip install -r /opt/wlc-monitor/requirements.txt +``` + +Edit `/opt/wlc-monitor/config.ini`, keeping +`state_file = /var/lib/wlc-monitor/state.json`, then install and start the +timer: + +```bash +sudo install -m 0644 wlc-monitor.service wlc-monitor.timer /etc/systemd/system/ +sudo systemctl daemon-reload +sudo systemctl enable --now wlc-monitor.timer +``` + +Verify: + +```bash +sudo systemctl start wlc-monitor.service +journalctl -u wlc-monitor.service -n 20 +systemctl list-timers wlc-monitor.timer +``` + +The unit creates `/var/lib/wlc-monitor` privately and applies a restrictive +umask. The configuration is readable only by root and the dedicated service +group. + +## Local command-line use + +```bash +python3 -m venv .venv +. .venv/bin/activate +python -m pip install -r requirements.txt +cp config.ini.example config.ini +chmod 600 config.ini +# Edit config.ini and use a writable state path such as ./state.json. + +python wlc_monitor.py --show +python wlc_monitor.py --dry-run --no-save +python wlc_monitor.py --test-email +python wlc_monitor.py --loop --interval 60 +python wlc_monitor.py --healthcheck +``` + +`--dry-run` prints alerts instead of sending them. It still updates state +unless combined with `--no-save`. `--show` returns a nonzero exit status when +the poll fails, making it suitable for scripts. + +The first successful poll is deliberately silent because there is no previous +state to compare. + +## Controller configuration + +Use a read-only SNMPv3 authPriv user and restrict its ACL to the single +monitoring host. Exact SHA-2 syntax varies by IOS XE release; consult the +controller documentation and make `WLC_SNMP_AUTH_PROTOCOL` match. + +```text +ip access-list standard SNMP-MON + permit host +snmp-server group WLCMON v3 priv read v1default access SNMP-MON +snmp-server user wlcmon WLCMON v3 auth \ + priv aes 128 +``` + +Do not paste real controller configuration, credentials, addresses, or device +output into issues. + +## Delivery and state behavior + +State is written atomically with mode `0600`. It can still contain controller +and AP names, software details, and queued alert bodies, so treat it as +operationally sensitive and do not commit or publish it. + +Transition alerts that fail SMTP delivery are stored in a small durable FIFO +outbox and retried on the next poll. Delivery stops at the first failure so a +recovery cannot arrive before its outage. A continuously unsaved configuration +alerts once when it first exceeds the threshold, then becomes eligible again +after it is saved. A crash between the SMTP server accepting a message and the +state write can still cause a duplicate; recipients should tolerate +at-least-once delivery. + +## Operational limitations + +- SNMP reveals that a configuration changed, not what changed. Inspect the + controller's configuration history for the actual diff. +- The poll interval is the detection resolution. An AP that disconnects and + recovers entirely between polls is invisible. +- AP renames look like a down and join pair because APs are keyed by name. +- The monitoring host remains a single point of failure. Use an external + dead-man/heartbeat service if silent monitor failure must be detected. +- A missing or partial AP table is treated as a failed poll to avoid false + reboot/rejoin alerts. + +## Development + +Run the same checks used by CI: + +```bash +python -m py_compile wlc_monitor.py +python -m unittest discover -v +python -m pip check +docker compose config --quiet +docker build -t wlc-monitor:test . +``` + +## Contributing and security + +Bug reports and pull requests are welcome; see +[CONTRIBUTING.md](CONTRIBUTING.md). Report vulnerabilities privately as +described in [SECURITY.md](SECURITY.md), not in a public issue. + +## License + +WLC Monitor is available under the [MIT License](LICENSE). diff --git a/SECURITY.md b/SECURITY.md new file mode 100644 index 0000000..c2327e9 --- /dev/null +++ b/SECURITY.md @@ -0,0 +1,22 @@ +# Security Policy + +## Supported versions + +Until tagged releases are published, only the current default branch receives +security fixes. Supported release versions will be listed here once releases +begin. + +## Reporting a vulnerability + +Do not open a public issue for a suspected vulnerability. Prefer GitHub's +private vulnerability reporting from the repository's Security tab. Include +the affected commit or version, impact, reproduction steps, and any suggested +mitigation. + +If private reporting is not enabled, ask the maintainer for a private reporting +channel without disclosing vulnerability details publicly. + +Do not include real SNMP credentials, `.env`, `config.ini`, `state.json`, +controller addresses, AP names, or unredacted logs. Reports will be +acknowledged and updated on a best-effort basis; please allow time for a fix +before public disclosure. diff --git a/config.ini.example b/config.ini.example new file mode 100644 index 0000000..ea5366b --- /dev/null +++ b/config.ini.example @@ -0,0 +1,34 @@ +# Copy to config.ini, fill in the required values, and keep it private: +# cp config.ini.example config.ini +# chmod 600 config.ini + +[wlc] +host = 192.0.2.10 +port = 161 +snmp_user = wlcmon +snmp_auth = +snmp_priv = +# Prefer sha256. Use sha only when required by an older controller. +auth_protocol = sha256 +timeout = 5 +retries = 1 + +[mail] +smtp_host = smtp.example.com +smtp_port = 587 +# One of: none, starttls, ssl +smtp_security = starttls +# Leave both blank for an unauthenticated trusted relay. +smtp_user = +smtp_password = +from = wlc-monitor@example.com +to = network-ops@example.com +subject_prefix = [WLC] + +[monitor] +# This path matches wlc-monitor.service. For a user-run checkout, change it to +# a private writable path such as ./state.json. +state_file = /var/lib/wlc-monitor/state.json +fail_threshold = 2 +unsaved_minutes = 60 +interval = 300 diff --git a/docker-compose.yml b/docker-compose.yml new file mode 100644 index 0000000..cae0187 --- /dev/null +++ b/docker-compose.yml @@ -0,0 +1,36 @@ +services: + wlc-monitor: + build: . + image: wlc-monitor:latest + container_name: wlc-monitor + restart: unless-stopped + + # Credentials come from .env (gitignored, chmod 600), never from the image. + env_file: [.env] + + environment: + MONITOR_STATE_FILE: /data/state.json + # Timestamps in alert bodies use the container's zone; without this they + # render as UTC and will not match the controller's log timestamps. + TZ: ${TZ:-UTC} + + volumes: + # Named volume: alert de-duplication and the consecutive-failure counter + # live here. Lose it and the container re-alerts from a blank slate. + - wlc-monitor-state:/data + + # Read-only rootfs; the only thing that needs writing is /data. + read_only: true + tmpfs: + - /tmp + security_opt: + - no-new-privileges:true + cap_drop: + - ALL + + logging: + driver: json-file + options: {max-size: "10m", max-file: "3"} + +volumes: + wlc-monitor-state: diff --git a/pyproject.toml b/pyproject.toml new file mode 100644 index 0000000..3a27d1b --- /dev/null +++ b/pyproject.toml @@ -0,0 +1,20 @@ +[build-system] +requires = ["setuptools>=77"] +build-backend = "setuptools.build_meta" + +[project] +name = "wlc-monitor" +version = "0.1.0" +description = "SNMPv3 health monitoring and email alerts for Cisco Catalyst 9800 controllers" +readme = "README.md" +requires-python = ">=3.10" +dependencies = [ + "cryptography==49.0.0", + "pysnmp==7.1.27", +] + +[project.scripts] +wlc-monitor = "wlc_monitor:main" + +[tool.setuptools] +py-modules = ["wlc_monitor"] diff --git a/requirements.txt b/requirements.txt new file mode 100644 index 0000000..8419790 --- /dev/null +++ b/requirements.txt @@ -0,0 +1,5 @@ +cffi==2.1.0 +cryptography==49.0.0 +pyasn1==0.6.4 +pycparser==3.0 +pysnmp==7.1.27 diff --git a/tests/__init__.py b/tests/__init__.py new file mode 100644 index 0000000..c637051 --- /dev/null +++ b/tests/__init__.py @@ -0,0 +1 @@ +"""WLC Monitor test suite.""" diff --git a/tests/test_wlc_monitor.py b/tests/test_wlc_monitor.py new file mode 100644 index 0000000..0a6d37c --- /dev/null +++ b/tests/test_wlc_monitor.py @@ -0,0 +1,374 @@ +import json +import os +import stat +import tempfile +import unittest +from contextlib import redirect_stderr, redirect_stdout +from io import StringIO +from pathlib import Path +from types import SimpleNamespace +from unittest.mock import ANY, AsyncMock, MagicMock, call, patch + +import wlc_monitor + + +def sample_state(**changes): + state = { + "ok": True, + "sys_name": "test-wlc", + "sys_descr": "Synthetic controller", + "sys_uptime": 2_000_000, + "run_last_changed": 1_000_000, + "run_last_saved": 1_000_000, + "aps": {}, + "consecutive_failures": 0, + "last_poll": "2026-01-01T00:00:00+00:00", + } + state.update(changes) + return state + + +class AlertConfig: + fail_threshold = 2 + unsaved_ticks = 60 * 60 * 100 + host = "192.0.2.10" + + +class CounterTests(unittest.TestCase): + def test_increasing_counter_is_not_reset(self): + self.assertFalse(wlc_monitor.counter_reset(100, 200, 1)) + + def test_small_decrease_is_reset(self): + self.assertTrue(wlc_monitor.counter_reset(50_000, 100, 60)) + + def test_32_bit_wrap_is_not_reset(self): + self.assertFalse( + wlc_monitor.counter_reset(2**32 - 100, 200, elapsed_s=5)) + + +class AlertTests(unittest.TestCase): + def test_first_poll_is_silent(self): + self.assertEqual( + wlc_monitor.build_alerts({}, sample_state(), AlertConfig()), []) + + def test_unsaved_config_alerts_once_until_saved(self): + prev = sample_state( + run_last_changed=1_000_000, run_last_saved=900_000) + cur = sample_state( + sys_uptime=2_000_000, + run_last_changed=1_000_000, + run_last_saved=900_000, + ) + first = wlc_monitor.build_alerts(prev, cur, AlertConfig()) + self.assertEqual([a[1] for a in first], [ + "test-wlc has unsaved config"]) + self.assertTrue(cur["_unsaved_alerted"]) + + later = sample_state( + sys_uptime=2_030_000, + run_last_changed=1_000_000, + run_last_saved=900_000, + ) + self.assertEqual( + wlc_monitor.build_alerts(cur, later, AlertConfig()), []) + + saved = sample_state( + sys_uptime=2_060_000, + run_last_changed=1_000_000, + run_last_saved=1_000_000, + ) + wlc_monitor.build_alerts(later, saved, AlertConfig()) + self.assertFalse(saved["_unsaved_alerted"]) + + +class StateTests(unittest.TestCase): + def test_state_is_written_atomically_with_private_permissions(self): + with tempfile.TemporaryDirectory() as directory: + path = Path(directory) / "state.json" + wlc_monitor.save_state(path, {"ok": True}) + self.assertEqual(json.loads(path.read_text()), {"ok": True}) + self.assertEqual( + stat.S_IMODE(path.stat().st_mode), 0o600) + + def test_naive_or_invalid_timestamp_is_tolerated(self): + self.assertIsNone( + wlc_monitor.elapsed_since({"last_poll": "2026-01-01T00:00:00"})) + self.assertIsNone( + wlc_monitor.elapsed_since({"last_poll": "not-a-timestamp"})) + + def test_non_object_state_is_ignored(self): + with tempfile.TemporaryDirectory() as directory: + path = Path(directory) / "state.json" + path.write_text("[]") + self.assertEqual(wlc_monitor.load_state(path), {}) + + +class ConfigTests(unittest.TestCase): + required_env = { + "WLC_SNMP_AUTH": "synthetic-auth", + "WLC_SNMP_PRIV": "synthetic-privacy", + } + + def write_config(self, directory): + path = Path(directory) / "config.ini" + path.write_text( + "[wlc]\n" + "host = 192.0.2.10\n" + "snmp_user = test\n" + "snmp_auth =\n" + "snmp_priv =\n" + "[mail]\n" + "smtp_host = smtp.example.com\n" + "from = monitor@example.com\n" + "to = ops@example.com\n" + "[monitor]\n" + "state_file = state.json\n" + ) + path.chmod(0o600) + return path + + def test_environment_overrides_file(self): + with tempfile.TemporaryDirectory() as directory: + path = self.write_config(directory) + with patch.dict( + os.environ, + {**self.required_env, "WLC_HOST": "198.51.100.20"}, + clear=True, + ): + cfg = wlc_monitor.Cfg(path) + self.assertEqual(cfg.host, "198.51.100.20") + + def test_invalid_interval_is_rejected(self): + with tempfile.TemporaryDirectory() as directory: + path = self.write_config(directory) + with patch.dict( + os.environ, + {**self.required_env, "MONITOR_INTERVAL": "0"}, + clear=True, + ): + with self.assertRaises(SystemExit): + wlc_monitor.Cfg(path) + + def test_negative_unsaved_threshold_is_rejected(self): + with tempfile.TemporaryDirectory() as directory: + path = self.write_config(directory) + with patch.dict( + os.environ, + { + **self.required_env, + "MONITOR_UNSAVED_MINUTES": "-1", + }, + clear=True, + ): + with self.assertRaises(SystemExit): + wlc_monitor.Cfg(path) + + def test_smtp_authentication_requires_encryption(self): + with tempfile.TemporaryDirectory() as directory: + path = self.write_config(directory) + with patch.dict( + os.environ, + { + **self.required_env, + "MAIL_SMTP_SECURITY": "none", + "MAIL_SMTP_USER": "test-user", + "MAIL_SMTP_PASSWORD": "synthetic-password", + }, + clear=True, + ): + with self.assertRaises(SystemExit): + wlc_monitor.Cfg(path) + + def test_secret_can_be_read_from_file(self): + with tempfile.TemporaryDirectory() as directory: + path = self.write_config(directory) + secret = Path(directory) / "auth" + secret.write_text("from-file\n") + with patch.dict( + os.environ, + { + "WLC_SNMP_PRIV": self.required_env["WLC_SNMP_PRIV"], + "WLC_SNMP_AUTH_FILE": str(secret), + }, + clear=True, + ): + cfg = wlc_monitor.Cfg(path) + self.assertEqual(cfg.auth, "from-file") + + +class DeliveryTests(unittest.TestCase): + def test_failed_transition_email_is_retried(self): + with tempfile.TemporaryDirectory() as directory: + state_path = Path(directory) / "state.json" + previous = sample_state( + aps={ + "ap-one": { + "index": "1", + "ap_uptime": 1_000, + "assoc_uptime": 900, + } + } + ) + wlc_monitor.save_state(state_path, previous) + current = sample_state(sys_uptime=2_030_000) + + cfg = SimpleNamespace( + state_file=state_path, + host="192.0.2.10", + port=161, + user="test", + auth="auth", + priv="priv", + auth_protocol="sha256", + timeout=1, + retries=0, + fail_threshold=2, + unsaved_ticks=360_000, + mail_from="monitor@example.com", + mail_to=["ops@example.com"], + subject_prefix="[TEST]", + smtp_host="smtp.example.com", + smtp_port=587, + smtp_security="starttls", + smtp_user=None, + smtp_password=None, + ) + args = SimpleNamespace(show=False, dry_run=False, no_save=False) + + poller = SimpleNamespace(poll=AsyncMock(return_value=current)) + with patch.object(wlc_monitor, "Poller", return_value=poller), \ + patch.object( + wlc_monitor, "send_email", + side_effect=OSError("temporary failure"), + ), redirect_stdout(StringIO()), redirect_stderr(StringIO()): + self.assertEqual(wlc_monitor.run_once(cfg, args), 1) + + failed_state = json.loads(state_path.read_text()) + self.assertEqual(len(failed_state["_pending_emails"]), 1) + + poller.poll = AsyncMock(return_value=sample_state( + sys_uptime=2_060_000)) + with patch.object(wlc_monitor, "Poller", return_value=poller), \ + patch.object(wlc_monitor, "send_email") as send, \ + redirect_stdout(StringIO()): + self.assertEqual(wlc_monitor.run_once(cfg, args), 0) + send.assert_called_once() + + recovered_state = json.loads(state_path.read_text()) + self.assertNotIn("_pending_emails", recovered_state) + + def test_outbox_stops_after_first_failure_to_preserve_order(self): + with tempfile.TemporaryDirectory() as directory: + state_path = Path(directory) / "state.json" + previous = sample_state( + ok=False, + consecutive_failures=2, + _pending_emails=[ + {"subject": "OUTAGE", "body": "controller unavailable"} + ], + ) + wlc_monitor.save_state(state_path, previous) + + cfg = SimpleNamespace( + state_file=state_path, + host="192.0.2.10", + port=161, + user="test", + auth="auth", + priv="priv", + auth_protocol="sha256", + timeout=1, + retries=0, + fail_threshold=2, + unsaved_ticks=360_000, + mail_from="monitor@example.com", + mail_to=["ops@example.com"], + subject_prefix="[TEST]", + smtp_host="smtp.example.com", + smtp_port=587, + smtp_security="starttls", + smtp_user=None, + smtp_password=None, + ) + args = SimpleNamespace(show=False, dry_run=False, no_save=False) + poller = SimpleNamespace( + poll=AsyncMock(return_value=sample_state())) + + with patch.object(wlc_monitor, "Poller", return_value=poller), \ + patch.object( + wlc_monitor, "send_email", + side_effect=OSError("still unavailable"), + ) as send, redirect_stdout(StringIO()), \ + redirect_stderr(StringIO()): + self.assertEqual(wlc_monitor.run_once(cfg, args), 1) + send.assert_called_once() + + saved = json.loads(state_path.read_text()) + queued = saved["_pending_emails"] + self.assertEqual(len(queued), 2) + self.assertEqual(queued[0]["subject"], "OUTAGE") + self.assertIn("RECOVERY", queued[1]["subject"]) + + +class EmailTests(unittest.TestCase): + def config(self, security, user=None, password=None): + return SimpleNamespace( + mail_from="monitor@example.com", + mail_to=["ops@example.com"], + smtp_host="smtp.example.com", + smtp_port=587, + smtp_security=security, + smtp_user=user, + smtp_password=password, + ) + + def test_starttls_happens_before_authentication(self): + server = MagicMock() + session = server.__enter__.return_value + context = object() + with patch.object(wlc_monitor.ssl, "create_default_context", + return_value=context), \ + patch.object(wlc_monitor.smtplib, "SMTP", + return_value=server): + wlc_monitor.send_email( + self.config("starttls", "user", "password"), + "subject", + "body", + ) + self.assertEqual( + session.method_calls, + [ + call.starttls(context=context), + call.login("user", "password"), + call.send_message(ANY), + ], + ) + + def test_implicit_tls_uses_smtp_ssl(self): + server = MagicMock() + context = object() + with patch.object(wlc_monitor.ssl, "create_default_context", + return_value=context), \ + patch.object( + wlc_monitor.smtplib, "SMTP_SSL", return_value=server + ) as smtp_ssl: + wlc_monitor.send_email( + self.config("ssl"), "subject", "body") + smtp_ssl.assert_called_once_with( + "smtp.example.com", 587, timeout=15, context=context) + server.__enter__.return_value.send_message.assert_called_once() + + def test_plain_relay_does_not_start_tls_or_authenticate(self): + server = MagicMock() + session = server.__enter__.return_value + with patch.object(wlc_monitor.smtplib, "SMTP", + return_value=server): + wlc_monitor.send_email( + self.config("none"), "subject", "body") + session.starttls.assert_not_called() + session.login.assert_not_called() + session.send_message.assert_called_once() + + +if __name__ == "__main__": + unittest.main() diff --git a/wlc-monitor.service b/wlc-monitor.service new file mode 100644 index 0000000..c1d3466 --- /dev/null +++ b/wlc-monitor.service @@ -0,0 +1,34 @@ +[Unit] +Description=Cisco Catalyst 9800 WLC health check (SNMP -> email) +Documentation=file:/opt/wlc-monitor/README.md +After=network-online.target +Wants=network-online.target + +[Service] +Type=oneshot +User=wlcmon +Group=wlcmon +WorkingDirectory=/opt/wlc-monitor +ExecStart=/opt/wlc-monitor/venv/bin/python /opt/wlc-monitor/wlc_monitor.py \ + --config /opt/wlc-monitor/config.ini + +# Creates and owns /var/lib/wlc-monitor for the state file. +StateDirectory=wlc-monitor +StateDirectoryMode=0700 +UMask=0077 + +# The service only needs to talk UDP/161 outbound and SMTP outbound, and to +# write one JSON file. Everything else is off. +NoNewPrivileges=true +PrivateTmp=true +PrivateDevices=true +ProtectSystem=strict +ProtectHome=true +ProtectKernelTunables=true +ProtectKernelModules=true +ProtectControlGroups=true +RestrictAddressFamilies=AF_INET AF_INET6 +RestrictNamespaces=true +LockPersonality=true +MemoryDenyWriteExecute=true +SystemCallArchitectures=native diff --git a/wlc-monitor.timer b/wlc-monitor.timer new file mode 100644 index 0000000..3a888ac --- /dev/null +++ b/wlc-monitor.timer @@ -0,0 +1,12 @@ +[Unit] +Description=Poll the 9800 WLC every 5 minutes + +[Timer] +# Give the network a moment after boot before the first poll, otherwise a +# cold start reliably produces a false "controller unreachable". +OnBootSec=3min +OnUnitActiveSec=5min +AccuracySec=30s + +[Install] +WantedBy=timers.target diff --git a/wlc_monitor.py b/wlc_monitor.py new file mode 100755 index 0000000..7097cb6 --- /dev/null +++ b/wlc_monitor.py @@ -0,0 +1,669 @@ +#!/usr/bin/env python3 +"""SNMP health check for the Catalyst 9800 WLC, with email alerting. + +One invocation = one poll. State is kept in a JSON file so the next run can +diff against it; drive it from a systemd timer (see wlc-monitor.timer). + +Why poll instead of receiving traps/syslog: a poll that FAILS is itself a +signal. Traps go silent when the controller dies, and silence is +indistinguishable from "everything is fine". This design alerts on absence. + + wlc_monitor.py --config config.ini # one poll, send alerts + wlc_monitor.py --config config.ini --dry-run # poll, print, send nothing + wlc_monitor.py --config config.ini --test-email + wlc_monitor.py --config config.ini --show # dump current state + +Requires: pysnmp (pip install pysnmp) +""" + +from __future__ import annotations + +import argparse +import asyncio +import configparser +import json +import os +import signal +import smtplib +import socket +import ssl +import sys +import threading +import warnings +from datetime import datetime, timezone +from email.message import EmailMessage +from pathlib import Path + +warnings.filterwarnings("ignore", message=".*CFB has been moved.*") + +try: + from pysnmp.hlapi.v3arch.asyncio import ( + SnmpEngine, UsmUserData, UdpTransportTarget, ContextData, + ObjectType, ObjectIdentity, get_cmd, bulk_cmd, + usmHMACSHAAuthProtocol, usmHMAC128SHA224AuthProtocol, + usmHMAC192SHA256AuthProtocol, usmHMAC256SHA384AuthProtocol, + usmHMAC384SHA512AuthProtocol, usmAesCfb128Protocol, + ) +except ImportError: + sys.exit("error: pysnmp not installed. pip install pysnmp") + +AUTH_PROTOCOLS = { + "sha": usmHMACSHAAuthProtocol, + "sha1": usmHMACSHAAuthProtocol, + "sha224": usmHMAC128SHA224AuthProtocol, + "sha256": usmHMAC192SHA256AuthProtocol, + "sha384": usmHMAC256SHA384AuthProtocol, + "sha512": usmHMAC384SHA512AuthProtocol, +} + +# ---------------------------------------------------------------- OIDs +# Scalars. +OID_SYS_DESCR = "1.3.6.1.2.1.1.1.0" +OID_SYS_UPTIME = "1.3.6.1.2.1.1.3.0" +OID_SYS_NAME = "1.3.6.1.2.1.1.5.0" +# CISCO-CONFIG-MAN-MIB: sysUpTime at which running/startup last changed. +OID_RUN_LAST_CHANGED = "1.3.6.1.4.1.9.9.43.1.1.1.0" +OID_RUN_LAST_SAVED = "1.3.6.1.4.1.9.9.43.1.1.2.0" + +# CISCO-LWAPP-AP-MIB cLApEntry columns. Column meanings were verified against +# `show ap uptime` on this controller rather than taken from a MIB reference: +# .5 = AP name .6 = AP up time .7 = association (CAPWAP) up time +# Only APs currently joined appear in this table, so presence == joined. +AP_TABLE = "1.3.6.1.4.1.9.9.513.1.1.1.1" +AP_COL_NAME = f"{AP_TABLE}.5" +AP_COL_UPTIME = f"{AP_TABLE}.6" +AP_COL_ASSOC_UPTIME = f"{AP_TABLE}.7" + +# TimeTicks are 32-bit hundredths of a second and wrap at ~497 days. A counter +# that "decreased" may simply have wrapped, so distinguish the two by how far +# it moved FORWARD (mod 2^32) versus how much wall-clock time actually passed +# between polls. Five minutes of real time cannot produce months of ticks. +TICKS_MAX = 2 ** 32 +WRAP_SLACK_S = 600 # tolerate clock skew / a missed poll or two +WRAP_FACTOR = 3 # and be generous about timer jitter + + +def ticks_to_human(t: int) -> str: + s = int(t) // 100 + d, s = divmod(s, 86400) + h, s = divmod(s, 3600) + m, s = divmod(s, 60) + if d: + return f"{d}d {h}h {m}m" + if h: + return f"{h}h {m}m" + return f"{m}m {s}s" + + +def counter_reset(old: int, new: int, elapsed_s: float | None = None) -> bool: + """True if `new` is a genuine reset of `old` rather than a 32-bit wrap. + + A wrapped counter has still moved forward by only the elapsed time; a reset + counter has effectively jumped backwards by nearly its whole range. Compare + the forward distance (mod 2^32) against the time that really passed. + """ + if new >= old: + return False + forward = (new - old) % TICKS_MAX + if elapsed_s is None: + # No timing reference (first diff after an upgrade, say). Only a value + # very close to the ceiling is plausibly a wrap. + return not (old > 4.2e9 and new < 1e7) + budget = (elapsed_s * WRAP_FACTOR + WRAP_SLACK_S) * 100 + return forward > budget + + +# ---------------------------------------------------------------- SNMP +class Poller: + def __init__(self, host, port, user, auth, priv, auth_protocol, + timeout, retries): + self.host, self.port = host, port + self.user, self.auth, self.priv = user, auth, priv + self.auth_protocol = auth_protocol + self.timeout, self.retries = timeout, retries + self.engine = SnmpEngine() + + def _creds(self): + return UsmUserData(self.user, self.auth, self.priv, + authProtocol=AUTH_PROTOCOLS[self.auth_protocol], + privProtocol=usmAesCfb128Protocol) + + async def _target(self): + return await UdpTransportTarget.create( + (self.host, self.port), timeout=self.timeout, retries=self.retries) + + async def get(self, tgt, *oids): + ei, es, _ix, vbs = await get_cmd( + self.engine, self._creds(), tgt, ContextData(), + *[ObjectType(ObjectIdentity(o)) for o in oids], lookupMib=False) + if ei: + raise RuntimeError(str(ei)) + if es: + raise RuntimeError(es.prettyPrint()) + return [vb[1] for vb in vbs] + + async def walk(self, tgt, base): + """Return {oid_suffix: value} for one table column.""" + out, objs = {}, [ObjectType(ObjectIdentity(base))] + while True: + ei, es, _ix, rows = await bulk_cmd( + self.engine, self._creds(), tgt, ContextData(), 0, 25, + *objs, lookupMib=False) + if ei: + raise RuntimeError(str(ei)) + if es: + raise RuntimeError(es.prettyPrint()) + if not rows: + break + done = False + for vb in rows: + oid, val = str(vb[0]), vb[1] + if not oid.startswith(base + "."): + done = True + break + out[oid[len(base) + 1:]] = val + if done: + break + objs = [ObjectType(ObjectIdentity(str(rows[-1][0])))] + return out + + async def poll(self) -> dict: + tgt = await self._target() + descr, uptime, name, changed, saved = await self.get( + tgt, OID_SYS_DESCR, OID_SYS_UPTIME, OID_SYS_NAME, + OID_RUN_LAST_CHANGED, OID_RUN_LAST_SAVED) + + names = await self.walk(tgt, AP_COL_NAME) + ups = await self.walk(tgt, AP_COL_UPTIME) + assoc = await self.walk(tgt, AP_COL_ASSOC_UPTIME) + + aps = {} + for idx, nm in names.items(): + missing = [ + column for column, values in ( + ("AP uptime", ups), ("association uptime", assoc) + ) if idx not in values + ] + if missing: + raise RuntimeError( + f"incomplete SNMP AP row {idx}: missing {', '.join(missing)}") + aps[str(nm)] = { + "index": idx, + "ap_uptime": int(ups[idx]), + "assoc_uptime": int(assoc[idx]), + } + return { + "ok": True, + "sys_name": str(name), + "sys_descr": str(descr)[:120], + "sys_uptime": int(uptime), + "run_last_changed": int(changed), + "run_last_saved": int(saved), + "aps": aps, + } + + +# ---------------------------------------------------------------- diff +def elapsed_since(prev: dict) -> float | None: + """Wall-clock seconds since the previous poll, if we can tell.""" + ts = prev.get("last_poll") + if not ts: + return None + try: + then = datetime.fromisoformat(ts) + except (TypeError, ValueError): + return None + if then.tzinfo is None: + return None + delta = (datetime.now(timezone.utc) - then).total_seconds() + return delta if delta >= 0 else None + + +def build_alerts(prev: dict, cur: dict, cfg) -> list[tuple[str, str, str]]: + """Return [(severity, title, detail)]. prev may be {} on first run.""" + alerts = [] + first_run = not prev.get("aps") and not prev.get("ok") + el = elapsed_since(prev) + + # --- reachability transitions ------------------------------------- + if cur["ok"] and prev.get("consecutive_failures", 0) >= cfg.fail_threshold: + alerts.append(( + "RECOVERY", f"{cur['sys_name']}: controller reachable again", + f"SNMP polling recovered after " + f"{prev.get('consecutive_failures')} consecutive failures.\n" + f"Controller uptime is now {ticks_to_human(cur['sys_uptime'])}.")) + + if not cur["ok"]: + fails = cur.get("consecutive_failures", 1) + if fails == cfg.fail_threshold: # alert once, on crossing + alerts.append(( + "CRITICAL", "WLC unreachable over SNMP", + f"{fails} consecutive failed polls of {cfg.host}.\n" + f"Last error: {cur.get('error')}\n\n" + "The controller may be down, rebooting, or unreachable " + "from this host.")) + return alerts + + if first_run: + return alerts # nothing to diff against; stay quiet + + # --- controller reboot -------------------------------------------- + if counter_reset(prev.get("sys_uptime", 0), cur["sys_uptime"], el): + alerts.append(( + "CRITICAL", f"{cur['sys_name']} rebooted", + f"Controller uptime went from " + f"{ticks_to_human(prev['sys_uptime'])} to " + f"{ticks_to_human(cur['sys_uptime'])}.\n" + f"{cur['sys_descr']}")) + + # --- config changes ----------------------------------------------- + # Only compare when we actually have a previous reading. Treating a missing + # baseline as 0 would report a config change after every failed poll. + prev_changed = prev.get("run_last_changed") + if prev_changed is not None and cur["run_last_changed"] > prev_changed and \ + not counter_reset(prev.get("sys_uptime", 0), cur["sys_uptime"], el): + alerts.append(( + "NOTICE", f"{cur['sys_name']} running-config changed", + f"running-config last changed at controller uptime " + f"{ticks_to_human(cur['run_last_changed'])} " + f"(previous poll saw {ticks_to_human(prev_changed)}).\n\n" + "SNMP reports THAT the config changed, not what changed. " + "Check `show archive config differences` or the session logs.")) + + # unsaved config that stays unsaved + if cur["run_last_changed"] > cur["run_last_saved"]: + unsaved_for = cur["sys_uptime"] - cur["run_last_changed"] + overdue = unsaved_for > cfg.unsaved_ticks + if overdue and not prev.get("_unsaved_alerted", False): + alerts.append(( + "NOTICE", f"{cur['sys_name']} has unsaved config", + f"running-config has differed from startup-config for " + f"{ticks_to_human(unsaved_for)}.\n" + "A reboot would lose those changes. Run `write memory`.")) + cur["_unsaved_alerted"] = overdue + else: + cur["_unsaved_alerted"] = False + + # --- AP state ------------------------------------------------------ + prev_aps, cur_aps = prev.get("aps", {}), cur["aps"] + + for nm in sorted(set(prev_aps) - set(cur_aps)): + alerts.append(( + "CRITICAL", f"AP down: {nm}", + f"{nm} is no longer registered to {cur['sys_name']}.\n" + f"At the previous poll it had been joined for " + f"{ticks_to_human(prev_aps[nm]['assoc_uptime'])}.\n\n" + f"APs still joined: {len(cur_aps)}")) + + for nm in sorted(set(cur_aps) - set(prev_aps)): + alerts.append(( + "RECOVERY", f"AP joined: {nm}", + f"{nm} registered to {cur['sys_name']}.\n" + f"AP has been up {ticks_to_human(cur_aps[nm]['ap_uptime'])}, " + f"joined {ticks_to_human(cur_aps[nm]['assoc_uptime'])} ago.\n\n" + f"APs joined: {len(cur_aps)}")) + + for nm in sorted(set(cur_aps) & set(prev_aps)): + p, c = prev_aps[nm], cur_aps[nm] + if counter_reset(p["ap_uptime"], c["ap_uptime"], el): + alerts.append(( + "WARNING", f"AP rebooted: {nm}", + f"{nm} uptime went from {ticks_to_human(p['ap_uptime'])} " + f"to {ticks_to_human(c['ap_uptime'])}.\n" + "The access point restarted (power, watchdog, or upgrade).")) + elif counter_reset(p["assoc_uptime"], c["assoc_uptime"], el): + # Tunnel flapped without the AP itself rebooting -- this is the + # case a naive up/down check silently misses. + alerts.append(( + "WARNING", f"AP re-joined controller: {nm}", + f"{nm} association uptime reset from " + f"{ticks_to_human(p['assoc_uptime'])} to " + f"{ticks_to_human(c['assoc_uptime'])}, but AP uptime " + f"({ticks_to_human(c['ap_uptime'])}) did not reset.\n\n" + "The CAPWAP tunnel dropped and re-established without the AP " + "rebooting -- typically a network path or controller issue " + "rather than an AP fault.")) + return alerts + + +# ---------------------------------------------------------------- email +def send_email(cfg, subject: str, body: str) -> None: + msg = EmailMessage() + msg["From"] = cfg.mail_from + msg["To"] = ", ".join(cfg.mail_to) + msg["Subject"] = subject + msg["X-WLC-Monitor"] = socket.gethostname() + msg.set_content(body) + context = ssl.create_default_context() + if cfg.smtp_security == "ssl": + server = smtplib.SMTP_SSL( + cfg.smtp_host, cfg.smtp_port, timeout=15, context=context) + else: + server = smtplib.SMTP(cfg.smtp_host, cfg.smtp_port, timeout=15) + with server as s: + if cfg.smtp_security == "starttls": + s.starttls(context=context) + if cfg.smtp_user: + s.login(cfg.smtp_user, cfg.smtp_password) + s.send_message(msg) + + +def format_body(detail: str, cur: dict, cfg) -> str: + now = datetime.now(timezone.utc).astimezone().strftime("%Y-%m-%d %H:%M:%S %Z") + lines = [detail, "", "-" * 58, + f"controller : {cfg.host}", + f"detected : {now}", + f"monitored : {socket.gethostname()}"] + if cur.get("ok"): + lines += [f"uptime : {ticks_to_human(cur['sys_uptime'])}", + f"APs joined : {len(cur['aps'])}"] + for nm, ap in sorted(cur["aps"].items()): + lines.append(f" {nm} up {ticks_to_human(ap['ap_uptime'])}, " + f"joined {ticks_to_human(ap['assoc_uptime'])}") + return "\n".join(lines) + + +# ---------------------------------------------------------------- config +class Cfg: + def __init__(self, path: Path): + cp = configparser.ConfigParser() + # The file is optional: in a container every value can come from the + # environment, so credentials never have to be baked into an image. + found = cp.read(path) if path else [] + if path and not found and not os.environ.get("WLC_HOST"): + sys.exit(f"error: cannot read config {path} and WLC_HOST is unset") + for section in ("wlc", "mail", "monitor"): + if not cp.has_section(section): + cp.add_section(section) + w, m, g = cp["wlc"], cp["mail"], cp["monitor"] + + def s(env, sect, key, default=None): + """Environment wins over the file, file wins over the default.""" + v = os.environ.get(env) + if v is not None and v != "": + return v + return sect.get(key, fallback=default) + + def i(env, sect, key, default): + v = s(env, sect, key) + if v in (None, ""): + return default + try: + return int(v) + except ValueError: + sys.exit(f"error: {env or key} must be an integer, got {v!r}") + + def secret(env, sect, key): + """Read a secret from ENV, ENV_FILE, or the config file.""" + direct = os.environ.get(env) + file_name = os.environ.get(f"{env}_FILE") + if direct and file_name: + sys.exit(f"error: set only one of {env} and {env}_FILE") + if file_name: + try: + return Path(file_name).read_text().strip() + except OSError as e: + sys.exit(f"error: cannot read {env}_FILE {file_name!r}: {e}") + if direct: + return direct + return sect.get(key, fallback=None) + + self.host = s("WLC_HOST", w, "host") + self.port = i("WLC_PORT", w, "port", 161) + self.user = s("WLC_SNMP_USER", w, "snmp_user") + self.auth = secret("WLC_SNMP_AUTH", w, "snmp_auth") + self.priv = secret("WLC_SNMP_PRIV", w, "snmp_priv") + self.auth_protocol = ( + s("WLC_SNMP_AUTH_PROTOCOL", w, "auth_protocol", "sha") + or "sha" + ).lower() + self.timeout = i("WLC_TIMEOUT", w, "timeout", 5) + self.retries = i("WLC_RETRIES", w, "retries", 1) + + self.smtp_host = s("MAIL_SMTP_HOST", m, "smtp_host") + self.smtp_port = i("MAIL_SMTP_PORT", m, "smtp_port", 25) + self.smtp_security = ( + s("MAIL_SMTP_SECURITY", m, "smtp_security", "none") or "none" + ).lower() + self.smtp_user = s("MAIL_SMTP_USER", m, "smtp_user") + self.smtp_password = secret("MAIL_SMTP_PASSWORD", m, "smtp_password") + self.mail_from = s("MAIL_FROM", m, "from") + to = s("MAIL_TO", m, "to") or "" + self.mail_to = [a.strip() for a in to.split(",") if a.strip()] + self.subject_prefix = s("MAIL_SUBJECT_PREFIX", m, "subject_prefix", "[WLC]") + + self.state_file = Path( + s("MONITOR_STATE_FILE", g, "state_file", "/data/state.json")).expanduser() + self.fail_threshold = i("MONITOR_FAIL_THRESHOLD", g, "fail_threshold", 2) + self.unsaved_minutes = i( + "MONITOR_UNSAVED_MINUTES", g, "unsaved_minutes", 60) + self.unsaved_ticks = self.unsaved_minutes * 60 * 100 + self.interval = i("MONITOR_INTERVAL", g, "interval", 300) + + missing = [n for n, v in ( + ("host", self.host), ("snmp_user", self.user), + ("snmp_auth", self.auth), ("snmp_priv", self.priv), + ("smtp_host", self.smtp_host), ("mail_from", self.mail_from) + ) if not v] + if missing: + sys.exit(f"error: missing required settings: {', '.join(missing)}") + if not self.mail_to: + sys.exit("error: no recipients configured (MAIL_TO / [mail] to)") + if self.auth_protocol not in AUTH_PROTOCOLS: + choices = ", ".join(sorted(AUTH_PROTOCOLS)) + sys.exit(f"error: unsupported SNMP auth protocol " + f"{self.auth_protocol!r}; choose {choices}") + if self.smtp_security not in {"none", "starttls", "ssl"}: + sys.exit("error: MAIL_SMTP_SECURITY must be none, starttls, or ssl") + if bool(self.smtp_user) != bool(self.smtp_password): + sys.exit("error: SMTP username and password must be set together") + if self.smtp_user and self.smtp_security == "none": + sys.exit("error: SMTP authentication requires starttls or ssl") + + for name, value, minimum, maximum in ( + ("WLC_PORT", self.port, 1, 65535), + ("MAIL_SMTP_PORT", self.smtp_port, 1, 65535), + ("WLC_TIMEOUT", self.timeout, 1, None), + ("WLC_RETRIES", self.retries, 0, None), + ("MONITOR_FAIL_THRESHOLD", self.fail_threshold, 1, None), + ("MONITOR_UNSAVED_MINUTES", self.unsaved_minutes, 0, None), + ("MONITOR_INTERVAL", self.interval, 1, None), + ): + if value < minimum or (maximum is not None and value > maximum): + expected = f"{minimum}..{maximum}" if maximum else f">= {minimum}" + sys.exit(f"error: {name} must be {expected}, got {value}") + + if found and path and path.stat().st_mode & 0o007: + print(f"warning: {path} is world-accessible; it holds SNMP " + f"credentials. Use mode 600, or 640 with a dedicated group.", + file=sys.stderr) + + +def load_state(p: Path) -> dict: + try: + state = json.loads(p.read_text()) + except (FileNotFoundError, json.JSONDecodeError): + return {} + return state if isinstance(state, dict) else {} + + +def save_state(p: Path, state: dict) -> None: + p.parent.mkdir(mode=0o700, parents=True, exist_ok=True) + tmp = p.with_suffix(p.suffix + ".tmp") + tmp.write_text(json.dumps(state, indent=2)) + tmp.chmod(0o600) + tmp.replace(p) # atomic; a killed run can't leave half a state file + + +# ---------------------------------------------------------------- loop +def run_loop(cfg, a) -> int: + """Poll forever. Used when running as a container instead of a timer.""" + stop = threading.Event() + + def handle(signum, _frame): + # Docker sends SIGTERM and waits ~10s before SIGKILL. Exit promptly + # and cleanly so a stop/restart never truncates the state file. + print(f"[{signal.Signals(signum).name}] shutting down", flush=True) + stop.set() + + signal.signal(signal.SIGTERM, handle) + signal.signal(signal.SIGINT, handle) + + print(f"polling {cfg.host} every {cfg.interval}s; " + f"state={cfg.state_file}; alerts -> {', '.join(cfg.mail_to)}", + flush=True) + + while not stop.is_set(): + started = datetime.now(timezone.utc) + try: + run_once(cfg, a) + except Exception as e: + # Never let one bad iteration kill the monitor; a crashed monitor + # is silent, and silence is the failure mode we are avoiding. + print(f"poll iteration failed: {type(e).__name__}: {e}", + file=sys.stderr, flush=True) + # Subtract the work time so the cadence stays honest under slow polls. + elapsed = (datetime.now(timezone.utc) - started).total_seconds() + stop.wait(max(1.0, cfg.interval - elapsed)) + return 0 + + +# ---------------------------------------------------------------- main +def main(argv=None) -> int: + ap = argparse.ArgumentParser( + description="SNMP health check + email alerts for a Catalyst 9800.") + ap.add_argument("--config", default=None, + help="path to config.ini. Optional: environment variables " + "can supply every setting (see README). Defaults to " + "config.ini beside this script if it exists.") + ap.add_argument("--dry-run", action="store_true", + help="poll and print alerts, send no mail, still save state") + ap.add_argument("--no-save", action="store_true", help="do not update state file") + ap.add_argument("--test-email", action="store_true", help="send one test message and exit") + ap.add_argument("--show", action="store_true", help="print the current poll as JSON and exit") + ap.add_argument("--loop", action="store_true", + help="poll forever every --interval seconds (container mode)") + ap.add_argument("--healthcheck", action="store_true", + help="exit 0 if the last poll is recent, 1 otherwise") + ap.add_argument("--interval", type=int, default=None, + help="seconds between polls in --loop mode " + "(default: MONITOR_INTERVAL or 300)") + a = ap.parse_args(argv) + + if a.config: + cfg_path = Path(a.config) + else: + beside = Path(__file__).with_name("config.ini") + cfg_path = beside if beside.exists() else None + cfg = Cfg(cfg_path) + if a.interval is not None: + if a.interval < 1: + ap.error("--interval must be at least 1 second") + cfg.interval = a.interval + + if a.healthcheck: + # Deliberately reports on the LOOP, not on the controller. An + # unreachable WLC is a working monitor doing its job -- restarting the + # container for that would throw away the state that suppresses + # duplicate alerts. Only a stalled poll loop is unhealthy. + st = load_state(cfg.state_file) + age = elapsed_since(st) + if age is None: + print("no state yet") + return 1 + limit = cfg.interval * 3 + 60 + print(f"last poll {int(age)}s ago (limit {limit}s)") + return 0 if age <= limit else 1 + + if a.test_email: + try: + send_email(cfg, f"{cfg.subject_prefix} test message", + "This is a test from wlc_monitor.py.\n" + f"Sent via {cfg.smtp_host}:{cfg.smtp_port} from " + f"{socket.gethostname()}.\n\n" + "If you are reading this, the alert path works.") + except Exception as e: + print(f"FAILED to send: {type(e).__name__}: {e}", file=sys.stderr) + return 1 + print(f"test message sent to {', '.join(cfg.mail_to)}") + return 0 + + if a.loop: + return run_loop(cfg, a) + return run_once(cfg, a) + + +def run_once(cfg, a) -> int: + prev = load_state(cfg.state_file) + pending = [ + item for item in prev.get("_pending_emails", []) + if isinstance(item, dict) and {"subject", "body"} <= item.keys() + ] + poller = Poller(cfg.host, cfg.port, cfg.user, cfg.auth, cfg.priv, + cfg.auth_protocol, cfg.timeout, cfg.retries) + try: + cur = asyncio.run(poller.poll()) + cur["consecutive_failures"] = 0 + except Exception as e: + # Carry the last known good readings forward. Without this, a failed + # poll erases the baseline and the next success looks like a change. + cur = {"ok": False, "error": f"{type(e).__name__}: {e}", + "consecutive_failures": prev.get("consecutive_failures", 0) + 1, + "sys_name": prev.get("sys_name", cfg.host), + "aps": prev.get("aps", {})} + for k in ("sys_uptime", "sys_descr", "run_last_changed", + "run_last_saved", "_unsaved_alerted"): + if k in prev: + cur[k] = prev[k] + + if a.show: + print(json.dumps(cur, indent=2, default=str)) + return 0 if cur["ok"] else 1 + + alerts = build_alerts(prev, cur, cfg) + messages = pending + [ + { + "subject": f"{cfg.subject_prefix} {severity}: {title}", + "body": format_body(detail, cur, cfg), + } + for severity, title, detail in alerts + ] + + rc = 0 + remaining = list(pending) if a.dry_run else [] + for index, message in enumerate(messages): + subject, body = message["subject"], message["body"] + if a.dry_run: + print(f"\n=== would send: {subject} ===\n{body}") + else: + try: + send_email(cfg, subject, body) + print(f"sent: {subject}") + except Exception as e: + print(f"FAILED to send {subject!r}: {type(e).__name__}: {e}", + file=sys.stderr) + # Preserve FIFO ordering: a later recovery must never arrive + # before the outage message whose delivery just failed. + remaining.extend(messages[index:]) + rc = 1 + break + + if not messages: + state = "ok" if cur["ok"] else f"UNREACHABLE (x{cur['consecutive_failures']})" + print(f"no change; {state}; " + f"APs joined: {len(cur.get('aps', {}))}") + + # Advance the comparison baseline even when delivery fails, but retain a + # small durable outbox so transition alerts are retried on the next poll. + if remaining: + cur["_pending_emails"] = remaining + else: + cur.pop("_pending_emails", None) + if not a.no_save: + cur["last_poll"] = datetime.now(timezone.utc).isoformat() + save_state(cfg.state_file, cur) + return rc + + +if __name__ == "__main__": + raise SystemExit(main())