From dbc7464b588e6b612ba06e74d7d98267ab524531 Mon Sep 17 00:00:00 2001 From: Johan Lundberg Date: Thu, 30 Jul 2026 23:22:10 +0200 Subject: [PATCH] Prepare project for public release --- .dockerignore | 26 + .env.example | 36 ++ .github/ISSUE_TEMPLATE/bug_report.yml | 67 +++ .github/dependabot.yml | 14 + .github/pull_request_template.md | 18 + .github/workflows/ci.yml | 35 ++ .gitignore | 27 ++ CONTRIBUTING.md | 54 +++ Dockerfile | 38 ++ LICENSE | 21 + README.md | 237 +++++++++ SECURITY.md | 22 + config.ini.example | 34 ++ docker-compose.yml | 36 ++ pyproject.toml | 20 + requirements.txt | 5 + tests/__init__.py | 1 + tests/test_wlc_monitor.py | 374 ++++++++++++++ wlc-monitor.service | 34 ++ wlc-monitor.timer | 12 + wlc_monitor.py | 669 ++++++++++++++++++++++++++ 21 files changed, 1780 insertions(+) create mode 100644 .dockerignore create mode 100644 .env.example create mode 100644 .github/ISSUE_TEMPLATE/bug_report.yml create mode 100644 .github/dependabot.yml create mode 100644 .github/pull_request_template.md create mode 100644 .github/workflows/ci.yml create mode 100644 .gitignore create mode 100644 CONTRIBUTING.md create mode 100644 Dockerfile create mode 100644 LICENSE create mode 100644 README.md create mode 100644 SECURITY.md create mode 100644 config.ini.example create mode 100644 docker-compose.yml create mode 100644 pyproject.toml create mode 100644 requirements.txt create mode 100644 tests/__init__.py create mode 100644 tests/test_wlc_monitor.py create mode 100644 wlc-monitor.service create mode 100644 wlc-monitor.timer create mode 100755 wlc_monitor.py diff --git a/.dockerignore b/.dockerignore new file mode 100644 index 0000000..1114078 --- /dev/null +++ b/.dockerignore @@ -0,0 +1,26 @@ +# Never let credentials into an image layer. config.ini holds the SNMP auth +# and priv passphrases; supply them at runtime via env instead. +config.ini +.env + +# Local runtime artefacts. +.git/ +.github/ +.agents/ +.codex/ +.venv/ +venv/ +tests/ +state.json +state.json.tmp +__pycache__/ +*.pyc +.coverage +htmlcov/ + +# Host-specific deployment files, irrelevant inside the container. +wlc-monitor.service +wlc-monitor.timer +README.md +CONTRIBUTING.md +SECURITY.md diff --git a/.env.example b/.env.example new file mode 100644 index 0000000..c43d7cc --- /dev/null +++ b/.env.example @@ -0,0 +1,36 @@ +# Copy to .env and fill in. Contains SNMP credentials -- chmod 600, never commit. +# cp .env.example .env && chmod 600 .env + +# --- controller --- +WLC_HOST=192.0.2.10 +WLC_PORT=161 +WLC_SNMP_USER=wlcmon +WLC_SNMP_AUTH= +WLC_SNMP_PRIV= +# sha256 is preferred when the controller supports it; use sha for older setups. +WLC_SNMP_AUTH_PROTOCOL=sha256 +WLC_TIMEOUT=5 +WLC_RETRIES=1 + +# --- mail --- +MAIL_SMTP_HOST=smtp.example.com +MAIL_SMTP_PORT=587 +MAIL_SMTP_SECURITY=starttls +MAIL_SMTP_USER= +MAIL_SMTP_PASSWORD= +MAIL_FROM=wlc-monitor@example.com +# Comma-separated for multiple recipients. +MAIL_TO=network-ops@example.com +MAIL_SUBJECT_PREFIX=[WLC] + +# --- monitor --- +TZ=UTC +# Seconds between polls. This is also your detection resolution: an AP that +# drops and recovers inside one interval is never seen. +MONITOR_INTERVAL=300 +# Consecutive failed polls before declaring the controller unreachable. +# 2 x 300s means roughly 10 minutes of silence before alerting, which rides +# out a single dropped UDP packet without crying wolf. +MONITOR_FAIL_THRESHOLD=2 +# Alert if running-config stays unsaved this many minutes. +MONITOR_UNSAVED_MINUTES=60 diff --git a/.github/ISSUE_TEMPLATE/bug_report.yml b/.github/ISSUE_TEMPLATE/bug_report.yml new file mode 100644 index 0000000..449c019 --- /dev/null +++ b/.github/ISSUE_TEMPLATE/bug_report.yml @@ -0,0 +1,67 @@ +name: Bug report +description: Report reproducible incorrect behavior +title: "[Bug]: " +labels: ["bug"] +body: + - type: markdown + attributes: + value: | + Do not include credentials, private addresses, AP names, `config.ini`, + `.env`, `state.json`, or unredacted logs. Report security issues using + the private process in SECURITY.md. + - type: input + id: version + attributes: + label: Version or commit + placeholder: v0.1.0 or commit SHA + validations: + required: true + - type: dropdown + id: installation + attributes: + label: Installation mode + options: + - Docker Compose + - systemd + - Manual Python invocation + validations: + required: true + - type: input + id: runtime + attributes: + label: Runtime + description: Python version, or Docker Engine and Compose versions + validations: + required: true + - type: input + id: controller + attributes: + label: Controller model and IOS XE version + description: Do not include hostnames, addresses, serial numbers, or AP names. + validations: + required: true + - type: textarea + id: steps + attributes: + label: Steps to reproduce + validations: + required: true + - type: textarea + id: expected + attributes: + label: Expected behavior + validations: + required: true + - type: textarea + id: actual + attributes: + label: Actual behavior and redacted logs + validations: + required: true + - type: checkboxes + id: privacy + attributes: + label: Data-safety confirmation + options: + - label: I removed credentials, private infrastructure details, and personal data. + required: true diff --git a/.github/dependabot.yml b/.github/dependabot.yml new file mode 100644 index 0000000..8ba1e84 --- /dev/null +++ b/.github/dependabot.yml @@ -0,0 +1,14 @@ +version: 2 +updates: + - package-ecosystem: pip + directory: / + schedule: + interval: monthly + - package-ecosystem: docker + directory: / + schedule: + interval: monthly + - package-ecosystem: github-actions + directory: / + schedule: + interval: monthly diff --git a/.github/pull_request_template.md b/.github/pull_request_template.md new file mode 100644 index 0000000..0a8f4f2 --- /dev/null +++ b/.github/pull_request_template.md @@ -0,0 +1,18 @@ +## Summary + +Describe the problem and the focused solution. + +## Verification + +- [ ] `python -m py_compile wlc_monitor.py` +- [ ] `python -m unittest discover -v` +- [ ] `python -m pip check` +- [ ] Docker checks run, or not applicable +- [ ] User-facing documentation updated, or not applicable +- [ ] No credentials, private infrastructure data, runtime state, or personal data + are present in the diff + +## Operational impact + +Call out compatibility, configuration, alerting, state-migration, or deployment +effects. diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml new file mode 100644 index 0000000..e2b0c26 --- /dev/null +++ b/.github/workflows/ci.yml @@ -0,0 +1,35 @@ +name: CI + +on: + push: + pull_request: + +permissions: + contents: read + +jobs: + test: + runs-on: ubuntu-latest + strategy: + fail-fast: false + matrix: + python-version: ["3.10", "3.12", "3.13"] + steps: + - uses: actions/checkout@v4 + - uses: actions/setup-python@v5 + with: + python-version: ${{ matrix.python-version }} + cache: pip + - run: python -m pip install --upgrade pip + - run: python -m pip install -r requirements.txt + - run: python -m py_compile wlc_monitor.py + - run: python -m unittest discover -v + - run: python -m pip check + + container: + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@v4 + - run: cp .env.example .env + - run: docker compose config --quiet + - run: docker build --tag wlc-monitor:test . diff --git a/.gitignore b/.gitignore new file mode 100644 index 0000000..136bfc6 --- /dev/null +++ b/.gitignore @@ -0,0 +1,27 @@ +# Local configuration and credentials +.env +.env.* +!.env.example +config.ini + +# Runtime state and logs +state.json +state.json.tmp +*.log + +# Python environments and generated files +.venv/ +venv/ +__pycache__/ +*.py[cod] +.pytest_cache/ +.ruff_cache/ +.mypy_cache/ +.coverage +htmlcov/ + +# Editor and operating-system metadata +.DS_Store +.idea/ +.vscode/ +*.swp diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md new file mode 100644 index 0000000..fa761c3 --- /dev/null +++ b/CONTRIBUTING.md @@ -0,0 +1,54 @@ +# Contributing + +Thanks for helping improve WLC Monitor. + +## Before you start + +Use GitHub Issues for reproducible bugs and focused feature proposals. Search +existing issues first. Use the private process in [SECURITY.md](SECURITY.md) +for vulnerabilities. + +## Development setup + +```bash +python3 -m venv .venv +. .venv/bin/activate +python -m pip install -r requirements.txt +cp config.ini.example config.ini +chmod 600 config.ini +``` + +Never commit real credentials or infrastructure data. Use reserved example +addresses/domains and synthetic state fixtures. + +## Making changes + +- Create a focused branch. +- Add or update tests for behavior changes. +- Update the README and configuration examples for user-visible changes. +- Keep unrelated formatting out of the patch. + +Run the local checks before opening a pull request: + +```bash +python -m py_compile wlc_monitor.py +python -m unittest discover -v +python -m pip check +``` + +If Docker is available, also run: + +```bash +cp .env.example .env +docker compose config --quiet +docker build -t wlc-monitor:test . +``` + +Remove the local `.env` afterward or keep it private; it is ignored by Git. + +## Pull requests + +Explain the problem and solution, link related issues, list the checks you ran, +and call out compatibility or operational impact. Confirm that the diff +contains no credentials, `.env`, `config.ini`, `state.json`, controller/AP +identifiers, or personal data. diff --git a/Dockerfile b/Dockerfile new file mode 100644 index 0000000..3821a53 --- /dev/null +++ b/Dockerfile @@ -0,0 +1,38 @@ +# syntax=docker/dockerfile:1 +FROM python:3.12-slim + +# Unbuffered so `docker logs` shows poll output as it happens rather than in +# 4KB bursts -- stdout is block-buffered when it is not a TTY. +ENV PYTHONUNBUFFERED=1 \ + PYTHONDONTWRITEBYTECODE=1 + +WORKDIR /app + +# Dependencies first so edits to the script do not invalidate the layer. +# cryptography is required, not optional: without it pysnmp cannot do AES +# privacy and every poll fails with "Ciphering services not available". +COPY requirements.txt . +RUN pip install --no-cache-dir -r requirements.txt + +COPY wlc_monitor.py . + +# Run unprivileged. The image needs no root at runtime: outbound UDP/161 and +# TCP/SMTP only, plus one JSON file in /data. +RUN useradd --system --uid 10001 --no-create-home --shell /usr/sbin/nologin wlcmon \ + && mkdir -p /data \ + && chown wlcmon:wlcmon /data +USER wlcmon + +# State lives here; mount a named volume so alert de-duplication and the +# "consecutive failures" counter survive a restart. Without persistence the +# container starts with a blank comparison baseline after recreation. +VOLUME ["/data"] +ENV MONITOR_STATE_FILE=/data/state.json + +# Reports on the poll loop, not on the controller: an unreachable WLC means the +# monitor is working, and restarting it would discard alert-suppression state. +HEALTHCHECK --interval=60s --timeout=10s --start-period=90s --retries=3 \ + CMD ["python", "/app/wlc_monitor.py", "--healthcheck"] + +ENTRYPOINT ["python", "/app/wlc_monitor.py"] +CMD ["--loop"] diff --git a/LICENSE b/LICENSE new file mode 100644 index 0000000..92a0c59 --- /dev/null +++ b/LICENSE @@ -0,0 +1,21 @@ +MIT License + +Copyright (c) 2026 WLC Monitor contributors + +Permission is hereby granted, free of charge, to any person obtaining a copy +of this software and associated documentation files (the "Software"), to deal +in the Software without restriction, including without limitation the rights +to use, copy, modify, merge, publish, distribute, sublicense, and/or sell +copies of the Software, and to permit persons to whom the Software is +furnished to do so, subject to the following conditions: + +The above copyright notice and this permission notice shall be included in all +copies or substantial portions of the Software. + +THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +SOFTWARE. diff --git a/README.md b/README.md new file mode 100644 index 0000000..f1949c8 --- /dev/null +++ b/README.md @@ -0,0 +1,237 @@ +# WLC Monitor + +WLC Monitor polls a Cisco Catalyst 9800 wireless LAN controller over SNMPv3 +and sends email when controller, access-point, or configuration state changes. +It can run continuously in Docker or as a one-shot systemd timer. + +Polling makes silence observable: a failed poll is itself a signal, whereas +traps and syslog stop when a controller becomes unreachable. + +## Alerts + +| Severity | Condition | Detection | +|---|---|---| +| CRITICAL | Controller unreachable | Configured number of consecutive failed polls | +| CRITICAL | Controller rebooted | `sysUpTime` reset | +| CRITICAL | AP down | AP disappears from `cLApTable` | +| WARNING | AP rebooted | AP uptime reset | +| WARNING | AP rejoined | Association uptime reset while AP uptime did not | +| NOTICE | Running configuration changed | `ccmHistoryRunningLastChanged` advanced | +| NOTICE | Configuration remains unsaved | Running change is newer than the saved change beyond the threshold | +| RECOVERY | Controller or AP returns | Transition back to the available state | + +TimeTicks wrap at about 497 days. WLC Monitor compares the apparent counter +movement with elapsed wall time so a normal wrap is not reported as a reboot. + +## Requirements and compatibility + +- Docker Engine with Compose v2; or Python 3.10 or newer. +- A Cisco Catalyst 9800 reachable over SNMPv3 authPriv. +- Access to the CISCO-LWAPP-AP-MIB and CISCO-CONFIG-MAN-MIB objects used by the + controller. +- A reachable SMTP server. + +The implementation was developed against a Catalyst 9800-CL running IOS XE +17.12. Other 9800 models and releases may expose different MIB behavior; +compatibility reports are welcome. + +SMTP supports STARTTLS, implicit TLS, optional username/password +authentication, and plaintext delivery to a trusted relay. Do not select +`none` across an untrusted network: alert bodies contain infrastructure names +and operational state. Authentication is rejected in plaintext mode so SMTP +credentials cannot be sent over an unencrypted connection. + +## Quick start with Docker + +```bash +cp .env.example .env +chmod 600 .env +# Edit .env and set the controller, SNMP credentials, mail server, +# sender, and recipients. +docker compose up -d --build +``` + +Verify the controller and mail paths: + +```bash +docker compose exec wlc-monitor python /app/wlc_monitor.py --show +docker compose exec wlc-monitor python /app/wlc_monitor.py --test-email +docker compose logs -f wlc-monitor +docker inspect --format '{{.State.Health.Status}}' wlc-monitor +``` + +The container runs unprivileged with a read-only root filesystem. Its named +`/data` volume holds comparison state, failure counts, and any email awaiting +retry. Preserve this volume across upgrades. Losing it creates a blank +baseline: the next successful poll is deliberately silent, and changes during +the gap cannot be reconstructed. + +The health check reports whether the poll loop is progressing, not whether the +controller is reachable. Controller failure is a condition the monitor is +expected to observe, not a reason to restart it. + +## Configuration + +Environment variables override values in `config.ini`. In containers, prefer +environment variables or mounted secret files. For a host install, begin with +`config.ini.example`. + +| Variable | Required/default | Meaning | +|---|---|---| +| `WLC_HOST` | required | Controller hostname or address | +| `WLC_PORT` | `161` | SNMP port | +| `WLC_SNMP_USER` | required | SNMPv3 username | +| `WLC_SNMP_AUTH` | required | Authentication passphrase | +| `WLC_SNMP_PRIV` | required | AES privacy passphrase | +| `WLC_SNMP_AUTH_PROTOCOL` | `sha` | `sha`, `sha224`, `sha256`, `sha384`, or `sha512` | +| `WLC_TIMEOUT` | `5` | Per-request timeout in seconds | +| `WLC_RETRIES` | `1` | SNMP retries | +| `MAIL_SMTP_HOST` | required | SMTP server | +| `MAIL_SMTP_PORT` | `25` | SMTP port | +| `MAIL_SMTP_SECURITY` | `none` | `none`, `starttls`, or `ssl` | +| `MAIL_SMTP_USER` | optional | SMTP username; password must also be set | +| `MAIL_SMTP_PASSWORD` | optional | SMTP password; username must also be set | +| `MAIL_FROM` | required | Envelope/header sender | +| `MAIL_TO` | required | Comma-separated recipients | +| `MAIL_SUBJECT_PREFIX` | `[WLC]` | Subject prefix | +| `MONITOR_INTERVAL` | `300` | Seconds between continuous polls | +| `MONITOR_FAIL_THRESHOLD` | `2` | Consecutive failures before alerting | +| `MONITOR_UNSAVED_MINUTES` | `60` | Time before unsaved configuration alerts | +| `MONITOR_STATE_FILE` | `/data/state.json` | Persistent state path | +| `TZ` | `UTC` in Compose | Time zone used in alert timestamps | + +For secrets, `WLC_SNMP_AUTH_FILE`, `WLC_SNMP_PRIV_FILE`, and +`MAIL_SMTP_PASSWORD_FILE` may point to mounted files instead of placing values +directly in the process environment. Set either the direct variable or its +`_FILE` form, never both. + +The SNMP authentication protocol must match the controller user. SHA-2 is +preferred where the IOS XE release supports it; `sha` remains available for +older configurations. + +## Install with systemd + +Create a dedicated account and install the files: + +```bash +sudo useradd --system --no-create-home --shell /usr/sbin/nologin wlcmon +sudo install -d -o root -g wlcmon -m 0750 /opt/wlc-monitor +sudo install -o root -g wlcmon -m 0755 wlc_monitor.py /opt/wlc-monitor/ +sudo install -o root -g wlcmon -m 0644 requirements.txt README.md /opt/wlc-monitor/ +sudo cp config.ini.example /opt/wlc-monitor/config.ini +sudo chown root:wlcmon /opt/wlc-monitor/config.ini +sudo chmod 0640 /opt/wlc-monitor/config.ini +sudo python3 -m venv /opt/wlc-monitor/venv +sudo /opt/wlc-monitor/venv/bin/pip install -r /opt/wlc-monitor/requirements.txt +``` + +Edit `/opt/wlc-monitor/config.ini`, keeping +`state_file = /var/lib/wlc-monitor/state.json`, then install and start the +timer: + +```bash +sudo install -m 0644 wlc-monitor.service wlc-monitor.timer /etc/systemd/system/ +sudo systemctl daemon-reload +sudo systemctl enable --now wlc-monitor.timer +``` + +Verify: + +```bash +sudo systemctl start wlc-monitor.service +journalctl -u wlc-monitor.service -n 20 +systemctl list-timers wlc-monitor.timer +``` + +The unit creates `/var/lib/wlc-monitor` privately and applies a restrictive +umask. The configuration is readable only by root and the dedicated service +group. + +## Local command-line use + +```bash +python3 -m venv .venv +. .venv/bin/activate +python -m pip install -r requirements.txt +cp config.ini.example config.ini +chmod 600 config.ini +# Edit config.ini and use a writable state path such as ./state.json. + +python wlc_monitor.py --show +python wlc_monitor.py --dry-run --no-save +python wlc_monitor.py --test-email +python wlc_monitor.py --loop --interval 60 +python wlc_monitor.py --healthcheck +``` + +`--dry-run` prints alerts instead of sending them. It still updates state +unless combined with `--no-save`. `--show` returns a nonzero exit status when +the poll fails, making it suitable for scripts. + +The first successful poll is deliberately silent because there is no previous +state to compare. + +## Controller configuration + +Use a read-only SNMPv3 authPriv user and restrict its ACL to the single +monitoring host. Exact SHA-2 syntax varies by IOS XE release; consult the +controller documentation and make `WLC_SNMP_AUTH_PROTOCOL` match. + +```text +ip access-list standard SNMP-MON + permit host +snmp-server group WLCMON v3 priv read v1default access SNMP-MON +snmp-server user wlcmon WLCMON v3 auth \ + priv aes 128 +``` + +Do not paste real controller configuration, credentials, addresses, or device +output into issues. + +## Delivery and state behavior + +State is written atomically with mode `0600`. It can still contain controller +and AP names, software details, and queued alert bodies, so treat it as +operationally sensitive and do not commit or publish it. + +Transition alerts that fail SMTP delivery are stored in a small durable FIFO +outbox and retried on the next poll. Delivery stops at the first failure so a +recovery cannot arrive before its outage. A continuously unsaved configuration +alerts once when it first exceeds the threshold, then becomes eligible again +after it is saved. A crash between the SMTP server accepting a message and the +state write can still cause a duplicate; recipients should tolerate +at-least-once delivery. + +## Operational limitations + +- SNMP reveals that a configuration changed, not what changed. Inspect the + controller's configuration history for the actual diff. +- The poll interval is the detection resolution. An AP that disconnects and + recovers entirely between polls is invisible. +- AP renames look like a down and join pair because APs are keyed by name. +- The monitoring host remains a single point of failure. Use an external + dead-man/heartbeat service if silent monitor failure must be detected. +- A missing or partial AP table is treated as a failed poll to avoid false + reboot/rejoin alerts. + +## Development + +Run the same checks used by CI: + +```bash +python -m py_compile wlc_monitor.py +python -m unittest discover -v +python -m pip check +docker compose config --quiet +docker build -t wlc-monitor:test . +``` + +## Contributing and security + +Bug reports and pull requests are welcome; see +[CONTRIBUTING.md](CONTRIBUTING.md). Report vulnerabilities privately as +described in [SECURITY.md](SECURITY.md), not in a public issue. + +## License + +WLC Monitor is available under the [MIT License](LICENSE). diff --git a/SECURITY.md b/SECURITY.md new file mode 100644 index 0000000..c2327e9 --- /dev/null +++ b/SECURITY.md @@ -0,0 +1,22 @@ +# Security Policy + +## Supported versions + +Until tagged releases are published, only the current default branch receives +security fixes. Supported release versions will be listed here once releases +begin. + +## Reporting a vulnerability + +Do not open a public issue for a suspected vulnerability. Prefer GitHub's +private vulnerability reporting from the repository's Security tab. Include +the affected commit or version, impact, reproduction steps, and any suggested +mitigation. + +If private reporting is not enabled, ask the maintainer for a private reporting +channel without disclosing vulnerability details publicly. + +Do not include real SNMP credentials, `.env`, `config.ini`, `state.json`, +controller addresses, AP names, or unredacted logs. Reports will be +acknowledged and updated on a best-effort basis; please allow time for a fix +before public disclosure. diff --git a/config.ini.example b/config.ini.example new file mode 100644 index 0000000..ea5366b --- /dev/null +++ b/config.ini.example @@ -0,0 +1,34 @@ +# Copy to config.ini, fill in the required values, and keep it private: +# cp config.ini.example config.ini +# chmod 600 config.ini + +[wlc] +host = 192.0.2.10 +port = 161 +snmp_user = wlcmon +snmp_auth = +snmp_priv = +# Prefer sha256. Use sha only when required by an older controller. +auth_protocol = sha256 +timeout = 5 +retries = 1 + +[mail] +smtp_host = smtp.example.com +smtp_port = 587 +# One of: none, starttls, ssl +smtp_security = starttls +# Leave both blank for an unauthenticated trusted relay. +smtp_user = +smtp_password = +from = wlc-monitor@example.com +to = network-ops@example.com +subject_prefix = [WLC] + +[monitor] +# This path matches wlc-monitor.service. For a user-run checkout, change it to +# a private writable path such as ./state.json. +state_file = /var/lib/wlc-monitor/state.json +fail_threshold = 2 +unsaved_minutes = 60 +interval = 300 diff --git a/docker-compose.yml b/docker-compose.yml new file mode 100644 index 0000000..cae0187 --- /dev/null +++ b/docker-compose.yml @@ -0,0 +1,36 @@ +services: + wlc-monitor: + build: . + image: wlc-monitor:latest + container_name: wlc-monitor + restart: unless-stopped + + # Credentials come from .env (gitignored, chmod 600), never from the image. + env_file: [.env] + + environment: + MONITOR_STATE_FILE: /data/state.json + # Timestamps in alert bodies use the container's zone; without this they + # render as UTC and will not match the controller's log timestamps. + TZ: ${TZ:-UTC} + + volumes: + # Named volume: alert de-duplication and the consecutive-failure counter + # live here. Lose it and the container re-alerts from a blank slate. + - wlc-monitor-state:/data + + # Read-only rootfs; the only thing that needs writing is /data. + read_only: true + tmpfs: + - /tmp + security_opt: + - no-new-privileges:true + cap_drop: + - ALL + + logging: + driver: json-file + options: {max-size: "10m", max-file: "3"} + +volumes: + wlc-monitor-state: diff --git a/pyproject.toml b/pyproject.toml new file mode 100644 index 0000000..3a27d1b --- /dev/null +++ b/pyproject.toml @@ -0,0 +1,20 @@ +[build-system] +requires = ["setuptools>=77"] +build-backend = "setuptools.build_meta" + +[project] +name = "wlc-monitor" +version = "0.1.0" +description = "SNMPv3 health monitoring and email alerts for Cisco Catalyst 9800 controllers" +readme = "README.md" +requires-python = ">=3.10" +dependencies = [ + "cryptography==49.0.0", + "pysnmp==7.1.27", +] + +[project.scripts] +wlc-monitor = "wlc_monitor:main" + +[tool.setuptools] +py-modules = ["wlc_monitor"] diff --git a/requirements.txt b/requirements.txt new file mode 100644 index 0000000..8419790 --- /dev/null +++ b/requirements.txt @@ -0,0 +1,5 @@ +cffi==2.1.0 +cryptography==49.0.0 +pyasn1==0.6.4 +pycparser==3.0 +pysnmp==7.1.27 diff --git a/tests/__init__.py b/tests/__init__.py new file mode 100644 index 0000000..c637051 --- /dev/null +++ b/tests/__init__.py @@ -0,0 +1 @@ +"""WLC Monitor test suite.""" diff --git a/tests/test_wlc_monitor.py b/tests/test_wlc_monitor.py new file mode 100644 index 0000000..0a6d37c --- /dev/null +++ b/tests/test_wlc_monitor.py @@ -0,0 +1,374 @@ +import json +import os +import stat +import tempfile +import unittest +from contextlib import redirect_stderr, redirect_stdout +from io import StringIO +from pathlib import Path +from types import SimpleNamespace +from unittest.mock import ANY, AsyncMock, MagicMock, call, patch + +import wlc_monitor + + +def sample_state(**changes): + state = { + "ok": True, + "sys_name": "test-wlc", + "sys_descr": "Synthetic controller", + "sys_uptime": 2_000_000, + "run_last_changed": 1_000_000, + "run_last_saved": 1_000_000, + "aps": {}, + "consecutive_failures": 0, + "last_poll": "2026-01-01T00:00:00+00:00", + } + state.update(changes) + return state + + +class AlertConfig: + fail_threshold = 2 + unsaved_ticks = 60 * 60 * 100 + host = "192.0.2.10" + + +class CounterTests(unittest.TestCase): + def test_increasing_counter_is_not_reset(self): + self.assertFalse(wlc_monitor.counter_reset(100, 200, 1)) + + def test_small_decrease_is_reset(self): + self.assertTrue(wlc_monitor.counter_reset(50_000, 100, 60)) + + def test_32_bit_wrap_is_not_reset(self): + self.assertFalse( + wlc_monitor.counter_reset(2**32 - 100, 200, elapsed_s=5)) + + +class AlertTests(unittest.TestCase): + def test_first_poll_is_silent(self): + self.assertEqual( + wlc_monitor.build_alerts({}, sample_state(), AlertConfig()), []) + + def test_unsaved_config_alerts_once_until_saved(self): + prev = sample_state( + run_last_changed=1_000_000, run_last_saved=900_000) + cur = sample_state( + sys_uptime=2_000_000, + run_last_changed=1_000_000, + run_last_saved=900_000, + ) + first = wlc_monitor.build_alerts(prev, cur, AlertConfig()) + self.assertEqual([a[1] for a in first], [ + "test-wlc has unsaved config"]) + self.assertTrue(cur["_unsaved_alerted"]) + + later = sample_state( + sys_uptime=2_030_000, + run_last_changed=1_000_000, + run_last_saved=900_000, + ) + self.assertEqual( + wlc_monitor.build_alerts(cur, later, AlertConfig()), []) + + saved = sample_state( + sys_uptime=2_060_000, + run_last_changed=1_000_000, + run_last_saved=1_000_000, + ) + wlc_monitor.build_alerts(later, saved, AlertConfig()) + self.assertFalse(saved["_unsaved_alerted"]) + + +class StateTests(unittest.TestCase): + def test_state_is_written_atomically_with_private_permissions(self): + with tempfile.TemporaryDirectory() as directory: + path = Path(directory) / "state.json" + wlc_monitor.save_state(path, {"ok": True}) + self.assertEqual(json.loads(path.read_text()), {"ok": True}) + self.assertEqual( + stat.S_IMODE(path.stat().st_mode), 0o600) + + def test_naive_or_invalid_timestamp_is_tolerated(self): + self.assertIsNone( + wlc_monitor.elapsed_since({"last_poll": "2026-01-01T00:00:00"})) + self.assertIsNone( + wlc_monitor.elapsed_since({"last_poll": "not-a-timestamp"})) + + def test_non_object_state_is_ignored(self): + with tempfile.TemporaryDirectory() as directory: + path = Path(directory) / "state.json" + path.write_text("[]") + self.assertEqual(wlc_monitor.load_state(path), {}) + + +class ConfigTests(unittest.TestCase): + required_env = { + "WLC_SNMP_AUTH": "synthetic-auth", + "WLC_SNMP_PRIV": "synthetic-privacy", + } + + def write_config(self, directory): + path = Path(directory) / "config.ini" + path.write_text( + "[wlc]\n" + "host = 192.0.2.10\n" + "snmp_user = test\n" + "snmp_auth =\n" + "snmp_priv =\n" + "[mail]\n" + "smtp_host = smtp.example.com\n" + "from = monitor@example.com\n" + "to = ops@example.com\n" + "[monitor]\n" + "state_file = state.json\n" + ) + path.chmod(0o600) + return path + + def test_environment_overrides_file(self): + with tempfile.TemporaryDirectory() as directory: + path = self.write_config(directory) + with patch.dict( + os.environ, + {**self.required_env, "WLC_HOST": "198.51.100.20"}, + clear=True, + ): + cfg = wlc_monitor.Cfg(path) + self.assertEqual(cfg.host, "198.51.100.20") + + def test_invalid_interval_is_rejected(self): + with tempfile.TemporaryDirectory() as directory: + path = self.write_config(directory) + with patch.dict( + os.environ, + {**self.required_env, "MONITOR_INTERVAL": "0"}, + clear=True, + ): + with self.assertRaises(SystemExit): + wlc_monitor.Cfg(path) + + def test_negative_unsaved_threshold_is_rejected(self): + with tempfile.TemporaryDirectory() as directory: + path = self.write_config(directory) + with patch.dict( + os.environ, + { + **self.required_env, + "MONITOR_UNSAVED_MINUTES": "-1", + }, + clear=True, + ): + with self.assertRaises(SystemExit): + wlc_monitor.Cfg(path) + + def test_smtp_authentication_requires_encryption(self): + with tempfile.TemporaryDirectory() as directory: + path = self.write_config(directory) + with patch.dict( + os.environ, + { + **self.required_env, + "MAIL_SMTP_SECURITY": "none", + "MAIL_SMTP_USER": "test-user", + "MAIL_SMTP_PASSWORD": "synthetic-password", + }, + clear=True, + ): + with self.assertRaises(SystemExit): + wlc_monitor.Cfg(path) + + def test_secret_can_be_read_from_file(self): + with tempfile.TemporaryDirectory() as directory: + path = self.write_config(directory) + secret = Path(directory) / "auth" + secret.write_text("from-file\n") + with patch.dict( + os.environ, + { + "WLC_SNMP_PRIV": self.required_env["WLC_SNMP_PRIV"], + "WLC_SNMP_AUTH_FILE": str(secret), + }, + clear=True, + ): + cfg = wlc_monitor.Cfg(path) + self.assertEqual(cfg.auth, "from-file") + + +class DeliveryTests(unittest.TestCase): + def test_failed_transition_email_is_retried(self): + with tempfile.TemporaryDirectory() as directory: + state_path = Path(directory) / "state.json" + previous = sample_state( + aps={ + "ap-one": { + "index": "1", + "ap_uptime": 1_000, + "assoc_uptime": 900, + } + } + ) + wlc_monitor.save_state(state_path, previous) + current = sample_state(sys_uptime=2_030_000) + + cfg = SimpleNamespace( + state_file=state_path, + host="192.0.2.10", + port=161, + user="test", + auth="auth", + priv="priv", + auth_protocol="sha256", + timeout=1, + retries=0, + fail_threshold=2, + unsaved_ticks=360_000, + mail_from="monitor@example.com", + mail_to=["ops@example.com"], + subject_prefix="[TEST]", + smtp_host="smtp.example.com", + smtp_port=587, + smtp_security="starttls", + smtp_user=None, + smtp_password=None, + ) + args = SimpleNamespace(show=False, dry_run=False, no_save=False) + + poller = SimpleNamespace(poll=AsyncMock(return_value=current)) + with patch.object(wlc_monitor, "Poller", return_value=poller), \ + patch.object( + wlc_monitor, "send_email", + side_effect=OSError("temporary failure"), + ), redirect_stdout(StringIO()), redirect_stderr(StringIO()): + self.assertEqual(wlc_monitor.run_once(cfg, args), 1) + + failed_state = json.loads(state_path.read_text()) + self.assertEqual(len(failed_state["_pending_emails"]), 1) + + poller.poll = AsyncMock(return_value=sample_state( + sys_uptime=2_060_000)) + with patch.object(wlc_monitor, "Poller", return_value=poller), \ + patch.object(wlc_monitor, "send_email") as send, \ + redirect_stdout(StringIO()): + self.assertEqual(wlc_monitor.run_once(cfg, args), 0) + send.assert_called_once() + + recovered_state = json.loads(state_path.read_text()) + self.assertNotIn("_pending_emails", recovered_state) + + def test_outbox_stops_after_first_failure_to_preserve_order(self): + with tempfile.TemporaryDirectory() as directory: + state_path = Path(directory) / "state.json" + previous = sample_state( + ok=False, + consecutive_failures=2, + _pending_emails=[ + {"subject": "OUTAGE", "body": "controller unavailable"} + ], + ) + wlc_monitor.save_state(state_path, previous) + + cfg = SimpleNamespace( + state_file=state_path, + host="192.0.2.10", + port=161, + user="test", + auth="auth", + priv="priv", + auth_protocol="sha256", + timeout=1, + retries=0, + fail_threshold=2, + unsaved_ticks=360_000, + mail_from="monitor@example.com", + mail_to=["ops@example.com"], + subject_prefix="[TEST]", + smtp_host="smtp.example.com", + smtp_port=587, + smtp_security="starttls", + smtp_user=None, + smtp_password=None, + ) + args = SimpleNamespace(show=False, dry_run=False, no_save=False) + poller = SimpleNamespace( + poll=AsyncMock(return_value=sample_state())) + + with patch.object(wlc_monitor, "Poller", return_value=poller), \ + patch.object( + wlc_monitor, "send_email", + side_effect=OSError("still unavailable"), + ) as send, redirect_stdout(StringIO()), \ + redirect_stderr(StringIO()): + self.assertEqual(wlc_monitor.run_once(cfg, args), 1) + send.assert_called_once() + + saved = json.loads(state_path.read_text()) + queued = saved["_pending_emails"] + self.assertEqual(len(queued), 2) + self.assertEqual(queued[0]["subject"], "OUTAGE") + self.assertIn("RECOVERY", queued[1]["subject"]) + + +class EmailTests(unittest.TestCase): + def config(self, security, user=None, password=None): + return SimpleNamespace( + mail_from="monitor@example.com", + mail_to=["ops@example.com"], + smtp_host="smtp.example.com", + smtp_port=587, + smtp_security=security, + smtp_user=user, + smtp_password=password, + ) + + def test_starttls_happens_before_authentication(self): + server = MagicMock() + session = server.__enter__.return_value + context = object() + with patch.object(wlc_monitor.ssl, "create_default_context", + return_value=context), \ + patch.object(wlc_monitor.smtplib, "SMTP", + return_value=server): + wlc_monitor.send_email( + self.config("starttls", "user", "password"), + "subject", + "body", + ) + self.assertEqual( + session.method_calls, + [ + call.starttls(context=context), + call.login("user", "password"), + call.send_message(ANY), + ], + ) + + def test_implicit_tls_uses_smtp_ssl(self): + server = MagicMock() + context = object() + with patch.object(wlc_monitor.ssl, "create_default_context", + return_value=context), \ + patch.object( + wlc_monitor.smtplib, "SMTP_SSL", return_value=server + ) as smtp_ssl: + wlc_monitor.send_email( + self.config("ssl"), "subject", "body") + smtp_ssl.assert_called_once_with( + "smtp.example.com", 587, timeout=15, context=context) + server.__enter__.return_value.send_message.assert_called_once() + + def test_plain_relay_does_not_start_tls_or_authenticate(self): + server = MagicMock() + session = server.__enter__.return_value + with patch.object(wlc_monitor.smtplib, "SMTP", + return_value=server): + wlc_monitor.send_email( + self.config("none"), "subject", "body") + session.starttls.assert_not_called() + session.login.assert_not_called() + session.send_message.assert_called_once() + + +if __name__ == "__main__": + unittest.main() diff --git a/wlc-monitor.service b/wlc-monitor.service new file mode 100644 index 0000000..c1d3466 --- /dev/null +++ b/wlc-monitor.service @@ -0,0 +1,34 @@ +[Unit] +Description=Cisco Catalyst 9800 WLC health check (SNMP -> email) +Documentation=file:/opt/wlc-monitor/README.md +After=network-online.target +Wants=network-online.target + +[Service] +Type=oneshot +User=wlcmon +Group=wlcmon +WorkingDirectory=/opt/wlc-monitor +ExecStart=/opt/wlc-monitor/venv/bin/python /opt/wlc-monitor/wlc_monitor.py \ + --config /opt/wlc-monitor/config.ini + +# Creates and owns /var/lib/wlc-monitor for the state file. +StateDirectory=wlc-monitor +StateDirectoryMode=0700 +UMask=0077 + +# The service only needs to talk UDP/161 outbound and SMTP outbound, and to +# write one JSON file. Everything else is off. +NoNewPrivileges=true +PrivateTmp=true +PrivateDevices=true +ProtectSystem=strict +ProtectHome=true +ProtectKernelTunables=true +ProtectKernelModules=true +ProtectControlGroups=true +RestrictAddressFamilies=AF_INET AF_INET6 +RestrictNamespaces=true +LockPersonality=true +MemoryDenyWriteExecute=true +SystemCallArchitectures=native diff --git a/wlc-monitor.timer b/wlc-monitor.timer new file mode 100644 index 0000000..3a888ac --- /dev/null +++ b/wlc-monitor.timer @@ -0,0 +1,12 @@ +[Unit] +Description=Poll the 9800 WLC every 5 minutes + +[Timer] +# Give the network a moment after boot before the first poll, otherwise a +# cold start reliably produces a false "controller unreachable". +OnBootSec=3min +OnUnitActiveSec=5min +AccuracySec=30s + +[Install] +WantedBy=timers.target diff --git a/wlc_monitor.py b/wlc_monitor.py new file mode 100755 index 0000000..7097cb6 --- /dev/null +++ b/wlc_monitor.py @@ -0,0 +1,669 @@ +#!/usr/bin/env python3 +"""SNMP health check for the Catalyst 9800 WLC, with email alerting. + +One invocation = one poll. State is kept in a JSON file so the next run can +diff against it; drive it from a systemd timer (see wlc-monitor.timer). + +Why poll instead of receiving traps/syslog: a poll that FAILS is itself a +signal. Traps go silent when the controller dies, and silence is +indistinguishable from "everything is fine". This design alerts on absence. + + wlc_monitor.py --config config.ini # one poll, send alerts + wlc_monitor.py --config config.ini --dry-run # poll, print, send nothing + wlc_monitor.py --config config.ini --test-email + wlc_monitor.py --config config.ini --show # dump current state + +Requires: pysnmp (pip install pysnmp) +""" + +from __future__ import annotations + +import argparse +import asyncio +import configparser +import json +import os +import signal +import smtplib +import socket +import ssl +import sys +import threading +import warnings +from datetime import datetime, timezone +from email.message import EmailMessage +from pathlib import Path + +warnings.filterwarnings("ignore", message=".*CFB has been moved.*") + +try: + from pysnmp.hlapi.v3arch.asyncio import ( + SnmpEngine, UsmUserData, UdpTransportTarget, ContextData, + ObjectType, ObjectIdentity, get_cmd, bulk_cmd, + usmHMACSHAAuthProtocol, usmHMAC128SHA224AuthProtocol, + usmHMAC192SHA256AuthProtocol, usmHMAC256SHA384AuthProtocol, + usmHMAC384SHA512AuthProtocol, usmAesCfb128Protocol, + ) +except ImportError: + sys.exit("error: pysnmp not installed. pip install pysnmp") + +AUTH_PROTOCOLS = { + "sha": usmHMACSHAAuthProtocol, + "sha1": usmHMACSHAAuthProtocol, + "sha224": usmHMAC128SHA224AuthProtocol, + "sha256": usmHMAC192SHA256AuthProtocol, + "sha384": usmHMAC256SHA384AuthProtocol, + "sha512": usmHMAC384SHA512AuthProtocol, +} + +# ---------------------------------------------------------------- OIDs +# Scalars. +OID_SYS_DESCR = "1.3.6.1.2.1.1.1.0" +OID_SYS_UPTIME = "1.3.6.1.2.1.1.3.0" +OID_SYS_NAME = "1.3.6.1.2.1.1.5.0" +# CISCO-CONFIG-MAN-MIB: sysUpTime at which running/startup last changed. +OID_RUN_LAST_CHANGED = "1.3.6.1.4.1.9.9.43.1.1.1.0" +OID_RUN_LAST_SAVED = "1.3.6.1.4.1.9.9.43.1.1.2.0" + +# CISCO-LWAPP-AP-MIB cLApEntry columns. Column meanings were verified against +# `show ap uptime` on this controller rather than taken from a MIB reference: +# .5 = AP name .6 = AP up time .7 = association (CAPWAP) up time +# Only APs currently joined appear in this table, so presence == joined. +AP_TABLE = "1.3.6.1.4.1.9.9.513.1.1.1.1" +AP_COL_NAME = f"{AP_TABLE}.5" +AP_COL_UPTIME = f"{AP_TABLE}.6" +AP_COL_ASSOC_UPTIME = f"{AP_TABLE}.7" + +# TimeTicks are 32-bit hundredths of a second and wrap at ~497 days. A counter +# that "decreased" may simply have wrapped, so distinguish the two by how far +# it moved FORWARD (mod 2^32) versus how much wall-clock time actually passed +# between polls. Five minutes of real time cannot produce months of ticks. +TICKS_MAX = 2 ** 32 +WRAP_SLACK_S = 600 # tolerate clock skew / a missed poll or two +WRAP_FACTOR = 3 # and be generous about timer jitter + + +def ticks_to_human(t: int) -> str: + s = int(t) // 100 + d, s = divmod(s, 86400) + h, s = divmod(s, 3600) + m, s = divmod(s, 60) + if d: + return f"{d}d {h}h {m}m" + if h: + return f"{h}h {m}m" + return f"{m}m {s}s" + + +def counter_reset(old: int, new: int, elapsed_s: float | None = None) -> bool: + """True if `new` is a genuine reset of `old` rather than a 32-bit wrap. + + A wrapped counter has still moved forward by only the elapsed time; a reset + counter has effectively jumped backwards by nearly its whole range. Compare + the forward distance (mod 2^32) against the time that really passed. + """ + if new >= old: + return False + forward = (new - old) % TICKS_MAX + if elapsed_s is None: + # No timing reference (first diff after an upgrade, say). Only a value + # very close to the ceiling is plausibly a wrap. + return not (old > 4.2e9 and new < 1e7) + budget = (elapsed_s * WRAP_FACTOR + WRAP_SLACK_S) * 100 + return forward > budget + + +# ---------------------------------------------------------------- SNMP +class Poller: + def __init__(self, host, port, user, auth, priv, auth_protocol, + timeout, retries): + self.host, self.port = host, port + self.user, self.auth, self.priv = user, auth, priv + self.auth_protocol = auth_protocol + self.timeout, self.retries = timeout, retries + self.engine = SnmpEngine() + + def _creds(self): + return UsmUserData(self.user, self.auth, self.priv, + authProtocol=AUTH_PROTOCOLS[self.auth_protocol], + privProtocol=usmAesCfb128Protocol) + + async def _target(self): + return await UdpTransportTarget.create( + (self.host, self.port), timeout=self.timeout, retries=self.retries) + + async def get(self, tgt, *oids): + ei, es, _ix, vbs = await get_cmd( + self.engine, self._creds(), tgt, ContextData(), + *[ObjectType(ObjectIdentity(o)) for o in oids], lookupMib=False) + if ei: + raise RuntimeError(str(ei)) + if es: + raise RuntimeError(es.prettyPrint()) + return [vb[1] for vb in vbs] + + async def walk(self, tgt, base): + """Return {oid_suffix: value} for one table column.""" + out, objs = {}, [ObjectType(ObjectIdentity(base))] + while True: + ei, es, _ix, rows = await bulk_cmd( + self.engine, self._creds(), tgt, ContextData(), 0, 25, + *objs, lookupMib=False) + if ei: + raise RuntimeError(str(ei)) + if es: + raise RuntimeError(es.prettyPrint()) + if not rows: + break + done = False + for vb in rows: + oid, val = str(vb[0]), vb[1] + if not oid.startswith(base + "."): + done = True + break + out[oid[len(base) + 1:]] = val + if done: + break + objs = [ObjectType(ObjectIdentity(str(rows[-1][0])))] + return out + + async def poll(self) -> dict: + tgt = await self._target() + descr, uptime, name, changed, saved = await self.get( + tgt, OID_SYS_DESCR, OID_SYS_UPTIME, OID_SYS_NAME, + OID_RUN_LAST_CHANGED, OID_RUN_LAST_SAVED) + + names = await self.walk(tgt, AP_COL_NAME) + ups = await self.walk(tgt, AP_COL_UPTIME) + assoc = await self.walk(tgt, AP_COL_ASSOC_UPTIME) + + aps = {} + for idx, nm in names.items(): + missing = [ + column for column, values in ( + ("AP uptime", ups), ("association uptime", assoc) + ) if idx not in values + ] + if missing: + raise RuntimeError( + f"incomplete SNMP AP row {idx}: missing {', '.join(missing)}") + aps[str(nm)] = { + "index": idx, + "ap_uptime": int(ups[idx]), + "assoc_uptime": int(assoc[idx]), + } + return { + "ok": True, + "sys_name": str(name), + "sys_descr": str(descr)[:120], + "sys_uptime": int(uptime), + "run_last_changed": int(changed), + "run_last_saved": int(saved), + "aps": aps, + } + + +# ---------------------------------------------------------------- diff +def elapsed_since(prev: dict) -> float | None: + """Wall-clock seconds since the previous poll, if we can tell.""" + ts = prev.get("last_poll") + if not ts: + return None + try: + then = datetime.fromisoformat(ts) + except (TypeError, ValueError): + return None + if then.tzinfo is None: + return None + delta = (datetime.now(timezone.utc) - then).total_seconds() + return delta if delta >= 0 else None + + +def build_alerts(prev: dict, cur: dict, cfg) -> list[tuple[str, str, str]]: + """Return [(severity, title, detail)]. prev may be {} on first run.""" + alerts = [] + first_run = not prev.get("aps") and not prev.get("ok") + el = elapsed_since(prev) + + # --- reachability transitions ------------------------------------- + if cur["ok"] and prev.get("consecutive_failures", 0) >= cfg.fail_threshold: + alerts.append(( + "RECOVERY", f"{cur['sys_name']}: controller reachable again", + f"SNMP polling recovered after " + f"{prev.get('consecutive_failures')} consecutive failures.\n" + f"Controller uptime is now {ticks_to_human(cur['sys_uptime'])}.")) + + if not cur["ok"]: + fails = cur.get("consecutive_failures", 1) + if fails == cfg.fail_threshold: # alert once, on crossing + alerts.append(( + "CRITICAL", "WLC unreachable over SNMP", + f"{fails} consecutive failed polls of {cfg.host}.\n" + f"Last error: {cur.get('error')}\n\n" + "The controller may be down, rebooting, or unreachable " + "from this host.")) + return alerts + + if first_run: + return alerts # nothing to diff against; stay quiet + + # --- controller reboot -------------------------------------------- + if counter_reset(prev.get("sys_uptime", 0), cur["sys_uptime"], el): + alerts.append(( + "CRITICAL", f"{cur['sys_name']} rebooted", + f"Controller uptime went from " + f"{ticks_to_human(prev['sys_uptime'])} to " + f"{ticks_to_human(cur['sys_uptime'])}.\n" + f"{cur['sys_descr']}")) + + # --- config changes ----------------------------------------------- + # Only compare when we actually have a previous reading. Treating a missing + # baseline as 0 would report a config change after every failed poll. + prev_changed = prev.get("run_last_changed") + if prev_changed is not None and cur["run_last_changed"] > prev_changed and \ + not counter_reset(prev.get("sys_uptime", 0), cur["sys_uptime"], el): + alerts.append(( + "NOTICE", f"{cur['sys_name']} running-config changed", + f"running-config last changed at controller uptime " + f"{ticks_to_human(cur['run_last_changed'])} " + f"(previous poll saw {ticks_to_human(prev_changed)}).\n\n" + "SNMP reports THAT the config changed, not what changed. " + "Check `show archive config differences` or the session logs.")) + + # unsaved config that stays unsaved + if cur["run_last_changed"] > cur["run_last_saved"]: + unsaved_for = cur["sys_uptime"] - cur["run_last_changed"] + overdue = unsaved_for > cfg.unsaved_ticks + if overdue and not prev.get("_unsaved_alerted", False): + alerts.append(( + "NOTICE", f"{cur['sys_name']} has unsaved config", + f"running-config has differed from startup-config for " + f"{ticks_to_human(unsaved_for)}.\n" + "A reboot would lose those changes. Run `write memory`.")) + cur["_unsaved_alerted"] = overdue + else: + cur["_unsaved_alerted"] = False + + # --- AP state ------------------------------------------------------ + prev_aps, cur_aps = prev.get("aps", {}), cur["aps"] + + for nm in sorted(set(prev_aps) - set(cur_aps)): + alerts.append(( + "CRITICAL", f"AP down: {nm}", + f"{nm} is no longer registered to {cur['sys_name']}.\n" + f"At the previous poll it had been joined for " + f"{ticks_to_human(prev_aps[nm]['assoc_uptime'])}.\n\n" + f"APs still joined: {len(cur_aps)}")) + + for nm in sorted(set(cur_aps) - set(prev_aps)): + alerts.append(( + "RECOVERY", f"AP joined: {nm}", + f"{nm} registered to {cur['sys_name']}.\n" + f"AP has been up {ticks_to_human(cur_aps[nm]['ap_uptime'])}, " + f"joined {ticks_to_human(cur_aps[nm]['assoc_uptime'])} ago.\n\n" + f"APs joined: {len(cur_aps)}")) + + for nm in sorted(set(cur_aps) & set(prev_aps)): + p, c = prev_aps[nm], cur_aps[nm] + if counter_reset(p["ap_uptime"], c["ap_uptime"], el): + alerts.append(( + "WARNING", f"AP rebooted: {nm}", + f"{nm} uptime went from {ticks_to_human(p['ap_uptime'])} " + f"to {ticks_to_human(c['ap_uptime'])}.\n" + "The access point restarted (power, watchdog, or upgrade).")) + elif counter_reset(p["assoc_uptime"], c["assoc_uptime"], el): + # Tunnel flapped without the AP itself rebooting -- this is the + # case a naive up/down check silently misses. + alerts.append(( + "WARNING", f"AP re-joined controller: {nm}", + f"{nm} association uptime reset from " + f"{ticks_to_human(p['assoc_uptime'])} to " + f"{ticks_to_human(c['assoc_uptime'])}, but AP uptime " + f"({ticks_to_human(c['ap_uptime'])}) did not reset.\n\n" + "The CAPWAP tunnel dropped and re-established without the AP " + "rebooting -- typically a network path or controller issue " + "rather than an AP fault.")) + return alerts + + +# ---------------------------------------------------------------- email +def send_email(cfg, subject: str, body: str) -> None: + msg = EmailMessage() + msg["From"] = cfg.mail_from + msg["To"] = ", ".join(cfg.mail_to) + msg["Subject"] = subject + msg["X-WLC-Monitor"] = socket.gethostname() + msg.set_content(body) + context = ssl.create_default_context() + if cfg.smtp_security == "ssl": + server = smtplib.SMTP_SSL( + cfg.smtp_host, cfg.smtp_port, timeout=15, context=context) + else: + server = smtplib.SMTP(cfg.smtp_host, cfg.smtp_port, timeout=15) + with server as s: + if cfg.smtp_security == "starttls": + s.starttls(context=context) + if cfg.smtp_user: + s.login(cfg.smtp_user, cfg.smtp_password) + s.send_message(msg) + + +def format_body(detail: str, cur: dict, cfg) -> str: + now = datetime.now(timezone.utc).astimezone().strftime("%Y-%m-%d %H:%M:%S %Z") + lines = [detail, "", "-" * 58, + f"controller : {cfg.host}", + f"detected : {now}", + f"monitored : {socket.gethostname()}"] + if cur.get("ok"): + lines += [f"uptime : {ticks_to_human(cur['sys_uptime'])}", + f"APs joined : {len(cur['aps'])}"] + for nm, ap in sorted(cur["aps"].items()): + lines.append(f" {nm} up {ticks_to_human(ap['ap_uptime'])}, " + f"joined {ticks_to_human(ap['assoc_uptime'])}") + return "\n".join(lines) + + +# ---------------------------------------------------------------- config +class Cfg: + def __init__(self, path: Path): + cp = configparser.ConfigParser() + # The file is optional: in a container every value can come from the + # environment, so credentials never have to be baked into an image. + found = cp.read(path) if path else [] + if path and not found and not os.environ.get("WLC_HOST"): + sys.exit(f"error: cannot read config {path} and WLC_HOST is unset") + for section in ("wlc", "mail", "monitor"): + if not cp.has_section(section): + cp.add_section(section) + w, m, g = cp["wlc"], cp["mail"], cp["monitor"] + + def s(env, sect, key, default=None): + """Environment wins over the file, file wins over the default.""" + v = os.environ.get(env) + if v is not None and v != "": + return v + return sect.get(key, fallback=default) + + def i(env, sect, key, default): + v = s(env, sect, key) + if v in (None, ""): + return default + try: + return int(v) + except ValueError: + sys.exit(f"error: {env or key} must be an integer, got {v!r}") + + def secret(env, sect, key): + """Read a secret from ENV, ENV_FILE, or the config file.""" + direct = os.environ.get(env) + file_name = os.environ.get(f"{env}_FILE") + if direct and file_name: + sys.exit(f"error: set only one of {env} and {env}_FILE") + if file_name: + try: + return Path(file_name).read_text().strip() + except OSError as e: + sys.exit(f"error: cannot read {env}_FILE {file_name!r}: {e}") + if direct: + return direct + return sect.get(key, fallback=None) + + self.host = s("WLC_HOST", w, "host") + self.port = i("WLC_PORT", w, "port", 161) + self.user = s("WLC_SNMP_USER", w, "snmp_user") + self.auth = secret("WLC_SNMP_AUTH", w, "snmp_auth") + self.priv = secret("WLC_SNMP_PRIV", w, "snmp_priv") + self.auth_protocol = ( + s("WLC_SNMP_AUTH_PROTOCOL", w, "auth_protocol", "sha") + or "sha" + ).lower() + self.timeout = i("WLC_TIMEOUT", w, "timeout", 5) + self.retries = i("WLC_RETRIES", w, "retries", 1) + + self.smtp_host = s("MAIL_SMTP_HOST", m, "smtp_host") + self.smtp_port = i("MAIL_SMTP_PORT", m, "smtp_port", 25) + self.smtp_security = ( + s("MAIL_SMTP_SECURITY", m, "smtp_security", "none") or "none" + ).lower() + self.smtp_user = s("MAIL_SMTP_USER", m, "smtp_user") + self.smtp_password = secret("MAIL_SMTP_PASSWORD", m, "smtp_password") + self.mail_from = s("MAIL_FROM", m, "from") + to = s("MAIL_TO", m, "to") or "" + self.mail_to = [a.strip() for a in to.split(",") if a.strip()] + self.subject_prefix = s("MAIL_SUBJECT_PREFIX", m, "subject_prefix", "[WLC]") + + self.state_file = Path( + s("MONITOR_STATE_FILE", g, "state_file", "/data/state.json")).expanduser() + self.fail_threshold = i("MONITOR_FAIL_THRESHOLD", g, "fail_threshold", 2) + self.unsaved_minutes = i( + "MONITOR_UNSAVED_MINUTES", g, "unsaved_minutes", 60) + self.unsaved_ticks = self.unsaved_minutes * 60 * 100 + self.interval = i("MONITOR_INTERVAL", g, "interval", 300) + + missing = [n for n, v in ( + ("host", self.host), ("snmp_user", self.user), + ("snmp_auth", self.auth), ("snmp_priv", self.priv), + ("smtp_host", self.smtp_host), ("mail_from", self.mail_from) + ) if not v] + if missing: + sys.exit(f"error: missing required settings: {', '.join(missing)}") + if not self.mail_to: + sys.exit("error: no recipients configured (MAIL_TO / [mail] to)") + if self.auth_protocol not in AUTH_PROTOCOLS: + choices = ", ".join(sorted(AUTH_PROTOCOLS)) + sys.exit(f"error: unsupported SNMP auth protocol " + f"{self.auth_protocol!r}; choose {choices}") + if self.smtp_security not in {"none", "starttls", "ssl"}: + sys.exit("error: MAIL_SMTP_SECURITY must be none, starttls, or ssl") + if bool(self.smtp_user) != bool(self.smtp_password): + sys.exit("error: SMTP username and password must be set together") + if self.smtp_user and self.smtp_security == "none": + sys.exit("error: SMTP authentication requires starttls or ssl") + + for name, value, minimum, maximum in ( + ("WLC_PORT", self.port, 1, 65535), + ("MAIL_SMTP_PORT", self.smtp_port, 1, 65535), + ("WLC_TIMEOUT", self.timeout, 1, None), + ("WLC_RETRIES", self.retries, 0, None), + ("MONITOR_FAIL_THRESHOLD", self.fail_threshold, 1, None), + ("MONITOR_UNSAVED_MINUTES", self.unsaved_minutes, 0, None), + ("MONITOR_INTERVAL", self.interval, 1, None), + ): + if value < minimum or (maximum is not None and value > maximum): + expected = f"{minimum}..{maximum}" if maximum else f">= {minimum}" + sys.exit(f"error: {name} must be {expected}, got {value}") + + if found and path and path.stat().st_mode & 0o007: + print(f"warning: {path} is world-accessible; it holds SNMP " + f"credentials. Use mode 600, or 640 with a dedicated group.", + file=sys.stderr) + + +def load_state(p: Path) -> dict: + try: + state = json.loads(p.read_text()) + except (FileNotFoundError, json.JSONDecodeError): + return {} + return state if isinstance(state, dict) else {} + + +def save_state(p: Path, state: dict) -> None: + p.parent.mkdir(mode=0o700, parents=True, exist_ok=True) + tmp = p.with_suffix(p.suffix + ".tmp") + tmp.write_text(json.dumps(state, indent=2)) + tmp.chmod(0o600) + tmp.replace(p) # atomic; a killed run can't leave half a state file + + +# ---------------------------------------------------------------- loop +def run_loop(cfg, a) -> int: + """Poll forever. Used when running as a container instead of a timer.""" + stop = threading.Event() + + def handle(signum, _frame): + # Docker sends SIGTERM and waits ~10s before SIGKILL. Exit promptly + # and cleanly so a stop/restart never truncates the state file. + print(f"[{signal.Signals(signum).name}] shutting down", flush=True) + stop.set() + + signal.signal(signal.SIGTERM, handle) + signal.signal(signal.SIGINT, handle) + + print(f"polling {cfg.host} every {cfg.interval}s; " + f"state={cfg.state_file}; alerts -> {', '.join(cfg.mail_to)}", + flush=True) + + while not stop.is_set(): + started = datetime.now(timezone.utc) + try: + run_once(cfg, a) + except Exception as e: + # Never let one bad iteration kill the monitor; a crashed monitor + # is silent, and silence is the failure mode we are avoiding. + print(f"poll iteration failed: {type(e).__name__}: {e}", + file=sys.stderr, flush=True) + # Subtract the work time so the cadence stays honest under slow polls. + elapsed = (datetime.now(timezone.utc) - started).total_seconds() + stop.wait(max(1.0, cfg.interval - elapsed)) + return 0 + + +# ---------------------------------------------------------------- main +def main(argv=None) -> int: + ap = argparse.ArgumentParser( + description="SNMP health check + email alerts for a Catalyst 9800.") + ap.add_argument("--config", default=None, + help="path to config.ini. Optional: environment variables " + "can supply every setting (see README). Defaults to " + "config.ini beside this script if it exists.") + ap.add_argument("--dry-run", action="store_true", + help="poll and print alerts, send no mail, still save state") + ap.add_argument("--no-save", action="store_true", help="do not update state file") + ap.add_argument("--test-email", action="store_true", help="send one test message and exit") + ap.add_argument("--show", action="store_true", help="print the current poll as JSON and exit") + ap.add_argument("--loop", action="store_true", + help="poll forever every --interval seconds (container mode)") + ap.add_argument("--healthcheck", action="store_true", + help="exit 0 if the last poll is recent, 1 otherwise") + ap.add_argument("--interval", type=int, default=None, + help="seconds between polls in --loop mode " + "(default: MONITOR_INTERVAL or 300)") + a = ap.parse_args(argv) + + if a.config: + cfg_path = Path(a.config) + else: + beside = Path(__file__).with_name("config.ini") + cfg_path = beside if beside.exists() else None + cfg = Cfg(cfg_path) + if a.interval is not None: + if a.interval < 1: + ap.error("--interval must be at least 1 second") + cfg.interval = a.interval + + if a.healthcheck: + # Deliberately reports on the LOOP, not on the controller. An + # unreachable WLC is a working monitor doing its job -- restarting the + # container for that would throw away the state that suppresses + # duplicate alerts. Only a stalled poll loop is unhealthy. + st = load_state(cfg.state_file) + age = elapsed_since(st) + if age is None: + print("no state yet") + return 1 + limit = cfg.interval * 3 + 60 + print(f"last poll {int(age)}s ago (limit {limit}s)") + return 0 if age <= limit else 1 + + if a.test_email: + try: + send_email(cfg, f"{cfg.subject_prefix} test message", + "This is a test from wlc_monitor.py.\n" + f"Sent via {cfg.smtp_host}:{cfg.smtp_port} from " + f"{socket.gethostname()}.\n\n" + "If you are reading this, the alert path works.") + except Exception as e: + print(f"FAILED to send: {type(e).__name__}: {e}", file=sys.stderr) + return 1 + print(f"test message sent to {', '.join(cfg.mail_to)}") + return 0 + + if a.loop: + return run_loop(cfg, a) + return run_once(cfg, a) + + +def run_once(cfg, a) -> int: + prev = load_state(cfg.state_file) + pending = [ + item for item in prev.get("_pending_emails", []) + if isinstance(item, dict) and {"subject", "body"} <= item.keys() + ] + poller = Poller(cfg.host, cfg.port, cfg.user, cfg.auth, cfg.priv, + cfg.auth_protocol, cfg.timeout, cfg.retries) + try: + cur = asyncio.run(poller.poll()) + cur["consecutive_failures"] = 0 + except Exception as e: + # Carry the last known good readings forward. Without this, a failed + # poll erases the baseline and the next success looks like a change. + cur = {"ok": False, "error": f"{type(e).__name__}: {e}", + "consecutive_failures": prev.get("consecutive_failures", 0) + 1, + "sys_name": prev.get("sys_name", cfg.host), + "aps": prev.get("aps", {})} + for k in ("sys_uptime", "sys_descr", "run_last_changed", + "run_last_saved", "_unsaved_alerted"): + if k in prev: + cur[k] = prev[k] + + if a.show: + print(json.dumps(cur, indent=2, default=str)) + return 0 if cur["ok"] else 1 + + alerts = build_alerts(prev, cur, cfg) + messages = pending + [ + { + "subject": f"{cfg.subject_prefix} {severity}: {title}", + "body": format_body(detail, cur, cfg), + } + for severity, title, detail in alerts + ] + + rc = 0 + remaining = list(pending) if a.dry_run else [] + for index, message in enumerate(messages): + subject, body = message["subject"], message["body"] + if a.dry_run: + print(f"\n=== would send: {subject} ===\n{body}") + else: + try: + send_email(cfg, subject, body) + print(f"sent: {subject}") + except Exception as e: + print(f"FAILED to send {subject!r}: {type(e).__name__}: {e}", + file=sys.stderr) + # Preserve FIFO ordering: a later recovery must never arrive + # before the outage message whose delivery just failed. + remaining.extend(messages[index:]) + rc = 1 + break + + if not messages: + state = "ok" if cur["ok"] else f"UNREACHABLE (x{cur['consecutive_failures']})" + print(f"no change; {state}; " + f"APs joined: {len(cur.get('aps', {}))}") + + # Advance the comparison baseline even when delivery fails, but retain a + # small durable outbox so transition alerts are retried on the next poll. + if remaining: + cur["_pending_emails"] = remaining + else: + cur.pop("_pending_emails", None) + if not a.no_save: + cur["last_poll"] = datetime.now(timezone.utc).isoformat() + save_state(cfg.state_file, cur) + return rc + + +if __name__ == "__main__": + raise SystemExit(main())