From 2073be2a78ae7487091cd3c25adb2404de02f242 Mon Sep 17 00:00:00 2001 From: rdwr-rahulk Date: Wed, 16 Sep 2026 16:58:02 +0530 Subject: [PATCH 1/7] V 1.5.3 --- .env.example | 8 ++++++++ DEPLOYMENT.md | 8 +++++++- Dockerfile | 6 ++++++ README.md | 9 +++++++-- docker-compose.build.yaml | 18 ++++++++++++++++++ docker-compose.yaml | 18 ++++++++++++++++++ install.sh | 27 +++++++++++++++++++++++++++ 7 files changed, 91 insertions(+), 3 deletions(-) diff --git a/.env.example b/.env.example index 8752222..900f0b9 100644 --- a/.env.example +++ b/.env.example @@ -36,5 +36,13 @@ SMTP_PASSWORD=your-smtp-password/api key value # No credentials needed in env — syslog is unauthenticated (UDP/TCP). # Configure host, port, protocol, and facility entirely in watchdog-config.yaml. +# ── Docker socket access (non-root hardening) ──────────────────────────────── +#GID = Group ID — a number Linux uses to identify a user group (the group equivalent of a user's UID). +# GID of the group that owns /var/run/docker.sock on this host. The container +# runs as a non-root user and joins this group (via group_add in +# docker-compose.yaml) to read the socket. install.sh auto-detects this; for +# manual setups find it with: stat -c '%g' /var/run/docker.sock +DOCKER_GID=999 + # ── Tuning ──────────────────────────────────────────────────────────────────── LOG_LEVEL=INFO diff --git a/DEPLOYMENT.md b/DEPLOYMENT.md index 85360c8..50613b9 100644 --- a/DEPLOYMENT.md +++ b/DEPLOYMENT.md @@ -49,6 +49,8 @@ This guide covers initial deployment, alert channel configuration, and ongoing o | Git | Required to clone the repository | | Host permissions | Root, or membership in the `docker` group (to read `/var/run/docker.sock`) | +> The watchdog container itself runs as a non-root user and joins the host's `docker` group at runtime (via the `DOCKER_GID` value in `.env`, auto-detected by `install.sh`) to read the socket — see [2.2 Configure Environment Variables (.env)](#22-configure-environment-variables-env). + ```bash docker compose version ``` @@ -115,6 +117,10 @@ SLACK_WEBHOOK_URL=https://hooks.slack.com/services/T.../B.../... SMTP_USERNAME=your-smtp-username/login email SMTP_PASSWORD=your-smtp-password/api key value +# Docker socket access (non-root container) — GID of the group that owns +# /var/run/docker.sock on this host. Find it with: stat -c '%g' /var/run/docker.sock +DOCKER_GID=999 + # Tuning (optional — defaults shown) LOG_LEVEL=INFO ``` @@ -723,7 +729,7 @@ docker compose logs docker-container-watchdog ``` Common causes: -- `/var/run/docker.sock` is not accessible — ensure the host socket exists and the container has read access +- `/var/run/docker.sock` is not accessible — ensure the host socket exists and the container has read access. The container runs as a non-root user and needs `DOCKER_GID` in `.env` to match the socket's actual group (`stat -c '%g' /var/run/docker.sock`) — see [2.2 Configure Environment Variables (.env)](#22-configure-environment-variables-env) - Missing `.env` file — run `cp .env.example .env` and fill in values ### Alert Notifications Not Received diff --git a/Dockerfile b/Dockerfile index 778f5fc..be62abe 100644 --- a/Dockerfile +++ b/Dockerfile @@ -13,4 +13,10 @@ RUN pip install --no-cache-dir \ # Copy agent COPY watchdog.py . +# Non-root — the docker.sock group membership needed to read the socket is +# granted at runtime via `group_add` in docker-compose.yaml (GID varies per host). +RUN useradd --uid 1000 --create-home --shell /usr/sbin/nologin watchdog \ + && chown -R watchdog:watchdog /app +USER watchdog + CMD ["python3", "-u", "watchdog.py"] diff --git a/README.md b/README.md index 11ff3c3..b22f35c 100644 --- a/README.md +++ b/README.md @@ -90,10 +90,14 @@ Two additional deduplication rules suppress redundant alerts at the event level: | Item | Size | |------|------| | Docker image (`watchdog:latest`) | ~180 MB (python:3.11-slim base + dependencies) | -| Running container (memory) | ~50–80 MB | +| Running container (memory) | ~50–80 MB baseline; capped at `mem_limit: 256m` in `docker-compose.yaml` | | `watchdog.tar` export | ~170 MB | -> **Note:** The values above are approximate and may vary depending on the host operating system, Docker version, and installed dependencies. +> **Note:** The values above are approximate and may vary depending on the host operating system, Docker version, and installed dependencies. If usage approaches the 256 MB limit (check with `docker stats docker-container-watchdog`), raise `mem_limit`/`mem_reservation` in `docker-compose.yaml` rather than removing the limit. + +### Container Hardening + +The container runs as a non-root user (UID 1000) with a read-only root filesystem, no extra Linux capabilities (`cap_drop: ALL`), and `no-new-privileges` set. Access to `/var/run/docker.sock` is granted by adding the container's user to the host's `docker` group via `group_add` — the GID is auto-detected by `install.sh` and stored as `DOCKER_GID` in `.env` (see [DEPLOYMENT.md](DEPLOYMENT.md#22-configure-environment-variables-env)). CPU and process count are also capped (`cpus: "0.50"`, `pids_limit: 200`) so a leak or runaway condition is contained to this container instead of the host. ### Python Dependencies @@ -556,6 +560,7 @@ Probe selection is automatic: containers with a Docker `HEALTHCHECK` are monitor | Version | Date | Author | Changes | |---------|------------|--------|---------| +| 1.5.3 | 2026-09-16 | Rahul Kumar | Resource Limits Enforced | | 1.5.2 | 2026-08-31 | Rahul Kumar | Updated error message"Suppressing expected Cyber Controller SQL dump syntax-check container termination (exit 137) alert" | | 1.5.1 | 2026-08-31 | Rahul Kumar | fixed ignore Dynamic container crash alert | | 1.5.0 | 2026-08-27 | Rahul Kumar | Added ignore Dynamic container crash alert | diff --git a/docker-compose.build.yaml b/docker-compose.build.yaml index 85a98fb..505b458 100644 --- a/docker-compose.build.yaml +++ b/docker-compose.build.yaml @@ -26,6 +26,24 @@ services: - /var/run/docker.sock:/var/run/docker.sock:ro - ./watchdog-config.yaml:/etc/watchdog/watchdog-config.yaml:ro - ./watchdog:/var/log/watchdog + # ── Hardening ──────────────────────────────────────────────────────────── + user: "1000:1000" # non-root + group_add: + - "${DOCKER_GID:-999}" # host docker.sock group GID — auto-detected into .env by install.sh + read_only: true # immutable root filesystem + tmpfs: + - /tmp + security_opt: + - no-new-privileges:true + cap_drop: + - ALL + # ── Resource limits — contain a leak/runaway to this container, not the host ─ + mem_limit: 256m + mem_reservation: 128m + memswap_limit: 256m # no swap beyond mem_limit — hits the limit and restarts instead of degrading the host + mem_swappiness: 0 # avoid swapping this container's pages — needed when host kernel lacks swap accounting (memswap_limit is then unenforceable) + cpus: "0.50" + pids_limit: 200 healthcheck: test: ["CMD", "python3", "-c", "import docker; docker.from_env().ping()"] interval: 30s diff --git a/docker-compose.yaml b/docker-compose.yaml index 25284ea..e5443e6 100644 --- a/docker-compose.yaml +++ b/docker-compose.yaml @@ -32,6 +32,24 @@ services: - ./watchdog.py:/app/watchdog.py:ro # Persistent log storage — written directly to host folder - ./watchdog:/var/log/watchdog + # ── Hardening ──────────────────────────────────────────────────────────── + user: "1000:1000" # non-root + group_add: + - "${DOCKER_GID:-999}" # host docker.sock group GID — auto-detected into .env by install.sh + read_only: true # immutable root filesystem + tmpfs: + - /tmp + security_opt: + - no-new-privileges:true + cap_drop: + - ALL + # ── Resource limits — contain a leak/runaway to this container, not the host ─ + mem_limit: 256m + mem_reservation: 128m + memswap_limit: 256m # no swap beyond mem_limit — hits the limit and restarts instead of degrading the host + mem_swappiness: 0 # avoid swapping this container's pages — needed when host kernel lacks swap accounting (memswap_limit is then unenforceable) + cpus: "0.50" + pids_limit: 200 healthcheck: test: ["CMD", "python3", "-c", "import docker; docker.from_env().ping()"] interval: 30s diff --git a/install.sh b/install.sh index 4200d3e..d7884b8 100644 --- a/install.sh +++ b/install.sh @@ -118,9 +118,30 @@ setup_log_directory() { else print_info "Log directory already exists: ${LOG_DIR}" fi + # Container runs as non-root (UID 1000) — own it directly instead of opening it to all local users + if chown 1000:1000 "$LOG_DIR" 2>/dev/null; then + chmod 750 "$LOG_DIR" + else + print_warning "Could not chown ${LOG_DIR} to UID 1000 (not running as root?) — falling back to group-writable permissions" + chmod 775 "$LOG_DIR" + fi print_info "Logs will be written to: ${LOG_DIR}/watchdog.log" } +################################################################################ +# Docker Socket Group Detection +################################################################################ + +detect_docker_gid() { + # GID that owns /var/run/docker.sock — the non-root container joins this + # group (via group_add in docker-compose.yaml) to read the socket. + if [ -S /var/run/docker.sock ]; then + stat -c '%g' /var/run/docker.sock 2>/dev/null || stat -f '%g' /var/run/docker.sock 2>/dev/null || echo "999" + else + echo "999" + fi +} + ################################################################################ # Docker Image ################################################################################ @@ -334,6 +355,11 @@ configure_all() { WATCHDOG_HOST=$(_prompt "Hostname to display in alerts" "$HOST_DEFAULT") echo "" + # ── Docker socket GID (container runs non-root; needs group access to the socket) ── + local DOCKER_GID + DOCKER_GID=$(detect_docker_gid) + print_info "Docker socket GID detected: ${DOCKER_GID}" + # ── Write .env ──────────────────────────────────────────────────────────── { printf "# .env — generated by install.sh on %s\n" "$(date '+%Y-%m-%d %H:%M:%S')" @@ -349,6 +375,7 @@ configure_all() { "$SNMP_V3_AUTH_KEY" "$SNMP_V3_PRIV_KEY" fi printf "# Alert identity\nWATCHDOG_HOST=%s\n\n" "$WATCHDOG_HOST" + printf "# Docker socket access (non-root container)\nDOCKER_GID=%s\n\n" "$DOCKER_GID" printf "# Tuning\nLOG_LEVEL=INFO\n" } > "$ENV_FILE" chmod 600 "$ENV_FILE" From 5b224a29a1e3a0311b6715160844646875eb5d96 Mon Sep 17 00:00:00 2001 From: rdwr-rahulk Date: Thu, 17 Sep 2026 13:51:12 +0530 Subject: [PATCH 2/7] Updated v 1.5.3 --- .env.example | 4 +++- DEPLOYMENT.md | 4 +++- README.md | 4 ++-- docker-compose.build.yaml | 5 ++++- docker-compose.yaml | 5 ++++- install.sh | 41 ++++++++++++++++++++++++++++++--------- 6 files changed, 48 insertions(+), 15 deletions(-) diff --git a/.env.example b/.env.example index 900f0b9..f0c363e 100644 --- a/.env.example +++ b/.env.example @@ -42,7 +42,9 @@ SMTP_PASSWORD=your-smtp-password/api key value # runs as a non-root user and joins this group (via group_add in # docker-compose.yaml) to read the socket. install.sh auto-detects this; for # manual setups find it with: stat -c '%g' /var/run/docker.sock -DOCKER_GID=999 +# Replace the placeholder below with that number — docker compose refuses to +# start the container while DOCKER_GID is unset or left as this placeholder. +DOCKER_GID=REPLACE_WITH_DOCKER_SOCKET_GID # ── Tuning ──────────────────────────────────────────────────────────────────── LOG_LEVEL=INFO diff --git a/DEPLOYMENT.md b/DEPLOYMENT.md index 50613b9..b5f87c7 100644 --- a/DEPLOYMENT.md +++ b/DEPLOYMENT.md @@ -119,7 +119,9 @@ SMTP_PASSWORD=your-smtp-password/api key value # Docker socket access (non-root container) — GID of the group that owns # /var/run/docker.sock on this host. Find it with: stat -c '%g' /var/run/docker.sock -DOCKER_GID=999 +# Replace the placeholder below with that number — docker compose refuses to +# start the container while DOCKER_GID is unset or left as this placeholder. +DOCKER_GID=REPLACE_WITH_DOCKER_SOCKET_GID # Tuning (optional — defaults shown) LOG_LEVEL=INFO diff --git a/README.md b/README.md index b22f35c..dc148a8 100644 --- a/README.md +++ b/README.md @@ -93,11 +93,11 @@ Two additional deduplication rules suppress redundant alerts at the event level: | Running container (memory) | ~50–80 MB baseline; capped at `mem_limit: 256m` in `docker-compose.yaml` | | `watchdog.tar` export | ~170 MB | -> **Note:** The values above are approximate and may vary depending on the host operating system, Docker version, and installed dependencies. If usage approaches the 256 MB limit (check with `docker stats docker-container-watchdog`), raise `mem_limit`/`mem_reservation` in `docker-compose.yaml` rather than removing the limit. +> **Note:** The values above are approximate and may vary depending on the host operating system, Docker version, and installed dependencies. If usage approaches the 256 MB limit (check with `docker stats docker-container-watchdog`), raise `mem_limit`/`mem_reservation` in `docker-compose.yaml` rather than removing the limit — and raise `memswap_limit` to at least the new `mem_limit` in **both** `docker-compose.yaml` and `docker-compose.build.yaml`, since Docker rejects a config where `memswap_limit` is lower than `mem_limit`. ### Container Hardening -The container runs as a non-root user (UID 1000) with a read-only root filesystem, no extra Linux capabilities (`cap_drop: ALL`), and `no-new-privileges` set. Access to `/var/run/docker.sock` is granted by adding the container's user to the host's `docker` group via `group_add` — the GID is auto-detected by `install.sh` and stored as `DOCKER_GID` in `.env` (see [DEPLOYMENT.md](DEPLOYMENT.md#22-configure-environment-variables-env)). CPU and process count are also capped (`cpus: "0.50"`, `pids_limit: 200`) so a leak or runaway condition is contained to this container instead of the host. +The container runs as a non-root user (UID 1000) with a read-only root filesystem, no extra Linux capabilities (`cap_drop: ALL`), and `no-new-privileges` set. These controls, together with the CPU/process caps (`cpus: "0.50"`, `pids_limit: 200`), contain a leak or runaway condition in the watchdog process to this container instead of the host — but they do **not** sandbox the Docker API. Access to `/var/run/docker.sock` is granted by adding the container's user to the host's `docker` group via `group_add` (GID auto-detected by `install.sh`, stored as `DOCKER_GID` in `.env` — see [DEPLOYMENT.md](DEPLOYMENT.md#22-configure-environment-variables-env)), and that socket is effectively host-root-equivalent: anything with access to it can create privileged containers or bind-mount the host filesystem. A compromise of the watchdog process itself is therefore **not** contained by `cap_drop`, `no-new-privileges`, or the read-only filesystem — treat `docker.sock` access as the primary trust boundary when deciding who/what can reach this host. ### Python Dependencies diff --git a/docker-compose.build.yaml b/docker-compose.build.yaml index 505b458..be7fd2b 100644 --- a/docker-compose.build.yaml +++ b/docker-compose.build.yaml @@ -29,7 +29,9 @@ services: # ── Hardening ──────────────────────────────────────────────────────────── user: "1000:1000" # non-root group_add: - - "${DOCKER_GID:-999}" # host docker.sock group GID — auto-detected into .env by install.sh + # host docker.sock group GID — auto-detected into .env by install.sh. + # Required (no fallback): a wrong/guessed GID silently breaks socket access. + - "${DOCKER_GID:?Set DOCKER_GID in .env to the docker.sock GID - run install.sh or see DEPLOYMENT.md}" read_only: true # immutable root filesystem tmpfs: - /tmp @@ -41,6 +43,7 @@ services: mem_limit: 256m mem_reservation: 128m memswap_limit: 256m # no swap beyond mem_limit — hits the limit and restarts instead of degrading the host + # must stay >= mem_limit if you raise it, or Docker rejects this config mem_swappiness: 0 # avoid swapping this container's pages — needed when host kernel lacks swap accounting (memswap_limit is then unenforceable) cpus: "0.50" pids_limit: 200 diff --git a/docker-compose.yaml b/docker-compose.yaml index e5443e6..8033eca 100644 --- a/docker-compose.yaml +++ b/docker-compose.yaml @@ -35,7 +35,9 @@ services: # ── Hardening ──────────────────────────────────────────────────────────── user: "1000:1000" # non-root group_add: - - "${DOCKER_GID:-999}" # host docker.sock group GID — auto-detected into .env by install.sh + # host docker.sock group GID — auto-detected into .env by install.sh. + # Required (no fallback): a wrong/guessed GID silently breaks socket access. + - "${DOCKER_GID:?Set DOCKER_GID in .env to the docker.sock GID - run install.sh or see DEPLOYMENT.md}" read_only: true # immutable root filesystem tmpfs: - /tmp @@ -47,6 +49,7 @@ services: mem_limit: 256m mem_reservation: 128m memswap_limit: 256m # no swap beyond mem_limit — hits the limit and restarts instead of degrading the host + # must stay >= mem_limit if you raise it, or Docker rejects this config mem_swappiness: 0 # avoid swapping this container's pages — needed when host kernel lacks swap accounting (memswap_limit is then unenforceable) cpus: "0.50" pids_limit: 200 diff --git a/install.sh b/install.sh index d7884b8..9390c3a 100644 --- a/install.sh +++ b/install.sh @@ -118,12 +118,18 @@ setup_log_directory() { else print_info "Log directory already exists: ${LOG_DIR}" fi - # Container runs as non-root (UID 1000) — own it directly instead of opening it to all local users - if chown 1000:1000 "$LOG_DIR" 2>/dev/null; then + # Container runs as non-root (UID 1000:GID 1000) — must own the dir and any + # pre-existing files (e.g. log/rotated files left by an older root-run + # version) directly; group-writable permissions don't help unless the + # container's GID is actually a member of that group. + if chown -R 1000:1000 "$LOG_DIR" 2>/dev/null; then chmod 750 "$LOG_DIR" else - print_warning "Could not chown ${LOG_DIR} to UID 1000 (not running as root?) — falling back to group-writable permissions" - chmod 775 "$LOG_DIR" + print_error "Could not chown ${LOG_DIR} to UID 1000:GID 1000 (not running as root?)" + echo "" + echo " Re-run this script with sudo, or fix ownership manually:" + echo " sudo chown -R 1000:1000 ${LOG_DIR}" + exit 1 fi print_info "Logs will be written to: ${LOG_DIR}/watchdog.log" } @@ -135,11 +141,19 @@ setup_log_directory() { detect_docker_gid() { # GID that owns /var/run/docker.sock — the non-root container joins this # group (via group_add in docker-compose.yaml) to read the socket. - if [ -S /var/run/docker.sock ]; then - stat -c '%g' /var/run/docker.sock 2>/dev/null || stat -f '%g' /var/run/docker.sock 2>/dev/null || echo "999" - else - echo "999" + # No silent fallback: a wrong guessed GID breaks socket access at container + # start with a confusing error, so treat detection failure as fatal here. + if [ ! -S /var/run/docker.sock ]; then + print_error "/var/run/docker.sock not found — cannot determine its group GID" + return 1 + fi + local gid + gid=$(stat -c '%g' /var/run/docker.sock 2>/dev/null || stat -f '%g' /var/run/docker.sock 2>/dev/null) + if [ -z "$gid" ]; then + print_error "Could not determine the GID that owns /var/run/docker.sock" + return 1 fi + echo "$gid" } ################################################################################ @@ -241,6 +255,15 @@ configure_all() { IFS= read -r RECONFIG if [[ ! "$RECONFIG" =~ ^[Yy]$ ]]; then print_info "Keeping existing configuration" + # Upgrade path: older installs may predate DOCKER_GID — add it now + # rather than silently starting with the unset/wrong-GID fallback. + if [ -f "$ENV_FILE" ] && ! grep -q '^DOCKER_GID=' "$ENV_FILE"; then + print_warning "Existing .env is missing DOCKER_GID — detecting and adding it" + local MIGRATED_GID + MIGRATED_GID=$(detect_docker_gid) || exit 1 + printf "\n# Docker socket access (non-root container)\nDOCKER_GID=%s\n" "$MIGRATED_GID" >> "$ENV_FILE" + print_success "Added DOCKER_GID=${MIGRATED_GID} to ${ENV_FILE}" + fi return 0 fi echo "" @@ -357,7 +380,7 @@ configure_all() { # ── Docker socket GID (container runs non-root; needs group access to the socket) ── local DOCKER_GID - DOCKER_GID=$(detect_docker_gid) + DOCKER_GID=$(detect_docker_gid) || exit 1 print_info "Docker socket GID detected: ${DOCKER_GID}" # ── Write .env ──────────────────────────────────────────────────────────── From e1f8b5d564469fef14cdfc12792164398596998a Mon Sep 17 00:00:00 2001 From: rdwr-rahulk Date: Thu, 17 Sep 2026 14:14:05 +0530 Subject: [PATCH 3/7] Fixed 1.5.3 --- DEPLOYMENT.md | 10 +++++++++ install.sh | 57 +++++++++++++++++++++++++++++++++++---------------- 2 files changed, 49 insertions(+), 18 deletions(-) diff --git a/DEPLOYMENT.md b/DEPLOYMENT.md index b5f87c7..d3a318f 100644 --- a/DEPLOYMENT.md +++ b/DEPLOYMENT.md @@ -362,6 +362,16 @@ Download [`watchdog.tar` — pre-built Docker image for offline installation](ht docker load -i watchdog.tar ``` +#### Prepare the log directory + +The container runs as UID/GID 1000 and writes to the bind-mounted `./watchdog` directory. Create it and set ownership **before** the first `docker compose up`, otherwise Compose creates it as `root` and the non-root container cannot create `watchdog.log` — `RotatingFileHandler` fails at startup and the `restart: always` container loops. (Option A's `install.sh` does this for you.) + +```bash +mkdir -p watchdog +sudo chown -R 1000:1000 watchdog +sudo chmod 750 watchdog +``` + #### Start the container ```bash diff --git a/install.sh b/install.sh index 9390c3a..d1c9f10 100644 --- a/install.sh +++ b/install.sh @@ -118,19 +118,30 @@ setup_log_directory() { else print_info "Log directory already exists: ${LOG_DIR}" fi - # Container runs as non-root (UID 1000:GID 1000) — must own the dir and any - # pre-existing files (e.g. log/rotated files left by an older root-run - # version) directly; group-writable permissions don't help unless the - # container's GID is actually a member of that group. - if chown -R 1000:1000 "$LOG_DIR" 2>/dev/null; then - chmod 750 "$LOG_DIR" - else - print_error "Could not chown ${LOG_DIR} to UID 1000:GID 1000 (not running as root?)" + # Container runs as non-root (UID 1000:GID 1000) — it must own the dir and any + # pre-existing files (logs left by an older root-run version), or + # RotatingFileHandler cannot create/append watchdog.log and the container + # restart-loops. Elevate with sudo when we are not already root. + local PRIV="" + if [ "$EUID" -ne 0 ]; then + if command -v sudo &>/dev/null; then + PRIV="sudo" + else + print_error "Not running as root and sudo is not available — cannot set ${LOG_DIR} ownership to UID 1000:GID 1000" + echo "" + echo " Re-run this script as root, or fix ownership manually before starting:" + echo " chown -R 1000:1000 ${LOG_DIR} && chmod 750 ${LOG_DIR}" + exit 1 + fi + fi + if ! $PRIV chown -R 1000:1000 "$LOG_DIR"; then + print_error "Failed to set ownership of ${LOG_DIR} to UID 1000:GID 1000" echo "" - echo " Re-run this script with sudo, or fix ownership manually:" - echo " sudo chown -R 1000:1000 ${LOG_DIR}" + echo " Fix ownership manually before starting:" + echo " sudo chown -R 1000:1000 ${LOG_DIR} && sudo chmod 750 ${LOG_DIR}" exit 1 fi + $PRIV chmod 750 "$LOG_DIR" print_info "Logs will be written to: ${LOG_DIR}/watchdog.log" } @@ -255,14 +266,24 @@ configure_all() { IFS= read -r RECONFIG if [[ ! "$RECONFIG" =~ ^[Yy]$ ]]; then print_info "Keeping existing configuration" - # Upgrade path: older installs may predate DOCKER_GID — add it now - # rather than silently starting with the unset/wrong-GID fallback. - if [ -f "$ENV_FILE" ] && ! grep -q '^DOCKER_GID=' "$ENV_FILE"; then - print_warning "Existing .env is missing DOCKER_GID — detecting and adding it" - local MIGRATED_GID - MIGRATED_GID=$(detect_docker_gid) || exit 1 - printf "\n# Docker socket access (non-root container)\nDOCKER_GID=%s\n" "$MIGRATED_GID" >> "$ENV_FILE" - print_success "Added DOCKER_GID=${MIGRATED_GID} to ${ENV_FILE}" + # Upgrade path: older installs predate DOCKER_GID, and `cp .env.example .env` + # leaves a non-numeric placeholder. Treat anything that isn't a bare number + # as "not configured" and (re)detect it, or group_add gets a bad value and + # the container fails to start. + if [ -f "$ENV_FILE" ]; then + local CURRENT_GID + CURRENT_GID=$(grep -E '^DOCKER_GID=' "$ENV_FILE" | tail -n1 | cut -d= -f2 | tr -d '[:space:]') + if ! [[ "$CURRENT_GID" =~ ^[0-9]+$ ]]; then + print_warning "Existing .env has a missing or invalid DOCKER_GID ('${CURRENT_GID}') — detecting and setting it" + local MIGRATED_GID + MIGRATED_GID=$(detect_docker_gid) || exit 1 + if grep -qE '^DOCKER_GID=' "$ENV_FILE"; then + sed -i.bak -E "s|^DOCKER_GID=.*|DOCKER_GID=${MIGRATED_GID}|" "$ENV_FILE" && rm -f "${ENV_FILE}.bak" + else + printf "\n# Docker socket access (non-root container)\nDOCKER_GID=%s\n" "$MIGRATED_GID" >> "$ENV_FILE" + fi + print_success "Set DOCKER_GID=${MIGRATED_GID} in ${ENV_FILE}" + fi fi return 0 fi From 212d19a285077ecd3e3c13619f5135d55e062768 Mon Sep 17 00:00:00 2001 From: rdwr-rahulk Date: Fri, 18 Sep 2026 16:26:40 +0530 Subject: [PATCH 4/7] version 1.5.4 --- .env.example | 18 +++++++------- DEPLOYMENT.md | 22 ++++------------- Dockerfile | 7 ++++-- README.md | 2 +- docker-compose.build.yaml | 9 +++---- docker-compose.yaml | 14 ++++++----- install.sh | 50 +++++++++++++++++++++++---------------- 7 files changed, 62 insertions(+), 60 deletions(-) diff --git a/.env.example b/.env.example index f0c363e..b1b7d2d 100644 --- a/.env.example +++ b/.env.example @@ -36,15 +36,15 @@ SMTP_PASSWORD=your-smtp-password/api key value # No credentials needed in env — syslog is unauthenticated (UDP/TCP). # Configure host, port, protocol, and facility entirely in watchdog-config.yaml. -# ── Docker socket access (non-root hardening) ──────────────────────────────── -#GID = Group ID — a number Linux uses to identify a user group (the group equivalent of a user's UID). -# GID of the group that owns /var/run/docker.sock on this host. The container -# runs as a non-root user and joins this group (via group_add in -# docker-compose.yaml) to read the socket. install.sh auto-detects this; for -# manual setups find it with: stat -c '%g' /var/run/docker.sock -# Replace the placeholder below with that number — docker compose refuses to -# start the container while DOCKER_GID is unset or left as this placeholder. -DOCKER_GID=REPLACE_WITH_DOCKER_SOCKET_GID +# ── Non-root hardening (optional, advanced) ────────────────────────────────── +# By default the container runs as root so `docker compose up` works with zero +# configuration — install.sh sets these three automatically instead, running +# the container as UID/GID 1000 joined to the docker.sock group. To opt into +# the same hardening for a manual install, uncomment and fill in DOCKER_GID +# (find it with: stat -c '%g' /var/run/docker.sock): +# WATCHDOG_UID=1000 +# WATCHDOG_GID=1000 +# DOCKER_GID= # ── Tuning ──────────────────────────────────────────────────────────────────── LOG_LEVEL=INFO diff --git a/DEPLOYMENT.md b/DEPLOYMENT.md index d3a318f..f03479d 100644 --- a/DEPLOYMENT.md +++ b/DEPLOYMENT.md @@ -49,7 +49,7 @@ This guide covers initial deployment, alert channel configuration, and ongoing o | Git | Required to clone the repository | | Host permissions | Root, or membership in the `docker` group (to read `/var/run/docker.sock`) | -> The watchdog container itself runs as a non-root user and joins the host's `docker` group at runtime (via the `DOCKER_GID` value in `.env`, auto-detected by `install.sh`) to read the socket — see [2.2 Configure Environment Variables (.env)](#22-configure-environment-variables-env). +> By default the container runs as root so a manual install needs no host-specific group setup. The automated installer ([3.1 Option A](#31-option-a-automated-installation-installsh-recommended)) instead hardens it to run non-root (UID/GID 1000, joined to the docker.sock group) automatically, with no input required. Either way it's still contained by a read-only root filesystem, dropped Linux capabilities (`cap_drop: ALL`), `no-new-privileges`, and CPU/memory/PID limits (see [Container Hardening in the README](README.md#container-hardening)). ```bash docker compose version @@ -117,18 +117,14 @@ SLACK_WEBHOOK_URL=https://hooks.slack.com/services/T.../B.../... SMTP_USERNAME=your-smtp-username/login email SMTP_PASSWORD=your-smtp-password/api key value -# Docker socket access (non-root container) — GID of the group that owns -# /var/run/docker.sock on this host. Find it with: stat -c '%g' /var/run/docker.sock -# Replace the placeholder below with that number — docker compose refuses to -# start the container while DOCKER_GID is unset or left as this placeholder. -DOCKER_GID=REPLACE_WITH_DOCKER_SOCKET_GID - # Tuning (optional — defaults shown) LOG_LEVEL=INFO ``` Only credentials and secrets belong in `.env`. Hosts, ports, recipients, and thresholds are configured in `watchdog-config.yaml`, never hardcoded in this guide or in scripts. +The container runs as root by default for a zero-configuration manual install. If you want the non-root hardening that [Option A](#31-option-a-automated-installation-installsh-recommended) applies automatically, uncomment the `WATCHDOG_UID`/`WATCHDOG_GID`/`DOCKER_GID` lines in `.env.example` and fill in `DOCKER_GID` (`stat -c '%g' /var/run/docker.sock`). + --- ### 2.3 Configure Watchdog Global Settings @@ -362,16 +358,6 @@ Download [`watchdog.tar` — pre-built Docker image for offline installation](ht docker load -i watchdog.tar ``` -#### Prepare the log directory - -The container runs as UID/GID 1000 and writes to the bind-mounted `./watchdog` directory. Create it and set ownership **before** the first `docker compose up`, otherwise Compose creates it as `root` and the non-root container cannot create `watchdog.log` — `RotatingFileHandler` fails at startup and the `restart: always` container loops. (Option A's `install.sh` does this for you.) - -```bash -mkdir -p watchdog -sudo chown -R 1000:1000 watchdog -sudo chmod 750 watchdog -``` - #### Start the container ```bash @@ -741,7 +727,7 @@ docker compose logs docker-container-watchdog ``` Common causes: -- `/var/run/docker.sock` is not accessible — ensure the host socket exists and the container has read access. The container runs as a non-root user and needs `DOCKER_GID` in `.env` to match the socket's actual group (`stat -c '%g' /var/run/docker.sock`) — see [2.2 Configure Environment Variables (.env)](#22-configure-environment-variables-env) +- `/var/run/docker.sock` is not accessible — ensure the host socket exists and Docker is running. Manual installs run the container as root, so no host-side group configuration is required; if you opted into non-root hardening (`WATCHDOG_UID`/`WATCHDOG_GID`/`DOCKER_GID` in `.env`), verify `DOCKER_GID` matches `stat -c '%g' /var/run/docker.sock`. - Missing `.env` file — run `cp .env.example .env` and fill in values ### Alert Notifications Not Received diff --git a/Dockerfile b/Dockerfile index be62abe..6109679 100644 --- a/Dockerfile +++ b/Dockerfile @@ -13,10 +13,13 @@ RUN pip install --no-cache-dir \ # Copy agent COPY watchdog.py . -# Non-root — the docker.sock group membership needed to read the socket is -# granted at runtime via `group_add` in docker-compose.yaml (GID varies per host). +# UID 1000 account for install.sh's non-root hardening path (docker-compose.yaml +# defaults to root; install.sh instead sets WATCHDOG_UID/GID to run as this user). RUN useradd --uid 1000 --create-home --shell /usr/sbin/nologin watchdog \ && chown -R watchdog:watchdog /app + +# Non-root by default so a bare `docker run` is restricted. Compose always sets +# `user:` explicitly, so manual installs still get their root default. USER watchdog CMD ["python3", "-u", "watchdog.py"] diff --git a/README.md b/README.md index dc148a8..ce5c7cc 100644 --- a/README.md +++ b/README.md @@ -97,7 +97,7 @@ Two additional deduplication rules suppress redundant alerts at the event level: ### Container Hardening -The container runs as a non-root user (UID 1000) with a read-only root filesystem, no extra Linux capabilities (`cap_drop: ALL`), and `no-new-privileges` set. These controls, together with the CPU/process caps (`cpus: "0.50"`, `pids_limit: 200`), contain a leak or runaway condition in the watchdog process to this container instead of the host — but they do **not** sandbox the Docker API. Access to `/var/run/docker.sock` is granted by adding the container's user to the host's `docker` group via `group_add` (GID auto-detected by `install.sh`, stored as `DOCKER_GID` in `.env` — see [DEPLOYMENT.md](DEPLOYMENT.md#22-configure-environment-variables-env)), and that socket is effectively host-root-equivalent: anything with access to it can create privileged containers or bind-mount the host filesystem. A compromise of the watchdog process itself is therefore **not** contained by `cap_drop`, `no-new-privileges`, or the read-only filesystem — treat `docker.sock` access as the primary trust boundary when deciding who/what can reach this host. +By default the container runs as **root** so `docker compose up` works with zero configuration — no host-specific group setup needed to read `/var/run/docker.sock`. [install.sh](install.sh) instead auto-detects the socket's GID and opts into running the container as **non-root UID/GID 1000** (joined to that group via `group_add`), with no manual input required; manual installs can do the same by setting `WATCHDOG_UID`/`WATCHDOG_GID`/`DOCKER_GID` in `.env` (see [DEPLOYMENT.md](DEPLOYMENT.md#22-configure-environment-variables-env)). Either way, the container is constrained by a read-only root filesystem, no extra Linux capabilities (`cap_drop: ALL`), `no-new-privileges`, and CPU/memory/PID caps (`cpus: "0.50"`, `pids_limit: 200`, `mem_limit: 256m`), which contain a leak or runaway condition to this container instead of the host. Note that access to `/var/run/docker.sock` is effectively host-root-equivalent for **any** user (root or not): anything that can reach the socket can create privileged containers or bind-mount the host filesystem, so a compromise of the watchdog process is **not** contained by these controls — treat `docker.sock` access as the primary trust boundary when deciding who/what can reach this host. ### Python Dependencies diff --git a/docker-compose.build.yaml b/docker-compose.build.yaml index be7fd2b..4f48538 100644 --- a/docker-compose.build.yaml +++ b/docker-compose.build.yaml @@ -27,11 +27,12 @@ services: - ./watchdog-config.yaml:/etc/watchdog/watchdog-config.yaml:ro - ./watchdog:/var/log/watchdog # ── Hardening ──────────────────────────────────────────────────────────── - user: "1000:1000" # non-root + # Defaults to root (0:0) for zero-config manual installs — the socket is + # host-root-equivalent for any user anyway. install.sh instead populates + # WATCHDOG_UID/WATCHDOG_GID/DOCKER_GID in .env to run this non-root. + user: "${WATCHDOG_UID:-0}:${WATCHDOG_GID:-0}" group_add: - # host docker.sock group GID — auto-detected into .env by install.sh. - # Required (no fallback): a wrong/guessed GID silently breaks socket access. - - "${DOCKER_GID:?Set DOCKER_GID in .env to the docker.sock GID - run install.sh or see DEPLOYMENT.md}" + - "${DOCKER_GID:-0}" read_only: true # immutable root filesystem tmpfs: - /tmp diff --git a/docker-compose.yaml b/docker-compose.yaml index 8033eca..4a6e275 100644 --- a/docker-compose.yaml +++ b/docker-compose.yaml @@ -28,16 +28,18 @@ services: - /var/run/docker.sock:/var/run/docker.sock:ro # Config file (read-only mount) - ./watchdog-config.yaml:/etc/watchdog/watchdog-config.yaml:ro - # Watchdog script — mount so code changes take effect without image rebuild - - ./watchdog.py:/app/watchdog.py:ro + # watchdog.py is deliberately NOT bind-mounted: this container holds the + # Docker socket, so write access to the host copy would be host root. + # Code ships inside the image — rebuild to pick up changes. # Persistent log storage — written directly to host folder - ./watchdog:/var/log/watchdog # ── Hardening ──────────────────────────────────────────────────────────── - user: "1000:1000" # non-root + # Defaults to root (0:0) for zero-config manual installs — the socket is + # host-root-equivalent for any user anyway. install.sh instead populates + # WATCHDOG_UID/WATCHDOG_GID/DOCKER_GID in .env to run this non-root. + user: "${WATCHDOG_UID:-0}:${WATCHDOG_GID:-0}" group_add: - # host docker.sock group GID — auto-detected into .env by install.sh. - # Required (no fallback): a wrong/guessed GID silently breaks socket access. - - "${DOCKER_GID:?Set DOCKER_GID in .env to the docker.sock GID - run install.sh or see DEPLOYMENT.md}" + - "${DOCKER_GID:-0}" read_only: true # immutable root filesystem tmpfs: - /tmp diff --git a/install.sh b/install.sh index d1c9f10..8423c61 100644 --- a/install.sh +++ b/install.sh @@ -118,10 +118,10 @@ setup_log_directory() { else print_info "Log directory already exists: ${LOG_DIR}" fi - # Container runs as non-root (UID 1000:GID 1000) — it must own the dir and any - # pre-existing files (logs left by an older root-run version), or - # RotatingFileHandler cannot create/append watchdog.log and the container - # restart-loops. Elevate with sudo when we are not already root. + # This installer hardens the container to run as UID/GID 1000 (see + # detect_docker_gid below) instead of the compose file's root default, so + # it must own the dir and any pre-existing files up front. Elevate with + # sudo when we are not already root. local PRIV="" if [ "$EUID" -ne 0 ]; then if command -v sudo &>/dev/null; then @@ -141,7 +141,10 @@ setup_log_directory() { echo " sudo chown -R 1000:1000 ${LOG_DIR} && sudo chmod 750 ${LOG_DIR}" exit 1 fi - $PRIV chmod 750 "$LOG_DIR" + if ! $PRIV chmod 750 "$LOG_DIR"; then + print_error "Failed to set permissions on ${LOG_DIR}" + exit 1 + fi print_info "Logs will be written to: ${LOG_DIR}/watchdog.log" } @@ -177,6 +180,10 @@ load_docker_image() { # Check if image is already present if docker image inspect "${IMAGE_NAME}" &>/dev/null; then print_info "Image ${IMAGE_NAME} already exists locally — using existing image" + # watchdog.py is baked into the image, not bind-mounted, so a stale image + # keeps running old code even after the host copy is updated. + print_warning "Agent code runs from the image, not ${INSTALL_DIR}/watchdog.py" + print_info " To pick up code changes: docker rmi ${IMAGE_NAME} && bash install.sh" return 0 fi @@ -266,23 +273,25 @@ configure_all() { IFS= read -r RECONFIG if [[ ! "$RECONFIG" =~ ^[Yy]$ ]]; then print_info "Keeping existing configuration" - # Upgrade path: older installs predate DOCKER_GID, and `cp .env.example .env` - # leaves a non-numeric placeholder. Treat anything that isn't a bare number - # as "not configured" and (re)detect it, or group_add gets a bad value and - # the container fails to start. + # Upgrade path: installs from before non-root hardening existed + # predate these vars, which would silently fall back to the + # compose file's root default instead of the intended hardening. if [ -f "$ENV_FILE" ]; then local CURRENT_GID CURRENT_GID=$(grep -E '^DOCKER_GID=' "$ENV_FILE" | tail -n1 | cut -d= -f2 | tr -d '[:space:]') if ! [[ "$CURRENT_GID" =~ ^[0-9]+$ ]]; then - print_warning "Existing .env has a missing or invalid DOCKER_GID ('${CURRENT_GID}') — detecting and setting it" + print_warning "Existing .env predates non-root hardening — detecting and adding it" local MIGRATED_GID MIGRATED_GID=$(detect_docker_gid) || exit 1 - if grep -qE '^DOCKER_GID=' "$ENV_FILE"; then - sed -i.bak -E "s|^DOCKER_GID=.*|DOCKER_GID=${MIGRATED_GID}|" "$ENV_FILE" && rm -f "${ENV_FILE}.bak" - else - printf "\n# Docker socket access (non-root container)\nDOCKER_GID=%s\n" "$MIGRATED_GID" >> "$ENV_FILE" - fi - print_success "Set DOCKER_GID=${MIGRATED_GID} in ${ENV_FILE}" + for kv in "WATCHDOG_UID=1000" "WATCHDOG_GID=1000" "DOCKER_GID=${MIGRATED_GID}"; do + local key="${kv%%=*}" + if grep -qE "^${key}=" "$ENV_FILE"; then + sed -i.bak -E "s|^${key}=.*|${kv}|" "$ENV_FILE" && rm -f "${ENV_FILE}.bak" + else + printf "%s\n" "$kv" >> "$ENV_FILE" + fi + done + print_success "Set WATCHDOG_UID=1000, WATCHDOG_GID=1000, DOCKER_GID=${MIGRATED_GID} in ${ENV_FILE}" fi fi return 0 @@ -399,10 +408,10 @@ configure_all() { WATCHDOG_HOST=$(_prompt "Hostname to display in alerts" "$HOST_DEFAULT") echo "" - # ── Docker socket GID (container runs non-root; needs group access to the socket) ── - local DOCKER_GID + # ── Non-root hardening (manual `docker compose up` defaults to root instead) ── + local WATCHDOG_UID="1000" WATCHDOG_GID="1000" DOCKER_GID DOCKER_GID=$(detect_docker_gid) || exit 1 - print_info "Docker socket GID detected: ${DOCKER_GID}" + print_info "Docker socket GID detected: ${DOCKER_GID} — running non-root as UID/GID ${WATCHDOG_UID}" # ── Write .env ──────────────────────────────────────────────────────────── { @@ -419,7 +428,8 @@ configure_all() { "$SNMP_V3_AUTH_KEY" "$SNMP_V3_PRIV_KEY" fi printf "# Alert identity\nWATCHDOG_HOST=%s\n\n" "$WATCHDOG_HOST" - printf "# Docker socket access (non-root container)\nDOCKER_GID=%s\n\n" "$DOCKER_GID" + printf "# Non-root hardening\nWATCHDOG_UID=%s\nWATCHDOG_GID=%s\nDOCKER_GID=%s\n\n" \ + "$WATCHDOG_UID" "$WATCHDOG_GID" "$DOCKER_GID" printf "# Tuning\nLOG_LEVEL=INFO\n" } > "$ENV_FILE" chmod 600 "$ENV_FILE" From 9824ff51c6241605b5bdebba2afb07f82400670c Mon Sep 17 00:00:00 2001 From: rdwr-rahulk Date: Fri, 25 Sep 2026 18:19:53 +0530 Subject: [PATCH 5/7] V 1.5.4 --- .env.example | 5 + .github/workflows/ci.yml | 40 ++++ DEPLOYMENT.md | 46 ++-- README.md | 13 +- VERSION | 1 + docker-compose.build.yaml | 11 +- docker-compose.yaml | 8 +- install.sh | 378 ++++++++++++++++++++++++------ uninstall.sh | 78 ++++--- watchdog-config.yaml.example | 9 + watchdog.py | 429 +++++++++++++++++++++++++++++------ 11 files changed, 819 insertions(+), 199 deletions(-) create mode 100644 .github/workflows/ci.yml create mode 100644 VERSION diff --git a/.env.example b/.env.example index b1b7d2d..c826a1a 100644 --- a/.env.example +++ b/.env.example @@ -46,5 +46,10 @@ SMTP_PASSWORD=your-smtp-password/api key value # WATCHDOG_GID=1000 # DOCKER_GID= +# ── Image version ───────────────────────────────────────────────────────────── +# Release tag the container runs (see the repository VERSION file). install.sh +# sets this to the version it loaded from watchdog.tar; unset means watchdog:latest. +# WATCHDOG_VERSION= + # ── Tuning ──────────────────────────────────────────────────────────────────── LOG_LEVEL=INFO diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml new file mode 100644 index 0000000..f8e65ee --- /dev/null +++ b/.github/workflows/ci.yml @@ -0,0 +1,40 @@ +name: CI + +on: + pull_request: + branches: + - main + +jobs: + validate: + runs-on: ubuntu-latest + + steps: + - name: Checkout repository + uses: actions/checkout@v4 + + - name: Validate shell scripts + run: | + bash -n install.sh + bash -n uninstall.sh + + - name: Validate Python syntax + run: | + python3 -m py_compile watchdog.py + + - name: Validate VERSION file + run: | + test -f VERSION + VERSION=$(tr -d '[:space:]' < VERSION) + echo "Repository version: ${VERSION}" + echo "${VERSION}" | grep -Eq '^[0-9]+\.[0-9]+\.[0-9]+$' + grep -q "| ${VERSION} |" README.md + + - name: Prepare CI environment + run: | + cp .env.example .env + + - name: Validate Docker Compose + run: | + docker compose -f docker-compose.yaml config -q + docker compose -f docker-compose.build.yaml config -q diff --git a/DEPLOYMENT.md b/DEPLOYMENT.md index f03479d..dec6bf5 100644 --- a/DEPLOYMENT.md +++ b/DEPLOYMENT.md @@ -324,14 +324,14 @@ The script walks through the following stages in order: |---|---| | **Prerequisites** | Verifies Docker and Docker Compose are installed and running | | **Log directory** | Creates `./watchdog/` for persistent log storage | -| **Load image** | Loads `watchdog.tar` into Docker (`watchdog:latest`); builds from source if archive is absent | -| **Host identification** | `WATCHDOG_HOST` is hardcoded to `CyberController-Server` in `docker-compose.yaml` | +| **Load image** | Loads `watchdog.tar` and requires it to contain `watchdog:` (from the root `VERSION` file); aborts on a version mismatch, and builds from source only when the archive is absent | +| **Host identification** | `WATCHDOG_HOST` is taken from `.env` (the installer prompts for it), defaulting to `CyberController-Server` | | **Configuration wizard** | Selects channels and prompts for Slack/SMTP/SNMP/Syslog values interactively | -| **Credentials + config write** | Generates/updates `.env` and `watchdog-config.yaml` from wizard answers | -| **Start** | Runs `docker compose up -d` in the background | -| **Verify** | Checks the container is running and prints a summary with common commands | +| **Credentials + config write** | Generates/updates `.env` (including `WATCHDOG_VERSION`) and `watchdog-config.yaml` from wizard answers | +| **Start** | Runs `docker compose up -d` in the background, aborting if Compose reports a failure | +| **Verify** | Fails the installation unless the container is running **and** uses `watchdog:` | -> **Re-running `install.sh` on an existing installation is safe.** If configuration files already exist, the installer asks whether to reconfigure from scratch. Choose `N` to keep existing files unchanged. +> **Re-running `install.sh` on an existing installation is safe.** If both configuration files exist, the installer asks whether to reconfigure from scratch. Choose `N` to keep them unchanged — the packaged image is still loaded and the deployment is still moved to the new version. If only one of the two files is present, the wizard regenerates both, since a half-present configuration cannot start. Continue to [4. Verification](#4-verification). @@ -430,17 +430,18 @@ Re-run [4.1 Confirm Container Status and Health](#41-confirm-container-status-an ### 5.2 Roll Back an Image Upgrade -The runtime image is always tagged `watchdog:latest`, so loading or pulling a new image overwrites the previous one. Before upgrading the image (see [6.3 Upgrade Guide](#63-upgrade-guide)), tag the current working image so it can be restored: +Each release is tagged `watchdog:` and the deployed tag is pinned by `WATCHDOG_VERSION` in `.env`. As long as the previous image is still on the host, rolling back is a version change: ```bash -docker tag watchdog:latest watchdog:rollback +docker images watchdog # list the versions available locally +sed -i 's/^WATCHDOG_VERSION=.*/WATCHDOG_VERSION=1.5.3/' .env +docker compose up -d ``` -If the new image misbehaves after `docker load -i watchdog.tar` or `docker compose pull`, restore the previous image and redeploy: +If the previous image was only ever tagged `watchdog:latest`, preserve it under a rollback tag before loading a new archive: ```bash -docker tag watchdog:rollback watchdog:latest -docker compose up -d +docker tag watchdog:latest watchdog:rollback ``` ### 5.3 Roll Back an Application Upgrade (Git) @@ -522,11 +523,10 @@ docker compose up -d #### Apply docker-compose.yaml changes (container settings) -For example, edit the `WATCHDOG_HOST` value in `docker-compose.yaml`: +For example, change the hostname shown in alerts by editing `WATCHDOG_HOST` in `.env`: -```yaml -environment: - WATCHDOG_HOST: my-new-server-name +```bash +WATCHDOG_HOST=my-new-server-name ``` Then redeploy to take effect: @@ -646,13 +646,16 @@ docker compose up -d ##### Option B: Offline Image Upgrade 1. Request the Radware RE team to create an updated pre-built Docker image for offline installation. See [8.4 Support Contacts](#84-support-contacts). -2. Load the image. Make sure the new image uses the same name, `watchdog:latest`. +2. Load the image. The archive carries the release tag `watchdog:` matching the package's `VERSION` file. ```bash docker load -i watchdog.tar + sed -i "s/^WATCHDOG_VERSION=.*/WATCHDOG_VERSION=$(cat VERSION)/" .env docker compose up -d ``` + Running `bash install.sh` performs all three steps (and verifies the result) automatically. + --- #### Checking the Installed Version @@ -696,7 +699,7 @@ Shows a removal plan (container name, image size, log directory size), asks for bash uninstall.sh --keep-logs ``` -Stops and removes the `watchdog` container and the `watchdog:latest` Docker image. The `./watchdog/` log directory is left intact so you can review historical logs later. +Stops and removes the `watchdog` container and the `watchdog:` and `watchdog:latest` Docker image tags. The `./watchdog/` log directory is left intact so you can review historical logs later. #### Remove everything including logs @@ -801,13 +804,14 @@ docker compose restart docker-container-watchdog #### Host Identification in Alerts -Set `WATCHDOG_HOST` in `docker-compose.yaml` under the watchdog service environment: +Set `WATCHDOG_HOST` in `.env` (written by `install.sh` from the hostname prompt), then run `docker compose up -d`: -```yaml -environment: - WATCHDOG_HOST: my-server-name +```bash +WATCHDOG_HOST=my-server-name ``` +When it is unset, `docker-compose.yaml` falls back to `CyberController-Server`. + --- ### 8.3 SNMP Trap Var-Binds diff --git a/README.md b/README.md index ce5c7cc..502995f 100644 --- a/README.md +++ b/README.md @@ -79,8 +79,9 @@ Two additional deduplication rules suppress redundant alerts at the event level: ├── requirements-watchdog.txt # Python dependencies (reference; Dockerfile pip-installs inline) ├── install.sh # Offline install script ├── uninstall.sh # Stop + remove script -├── watchdog.tar # Pre-built Docker image (provided, no internet needed) -└── .env # Secrets — NOT committed to git +├── VERSION # Release version — drives the image tag and installer +├── watchdog.tar # Pre-built Docker image (provided, no internet needed) +└── .env # Secrets — NOT committed to git ``` --- @@ -89,7 +90,7 @@ Two additional deduplication rules suppress redundant alerts at the event level: | Item | Size | |------|------| -| Docker image (`watchdog:latest`) | ~180 MB (python:3.11-slim base + dependencies) | +| Docker image (`watchdog:`) | ~180 MB (python:3.11-slim base + dependencies) | | Running container (memory) | ~50–80 MB baseline; capped at `mem_limit: 256m` in `docker-compose.yaml` | | `watchdog.tar` export | ~170 MB | @@ -133,8 +134,12 @@ All non-secret settings live in `watchdog-config.yaml`. Secrets (webhook URLs, p | `alert_on_recovery` | `true` | Send an INFO "recovered" alert once a previously-alarmed container returns to normal | | `excluded_containers` | `[]` | Container names to never alert on | | `ignored_container_events` | MariaDB syntax-check label rule | Label-based event suppressions for intentional temporary containers | +| `auto_health_check` | enabled | Probe containers reporting `health=none` by discovering an HTTP/TCP endpoint | +| `container_health_checks` | `{}` | Per-container probe overrides (`type: http` or `type: exec`) for `health=none` containers; takes precedence over `auto_health_check` | | `log_level` | `INFO` | `DEBUG` / `INFO` / `WARNING` / `ERROR` | | `log_file` | `/var/log/watchdog/watchdog.log` | Bind-mounted to `./watchdog/watchdog.log` on host. Rotates at 10 MB, 5 backups. Set to `null` to disable | +| `state_file` | `/var/log/watchdog/state.json` | Persists open alerts and the host boot time so recovery and reboot reporting survive a restart. Set to `""` to disable | +| `boot_grace_seconds` | `180` | After a detected host reboot, hold alerts this long while services start, then send one `reboot-summary`. `0` alerts immediately and sends the summary on the first poll | | `runbook_base_url` | — | URL included in every alert | Preferred suppression: Temporary MariaDB HA syntax-check containers should be created with Docker label `com.radware.cybercontroller.role=mariadb-ha-syntax-check`. The watchdog ignores only configured events for containers carrying that explicit label; normal service containers still alert. @@ -252,6 +257,7 @@ bash install.sh | `unhealthy` | HIGH | Health probe failing for N consecutive cycles | | `restart-loop` | HIGH | Container restarted ≥ threshold times within window | | `recovered` | INFO | Previously-alarmed container (`crashed`/`oom`/`unhealthy`/`restart-loop`) is healthy/running again. Fires once per incident; controlled by `alert_on_recovery` (default `true`) | +| `reboot-summary` | INFO / HIGH | Sent once after a host reboot is detected, when the `boot_grace_seconds` window closes. Reports boot time, how many containers are running, which recovered, and which are still failing. Always sent after a reboot — an INFO summary confirms the node came back cleanly. HIGH if any container is still failing | OOM alerts include **memory stats** (usage / limit / peak) prepended to the log snippet. @@ -560,6 +566,7 @@ Probe selection is automatic: containers with a Docker `HEALTHCHECK` are monitor | Version | Date | Author | Changes | |---------|------------|--------|---------| +| 1.5.4 | 2026-09-23 | Rahul Kumar | Versioned Docker image (`watchdog:`) driven by the root `VERSION` file; installer loads the packaged `watchdog.tar`, rejects an archive that does not carry the release tag, aborts on Compose failures, and verifies the running image; restart-loop detection now uses Docker `RestartCount` instead of counting poll observations; `container_health_checks` overrides implemented; fixed `.env` migration on root-owned files; `watchdog-config.yaml` made readable by the non-root container user | | 1.5.3 | 2026-09-16 | Rahul Kumar | Resource Limits Enforced | | 1.5.2 | 2026-08-31 | Rahul Kumar | Updated error message"Suppressing expected Cyber Controller SQL dump syntax-check container termination (exit 137) alert" | | 1.5.1 | 2026-08-31 | Rahul Kumar | fixed ignore Dynamic container crash alert | diff --git a/VERSION b/VERSION new file mode 100644 index 0000000..94fe62c --- /dev/null +++ b/VERSION @@ -0,0 +1 @@ +1.5.4 diff --git a/docker-compose.build.yaml b/docker-compose.build.yaml index 4f48538..5b9116d 100644 --- a/docker-compose.build.yaml +++ b/docker-compose.build.yaml @@ -5,8 +5,11 @@ # docker compose -f docker-compose.build.yaml build # docker compose -f docker-compose.build.yaml up -d # -# After a successful build, export the image for offline customers: -# docker save watchdog:latest -o watchdog.tar +# After a successful build, export the image for offline customers — always +# ship the versioned tag so install.sh can pin the deployment to it: +# VERSION=$(cat VERSION) +# docker tag watchdog:"$VERSION" watchdog:latest +# docker save watchdog:"$VERSION" watchdog:latest -o watchdog.tar # ───────────────────────────────────────────────────────────────────────────── services: @@ -14,14 +17,14 @@ services: docker-container-watchdog: build: context: . - image: watchdog:latest + image: watchdog:${WATCHDOG_VERSION:-latest} container_name: docker-container-watchdog restart: always network_mode: host env_file: .env environment: WATCHDOG_CONFIG: /etc/watchdog/watchdog-config.yaml - WATCHDOG_HOST: cyber-controller-server + WATCHDOG_HOST: ${WATCHDOG_HOST:-CyberController-Server} volumes: - /var/run/docker.sock:/var/run/docker.sock:ro - ./watchdog-config.yaml:/etc/watchdog/watchdog-config.yaml:ro diff --git a/docker-compose.yaml b/docker-compose.yaml index 4a6e275..9ac162d 100644 --- a/docker-compose.yaml +++ b/docker-compose.yaml @@ -15,14 +15,18 @@ services: # ── Watchdog Agent ────────────────────────────────────────────────────────── docker-container-watchdog: - image: watchdog:latest # pre-built image — does not require internet or build + # Pinned to the release in WATCHDOG_VERSION (written to .env by install.sh) + # so an upgrade cannot keep running a stale image that still answers to + # :latest. Falls back to :latest for manual, unversioned installs. + image: watchdog:${WATCHDOG_VERSION:-latest} container_name: docker-container-watchdog restart: always # always restart — watchdog must never stay down network_mode: host # must reach all container IPs across all Docker networks env_file: .env environment: WATCHDOG_CONFIG: /etc/watchdog/watchdog-config.yaml - WATCHDOG_HOST: CyberController-Server # identifies this node in alerts + # install.sh writes the operator's answer to .env; this is only the default. + WATCHDOG_HOST: ${WATCHDOG_HOST:-CyberController-Server} volumes: # Read Docker socket to monitor all containers on the host - /var/run/docker.sock:/var/run/docker.sock:ro diff --git a/install.sh b/install.sh index 8423c61..4f161a0 100644 --- a/install.sh +++ b/install.sh @@ -2,7 +2,6 @@ ################################################################################ # CyberController Container Watchdog - Installation Script -# Version: 1.0.0 # # Installs the watchdog Docker container that monitors all containers on the # host and fires alerts (Slack, SMTP, SNMP) on crashes, OOM-kills, @@ -30,8 +29,21 @@ SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" INSTALL_DIR="$SCRIPT_DIR" # package is already extracted to the install directory IMAGE_ARCHIVE="${SCRIPT_DIR}/watchdog.tar" CONTAINER_NAME="docker-container-watchdog" -IMAGE_NAME="watchdog:latest" -VERSION="1.0.0" +VERSION_FILE="${SCRIPT_DIR}/VERSION" + +# Single source of truth for the release: the packaged VERSION file. The image +# is pinned to it so an upgrade can never silently keep serving the previous +# build that also happens to be tagged :latest. +VERSION="$(tr -d '[:space:]' < "$VERSION_FILE" 2>/dev/null || true)" +if [ -z "$VERSION" ]; then + echo "ERROR: VERSION file missing or empty: ${VERSION_FILE}" >&2 + echo " Re-extract the installation package and try again." >&2 + exit 1 +fi +IMAGE_NAME="watchdog:${VERSION}" +IMAGE_LATEST="watchdog:latest" # convenience tag only — never the deployed reference +RUN_UID="1000" # non-root runtime user baked in by this installer +RUN_GID="1000" DC="" # compose command — set automatically in check_prerequisites ################################################################################ @@ -57,6 +69,155 @@ print_section() { echo "" } +################################################################################ +# Privileged File Helpers +# +# Config files left behind by an older, root-run installation are commonly +# 600 root:root. A docker-group user re-running this installer must still be +# able to migrate them, so every write goes through these helpers: attempt +# unprivileged first, escalate to sudo, and fail loudly instead of leaving a +# half-migrated file behind. +################################################################################ + +# Prints the privilege prefix to use ("sudo" or nothing); non-zero if elevation +# is needed but unavailable. +_priv_prefix() { + [ "$EUID" -eq 0 ] && return 0 + command -v sudo &>/dev/null && { printf 'sudo'; return 0; } + return 1 +} + +# Usage: _run_priv — run unprivileged, retry with sudo on failure. +_run_priv() { + "$@" 2>/dev/null && return 0 + local priv + priv=$(_priv_prefix) || return 1 + [ -z "$priv" ] && return 1 # already root: retrying changes nothing + $priv "$@" +} + +# Usage: _publish_file SRC DST MODE [OWNER:GROUP] +# Replaces DST with SRC's contents, then applies MODE/OWNER. +_publish_file() { + local src="$1" dst="$2" mode="$3" owner="${4:-}" + if ! _run_priv cp "$src" "$dst"; then + print_error "Failed to write ${dst}" + echo "" + echo " The file is not writable by $(id -un) and sudo is unavailable." + echo " Re-run this installer as root." + return 1 + fi + if [ -n "$owner" ] && ! _run_priv chown "$owner" "$dst"; then + print_error "Failed to set ownership ${owner} on ${dst}" + return 1 + fi + if ! _run_priv chmod "$mode" "$dst"; then + print_error "Failed to set permissions ${mode} on ${dst}" + return 1 + fi + return 0 +} + +# Usage: _read_file_priv FILE — cat a file, escalating if it is not readable. +_read_file_priv() { + local file="$1" + [ -e "$file" ] || return 1 # never prompt for sudo just to miss a file + [ -r "$file" ] && { cat "$file"; return $?; } + local priv + priv=$(_priv_prefix) || return 1 + [ -z "$priv" ] && return 1 + $priv cat "$file" +} + +# Usage: update_env_vars FILE KEY=VALUE... +# Upserts each KEY=VALUE, preserving the file's existing owner, then verifies +# every value really landed on disk before returning success. +update_env_vars() { + local file="$1"; shift + local tmp tmp2 kv key rc=0 + + tmp=$(mktemp) || { print_error "Could not create a temporary file"; return 1; } + tmp2=$(mktemp) || { rm -f "$tmp"; print_error "Could not create a temporary file"; return 1; } + chmod 600 "$tmp" "$tmp2" + + if ! _read_file_priv "$file" > "$tmp"; then + print_error "Cannot read ${file} — not readable by $(id -un) and sudo is unavailable" + rm -f "$tmp" "$tmp2" + return 1 + fi + + for kv in "$@"; do + key="${kv%%=*}" + if grep -q -e "^${key}=" "$tmp"; then + # index() instead of sed: values may contain / & \ and other + # characters that would be interpreted in a substitution. + awk -v key="$key" -v line="$kv" \ + 'index($0, key "=") == 1 { print line; next } { print }' "$tmp" > "$tmp2" \ + && cat "$tmp2" > "$tmp" + else + printf '%s\n' "$kv" >> "$tmp" + fi + done + + if ! _publish_file "$tmp" "$file" 600; then + rm -f "$tmp" "$tmp2" + return 1 + fi + + # Read back: a successful cp is not proof the values are present, and + # silently continuing here is what left upgrades running as root before. + if ! _read_file_priv "$file" > "$tmp2"; then + print_error "Cannot verify ${file} after writing it" + rm -f "$tmp" "$tmp2" + return 1 + fi + for kv in "$@"; do + if ! grep -qxF "$kv" "$tmp2"; then + print_error "Failed to save ${kv%%=*} in ${file}" + rc=1 + fi + done + rm -f "$tmp" "$tmp2" + + if [ "$rc" -ne 0 ]; then + echo "" + echo " Add the following lines to ${file} manually and re-run:" + printf ' %s\n' "$@" + return 1 + fi + + # Compose reads .env as the user running it — that is this user, right now. + if [ ! -r "$file" ]; then + print_warning "${file} is not readable by $(id -un) — handing it to this user" + if ! _run_priv chown "$(id -u):$(id -g)" "$file"; then + print_error "Could not make ${file} readable by $(id -un) — docker compose will fail" + return 1 + fi + fi + return 0 +} + +# watchdog-config.yaml is bind-mounted read-only into a container running as +# RUN_UID:RUN_GID. A config left 600 root:root by an older install makes the +# agent exit on startup, so normalise ownership/mode on every run. +harden_config_permissions() { + local file="$1" + [ -f "$file" ] || return 0 + if ! _run_priv chown "${RUN_UID}:${RUN_GID}" "$file"; then + print_error "Failed to set ownership of ${file} to UID ${RUN_UID}:GID ${RUN_GID}" + echo "" + echo " The container runs as ${RUN_UID}:${RUN_GID} and could not read it. Fix manually:" + echo " sudo chown ${RUN_UID}:${RUN_GID} ${file} && sudo chmod 640 ${file}" + return 1 + fi + if ! _run_priv chmod 640 "$file"; then + print_error "Failed to set permissions 640 on ${file}" + return 1 + fi + print_success "Config readable by the container runtime user (${RUN_UID}:${RUN_GID}, mode 640)" + return 0 +} + ################################################################################ # Pre-flight Checks ################################################################################ @@ -120,28 +281,15 @@ setup_log_directory() { fi # This installer hardens the container to run as UID/GID 1000 (see # detect_docker_gid below) instead of the compose file's root default, so - # it must own the dir and any pre-existing files up front. Elevate with - # sudo when we are not already root. - local PRIV="" - if [ "$EUID" -ne 0 ]; then - if command -v sudo &>/dev/null; then - PRIV="sudo" - else - print_error "Not running as root and sudo is not available — cannot set ${LOG_DIR} ownership to UID 1000:GID 1000" - echo "" - echo " Re-run this script as root, or fix ownership manually before starting:" - echo " chown -R 1000:1000 ${LOG_DIR} && chmod 750 ${LOG_DIR}" - exit 1 - fi - fi - if ! $PRIV chown -R 1000:1000 "$LOG_DIR"; then - print_error "Failed to set ownership of ${LOG_DIR} to UID 1000:GID 1000" + # it must own the dir and any pre-existing files up front. + if ! _run_priv chown -R "${RUN_UID}:${RUN_GID}" "$LOG_DIR"; then + print_error "Failed to set ownership of ${LOG_DIR} to UID ${RUN_UID}:GID ${RUN_GID}" echo "" echo " Fix ownership manually before starting:" - echo " sudo chown -R 1000:1000 ${LOG_DIR} && sudo chmod 750 ${LOG_DIR}" + echo " sudo chown -R ${RUN_UID}:${RUN_GID} ${LOG_DIR} && sudo chmod 750 ${LOG_DIR}" exit 1 fi - if ! $PRIV chmod 750 "$LOG_DIR"; then + if ! _run_priv chmod 750 "$LOG_DIR"; then print_error "Failed to set permissions on ${LOG_DIR}" exit 1 fi @@ -177,40 +325,67 @@ detect_docker_gid() { load_docker_image() { print_section "Loading Docker Image" - # Check if image is already present - if docker image inspect "${IMAGE_NAME}" &>/dev/null; then - print_info "Image ${IMAGE_NAME} already exists locally — using existing image" - # watchdog.py is baked into the image, not bind-mounted, so a stale image - # keeps running old code even after the host copy is updated. - print_warning "Agent code runs from the image, not ${INSTALL_DIR}/watchdog.py" - print_info " To pick up code changes: docker rmi ${IMAGE_NAME} && bash install.sh" - return 0 - fi + # Always load the packaged archive first. The previous behaviour — skipping + # the load whenever any watchdog image already existed — meant an upgrade + # kept running the old code, because watchdog.py ships inside the image. + if [ -f "$IMAGE_ARCHIVE" ]; then + print_info "Loading image from archive: $(basename "$IMAGE_ARCHIVE")" + local load_output + if ! load_output=$(docker load -i "$IMAGE_ARCHIVE" 2>&1); then + print_error "Failed to load ${IMAGE_ARCHIVE}" + echo "$load_output" + exit 1 + fi + echo "$load_output" - if [ ! -f "$IMAGE_ARCHIVE" ]; then + # The archive must carry the release tag. Promoting whatever it happened + # to contain would silently relabel an older build as this version, and + # the post-install image check would then confirm the wrong release. + if ! docker image inspect "$IMAGE_NAME" &>/dev/null; then + print_error "${IMAGE_ARCHIVE} does not contain ${IMAGE_NAME}" + echo "" + echo " The archive loaded:" + printf '%s\n' "$load_output" \ + | sed -n -e 's/^Loaded image: */ /p' -e 's/^Loaded image ID: */ /p' + echo "" + echo " This package is version ${VERSION} (from ${VERSION_FILE})." + echo " Obtain a watchdog.tar built for ${VERSION}, or rebuild it:" + echo " docker build -t ${IMAGE_NAME} ." + echo " docker tag ${IMAGE_NAME} ${IMAGE_LATEST}" + echo " docker save ${IMAGE_NAME} ${IMAGE_LATEST} -o watchdog.tar" + exit 1 + fi + else print_warning "Image archive not found: ${IMAGE_ARCHIVE}" - echo "" + + # No archive — build from source if the Dockerfile shipped. if [ -f "${INSTALL_DIR}/Dockerfile" ]; then - print_info "Building image from source — this may take a few minutes..." + print_info "Building ${IMAGE_NAME} from source — this may take a few minutes..." # Use a clean temp dir so any .dockerignore in INSTALL_DIR is bypassed local tmpdir tmpdir=$(mktemp -d) trap "rm -rf '${tmpdir}'" EXIT cp "${INSTALL_DIR}/Dockerfile" "${tmpdir}/" cp "${INSTALL_DIR}/watchdog.py" "${tmpdir}/" - docker build -t "${IMAGE_NAME}" "${tmpdir}" + if ! docker build -t "${IMAGE_NAME}" "${tmpdir}"; then + print_error "Build failed — cannot produce ${IMAGE_NAME}" + exit 1 + fi print_success "Image built: ${IMAGE_NAME}" - return 0 + else + print_error "Cannot proceed: ${IMAGE_NAME} is not available" + echo "" + echo " watchdog.tar not found and Dockerfile not found." + echo " Copy watchdog.tar into: $(dirname "$IMAGE_ARCHIVE") and re-run." + exit 1 fi - print_error "Cannot proceed: watchdog.tar not found and Dockerfile not found" - echo "" - echo " Copy watchdog.tar into: $(dirname "$IMAGE_ARCHIVE") and re-run." - exit 1 fi - print_info "Loading image from archive: $(basename "$IMAGE_ARCHIVE")" - docker load -i "$IMAGE_ARCHIVE" - print_success "Image loaded: ${IMAGE_NAME}" + # Convenience alias only — docker-compose.yaml deploys ${IMAGE_NAME}. + docker tag "$IMAGE_NAME" "$IMAGE_LATEST" &>/dev/null || \ + print_warning "Could not update the ${IMAGE_LATEST} convenience tag" + + print_success "Image ready: ${IMAGE_NAME}" } ################################################################################ @@ -262,12 +437,20 @@ configure_all() { ENV_FILE="${INSTALL_DIR}/.env" CONFIG_FILE="${INSTALL_DIR}/watchdog-config.yaml" + # Keeping a half-present configuration breaks the deployment: a missing .env + # aborts compose on env_file, and a missing watchdog-config.yaml makes Docker + # create a directory at the bind-mount path. Only offer to keep when both exist. + if { [ -f "$ENV_FILE" ] || [ -f "$CONFIG_FILE" ]; } && \ + { [ ! -f "$ENV_FILE" ] || [ ! -f "$CONFIG_FILE" ]; }; then + local missing="" + [ -f "$ENV_FILE" ] || missing+=" .env" + [ -f "$CONFIG_FILE" ] || missing+=" watchdog-config.yaml" + print_warning "Incomplete configuration — missing:${missing}" + print_warning "The wizard regenerates both files — re-enter your settings when prompted" + echo "" # Offer to reconfigure if files already exist - if [ -f "$ENV_FILE" ] || [ -f "$CONFIG_FILE" ]; then - local existing="" - [ -f "$ENV_FILE" ] && existing+=" .env" - [ -f "$CONFIG_FILE" ] && existing+=" watchdog-config.yaml" - print_info "Existing configuration detected:${existing}" + elif [ -f "$ENV_FILE" ] && [ -f "$CONFIG_FILE" ]; then + print_info "Existing configuration detected: .env watchdog-config.yaml" printf " Reconfigure from scratch? [y/N]: " local RECONFIG IFS= read -r RECONFIG @@ -276,24 +459,29 @@ configure_all() { # Upgrade path: installs from before non-root hardening existed # predate these vars, which would silently fall back to the # compose file's root default instead of the intended hardening. + # WATCHDOG_VERSION is refreshed on every run so the container is + # redeployed on the release that was just loaded. if [ -f "$ENV_FILE" ]; then - local CURRENT_GID - CURRENT_GID=$(grep -E '^DOCKER_GID=' "$ENV_FILE" | tail -n1 | cut -d= -f2 | tr -d '[:space:]') - if ! [[ "$CURRENT_GID" =~ ^[0-9]+$ ]]; then + local CURRENT_GID MIGRATED_GID + CURRENT_GID=$(_read_file_priv "$ENV_FILE" | grep -E '^DOCKER_GID=' | tail -n1 | cut -d= -f2 | tr -d '[:space:]') + if [[ "$CURRENT_GID" =~ ^[0-9]+$ ]]; then + MIGRATED_GID="$CURRENT_GID" + else print_warning "Existing .env predates non-root hardening — detecting and adding it" - local MIGRATED_GID MIGRATED_GID=$(detect_docker_gid) || exit 1 - for kv in "WATCHDOG_UID=1000" "WATCHDOG_GID=1000" "DOCKER_GID=${MIGRATED_GID}"; do - local key="${kv%%=*}" - if grep -qE "^${key}=" "$ENV_FILE"; then - sed -i.bak -E "s|^${key}=.*|${kv}|" "$ENV_FILE" && rm -f "${ENV_FILE}.bak" - else - printf "%s\n" "$kv" >> "$ENV_FILE" - fi - done - print_success "Set WATCHDOG_UID=1000, WATCHDOG_GID=1000, DOCKER_GID=${MIGRATED_GID} in ${ENV_FILE}" fi + if ! update_env_vars "$ENV_FILE" \ + "WATCHDOG_UID=${RUN_UID}" \ + "WATCHDOG_GID=${RUN_GID}" \ + "DOCKER_GID=${MIGRATED_GID}" \ + "WATCHDOG_VERSION=${VERSION}"; then + exit 1 + fi + print_success "Set WATCHDOG_UID=${RUN_UID}, WATCHDOG_GID=${RUN_GID}, DOCKER_GID=${MIGRATED_GID}, WATCHDOG_VERSION=${VERSION} in ${ENV_FILE}" fi + # An older install may have left the config root-owned and 600, + # which the non-root container cannot read. + harden_config_permissions "$CONFIG_FILE" || exit 1 return 0 fi echo "" @@ -409,11 +597,16 @@ configure_all() { echo "" # ── Non-root hardening (manual `docker compose up` defaults to root instead) ── - local WATCHDOG_UID="1000" WATCHDOG_GID="1000" DOCKER_GID + local WATCHDOG_UID="$RUN_UID" WATCHDOG_GID="$RUN_GID" DOCKER_GID DOCKER_GID=$(detect_docker_gid) || exit 1 print_info "Docker socket GID detected: ${DOCKER_GID} — running non-root as UID/GID ${WATCHDOG_UID}" # ── Write .env ──────────────────────────────────────────────────────────── + # Staged in a temp file so a root-owned .env from a previous install can + # still be replaced (via sudo) instead of failing on an unwritable target. + local ENV_TMP + ENV_TMP=$(mktemp) || { print_error "Could not create a temporary file"; exit 1; } + chmod 600 "$ENV_TMP" { printf "# .env — generated by install.sh on %s\n" "$(date '+%Y-%m-%d %H:%M:%S')" printf "# Secrets only. All other settings are in watchdog-config.yaml.\n\n" @@ -430,9 +623,15 @@ configure_all() { printf "# Alert identity\nWATCHDOG_HOST=%s\n\n" "$WATCHDOG_HOST" printf "# Non-root hardening\nWATCHDOG_UID=%s\nWATCHDOG_GID=%s\nDOCKER_GID=%s\n\n" \ "$WATCHDOG_UID" "$WATCHDOG_GID" "$DOCKER_GID" + printf "# Deployed release — docker-compose.yaml pins watchdog:\${WATCHDOG_VERSION}\n" + printf "WATCHDOG_VERSION=%s\n\n" "$VERSION" printf "# Tuning\nLOG_LEVEL=INFO\n" - } > "$ENV_FILE" - chmod 600 "$ENV_FILE" + } > "$ENV_TMP" + if ! _publish_file "$ENV_TMP" "$ENV_FILE" 600 "$(id -u):$(id -g)"; then + rm -f "$ENV_TMP" + exit 1 + fi + rm -f "$ENV_TMP" print_success ".env written: ${ENV_FILE}" # ── Build alert_channels YAML list ──────────────────────────────────────── @@ -451,6 +650,8 @@ configure_all() { [ -z "$RECIPIENTS_YAML" ] && RECIPIENTS_YAML=" - ops-team@radware.com"$'\n' # ── Write watchdog-config.yaml ──────────────────────────────────────────── + local CONFIG_TMP + CONFIG_TMP=$(mktemp) || { print_error "Could not create a temporary file"; exit 1; } { printf "# watchdog-config.yaml — generated by install.sh on %s\n" "$(date '+%Y-%m-%d %H:%M:%S')" printf "# Edit values, then restart: %s -f %s/docker-compose.yaml restart\n\n" \ @@ -473,6 +674,8 @@ configure_all() { printf "\nrunbook_base_url: \"\"\n" printf "\nlog_level: INFO\n" printf "log_file: /var/log/watchdog/watchdog.log\n" + printf "state_file: /var/log/watchdog/state.json\n" + printf "boot_grace_seconds: 180\n" printf "\nsyslog:\n" printf " enabled: %s\n" "$SYSLOG_ENABLED" printf " host: %s\n" "${SYSLOG_HOST:-127.0.0.1}" @@ -516,7 +719,14 @@ configure_all() { printf " - /\n" printf " timeout_seconds: 2\n" printf "\ncontainer_health_checks: {}\n" - } > "$CONFIG_FILE" + } > "$CONFIG_TMP" + # Owned by the container's runtime user: it is bind-mounted read-only and + # the agent cannot start if it is unreadable at UID/GID 1000. + if ! _publish_file "$CONFIG_TMP" "$CONFIG_FILE" 640 "${RUN_UID}:${RUN_GID}"; then + rm -f "$CONFIG_TMP" + exit 1 + fi + rm -f "$CONFIG_TMP" print_success "watchdog-config.yaml written: ${CONFIG_FILE}" } @@ -527,16 +737,28 @@ configure_all() { start_watchdog() { print_section "Starting Watchdog" - cd "${INSTALL_DIR}" + if ! cd "${INSTALL_DIR}"; then + print_error "Cannot enter installation directory: ${INSTALL_DIR}" + exit 1 + fi # Stop and remove existing container if present if docker ps -a --format '{{.Names}}' | grep -q "^${CONTAINER_NAME}$"; then print_info "Stopping existing container..." - $DC down + if ! $DC down; then + print_error "Failed to stop the existing container — aborting before redeploy" + exit 1 + fi fi print_info "Starting container with docker compose..." - $DC up -d + if ! $DC up -d; then + print_error "docker compose up failed — the watchdog was not deployed" + echo "" + echo " Inspect the failure with:" + echo " $DC -f ${INSTALL_DIR}/docker-compose.yaml logs" + exit 1 + fi print_success "Watchdog container started" } @@ -553,8 +775,26 @@ verify_deployment() { if [ -n "$STATUS" ]; then print_success "Container is running: ${STATUS}" else - print_warning "Container does not appear to be running — check logs for errors:" - print_info " $DC logs watchdog" + print_error "Container ${CONTAINER_NAME} is not running — the installation did not succeed" + echo "" + echo " Check the logs for the reason:" + echo " $DC -f ${INSTALL_DIR}/docker-compose.yaml logs" + echo " docker logs ${CONTAINER_NAME}" + exit 1 + fi + + # Confirm the deployed release, not just that something is up: a stale + # image left over from a previous version would otherwise look healthy. + local RUNNING_IMAGE + RUNNING_IMAGE=$(docker inspect --format '{{.Config.Image}}' "$CONTAINER_NAME" 2>/dev/null || echo "") + if [ "$RUNNING_IMAGE" = "$IMAGE_NAME" ]; then + print_success "Container is running the expected image: ${IMAGE_NAME}" + else + print_error "Container is running '${RUNNING_IMAGE:-unknown}', expected '${IMAGE_NAME}'" + echo "" + echo " Check WATCHDOG_VERSION in ${INSTALL_DIR}/.env, then redeploy:" + echo " $DC -f ${INSTALL_DIR}/docker-compose.yaml up -d --force-recreate" + exit 1 fi } diff --git a/uninstall.sh b/uninstall.sh index 8460174..7e8d1d2 100644 --- a/uninstall.sh +++ b/uninstall.sh @@ -2,7 +2,6 @@ ################################################################################ # CyberController Container Watchdog - Uninstallation Script -# Version: 1.0.0 # # Stops and removes the watchdog Docker container and image. # Log files are preserved by default. @@ -11,7 +10,7 @@ # sudo bash uninstall.sh [OPTIONS] # # Options: -# --keep-image Preserve the watchdog:latest Docker image (skip image removal) +# --keep-image Preserve the watchdog Docker image (skip image removal) # --keep-logs Preserve log files (default behaviour) # --remove-all Full removal: container + image + log files # --force Skip confirmation prompts @@ -35,8 +34,16 @@ fi SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" INSTALL_DIR="$SCRIPT_DIR" CONTAINER_NAME="docker-container-watchdog" -IMAGE_NAME="watchdog:latest" -VERSION="1.0.0" + +# Prefer the release actually deployed (.env), then the packaged VERSION file. +VERSION="$(grep -E '^WATCHDOG_VERSION=' "${SCRIPT_DIR}/.env" 2>/dev/null | tail -n1 | cut -d= -f2 | tr -d '[:space:]')" +[ -z "$VERSION" ] && VERSION="$(tr -d '[:space:]' < "${SCRIPT_DIR}/VERSION" 2>/dev/null || true)" +[ -z "$VERSION" ] && VERSION="latest" +IMAGE_NAME="watchdog:${VERSION}" +IMAGE_LATEST="watchdog:latest" +# Both tags usually point at the same image ID; removing only one just untags it. +IMAGE_TAGS=("$IMAGE_NAME") +[ "$IMAGE_NAME" != "$IMAGE_LATEST" ] && IMAGE_TAGS+=("$IMAGE_LATEST") DC="" # compose command — detected at runtime # ── Flags ───────────────────────────────────────────────────────────────────── @@ -80,7 +87,7 @@ usage() { echo "Uninstall the CyberController Container Watchdog." echo "" echo "OPTIONS:" - echo " --keep-image Preserve the watchdog:latest Docker image (skip image removal)" + echo " --keep-image Preserve the watchdog Docker image (skip image removal)" echo " --keep-logs Remove container and image; preserve log files" echo " --remove-all Full removal: container + image + log directory" echo " --force Skip confirmation prompts" @@ -196,13 +203,17 @@ plan_and_confirm() { echo -e " ${GREEN}○${NC} Container : ${CONTAINER_NAME} — not found" fi - if docker image inspect "${IMAGE_NAME}" &>/dev/null; then + FOUND_TAG="" + for TAG in "${IMAGE_TAGS[@]}"; do + docker image inspect "${TAG}" &>/dev/null && { FOUND_TAG="$TAG"; break; } + done + if [ -n "$FOUND_TAG" ]; then IMAGE_EXISTS=true - ISIZE=$(docker image inspect "${IMAGE_NAME}" --format '{{.Size}}' | \ + ISIZE=$(docker image inspect "${FOUND_TAG}" --format '{{.Size}}' | \ awk '{printf "%.0f MB", $1/1024/1024}') - echo -e " ${YELLOW}?${NC} Image : ${IMAGE_NAME} (~${ISIZE})" + echo -e " ${YELLOW}?${NC} Image : ${IMAGE_TAGS[*]} (~${ISIZE})" else - echo -e " ${GREEN}○${NC} Image : ${IMAGE_NAME} — not found" + echo -e " ${GREEN}○${NC} Image : ${IMAGE_TAGS[*]} — not found" fi fi @@ -240,7 +251,7 @@ plan_and_confirm() { # Interactive prompts — only for items not already decided by flags if [ "$FORCE" = false ]; then if [ "$IMAGE_EXISTS" = true ] && [ "$KEEP_IMAGE" = false ] && [ "$REMOVE_ALL" = false ]; then - read -p " Remove Docker image (${IMAGE_NAME}, ~${ISIZE})? [Y/n]: " RI + read -p " Remove Docker image (${IMAGE_TAGS[*]}, ~${ISIZE})? [Y/n]: " RI [[ "${RI:-y}" =~ ^[Nn] ]] || REMOVE_IMAGE_CONFIRMED=true fi @@ -324,14 +335,14 @@ remove_docker_image() { return 0 fi - if ! docker image inspect "${IMAGE_NAME}" &>/dev/null; then - print_info "Docker image not found: ${IMAGE_NAME}" + if [ "$IMAGE_EXISTS" != true ]; then + print_info "Docker image not found: ${IMAGE_TAGS[*]}" return 0 fi if [ "$KEEP_IMAGE" = true ]; then IMAGE_KEPT=true - print_success "Keeping Docker image (--keep-image): ${IMAGE_NAME}" + print_success "Keeping Docker image (--keep-image): ${IMAGE_TAGS[*]}" return 0 fi @@ -339,19 +350,20 @@ remove_docker_image() { _do_remove_image else IMAGE_KEPT=true - print_success "Docker image preserved: ${IMAGE_NAME}" - print_info "Remove manually with: docker rmi ${IMAGE_NAME}" + print_success "Docker image preserved: ${IMAGE_TAGS[*]}" + print_info "Remove manually with: docker rmi ${IMAGE_TAGS[*]}" fi } # Internal helper — stops/removes ALL containers that reference the image, then removes the image. _do_remove_image() { - print_info "Removing Docker image: ${IMAGE_NAME}..." + print_info "Removing Docker image: ${IMAGE_TAGS[*]}..." # Stop and remove every container (running or stopped) that uses this image. # We iterate rather than rely on --filter ancestor because Docker 20.10 combos are unreliable. - ALL_CTRS=$(docker ps -aq --filter "ancestor=${IMAGE_NAME}" 2>/dev/null || true) - if [ -n "$ALL_CTRS" ]; then + for TAG in "${IMAGE_TAGS[@]}"; do + ALL_CTRS=$(docker ps -aq --filter "ancestor=${TAG}" 2>/dev/null || true) + [ -n "$ALL_CTRS" ] || continue for CID in $ALL_CTRS; do CSTATUS=$(docker inspect --format '{{.State.Status}}' "$CID" 2>/dev/null || true) CNAME=$(docker inspect --format '{{.Name}}' "$CID" 2>/dev/null | sed 's|^/||' || true) @@ -362,13 +374,17 @@ _do_remove_image() { print_info " Removing container ${CNAME:-$CID}..." docker rm "$CID" 2>/dev/null || true done - fi + done - if docker rmi "${IMAGE_NAME}" 2>&1; then - print_success "Docker image removed: ${IMAGE_NAME}" - else - print_warning "Could not remove image — check for containers using it: docker ps -a --filter ancestor=${IMAGE_NAME}" - fi + # Remove every tag — leaving the :latest alias behind keeps the layers on disk. + for TAG in "${IMAGE_TAGS[@]}"; do + docker image inspect "${TAG}" &>/dev/null || continue + if docker rmi "${TAG}" 2>&1; then + print_success "Docker image removed: ${TAG}" + else + print_warning "Could not remove image — check for containers using it: docker ps -a --filter ancestor=${TAG}" + fi + done } remove_logs() { @@ -416,16 +432,20 @@ display_completion_summary() { echo -e " ${GREEN}○${NC} Container : ${CONTAINER_NAME} — not found (nothing to remove)" fi - if docker image inspect "${IMAGE_NAME}" &>/dev/null; then + REMAINING_TAG="" + for TAG in "${IMAGE_TAGS[@]}"; do + docker image inspect "${TAG}" &>/dev/null && { REMAINING_TAG="$TAG"; break; } + done + if [ -n "$REMAINING_TAG" ]; then if [ "$IMAGE_KEPT" = true ]; then - echo -e " ${GREEN}✓${NC} Image : ${IMAGE_NAME} — ${GREEN}PRESERVED${NC}" + echo -e " ${GREEN}✓${NC} Image : ${REMAINING_TAG} — ${GREEN}PRESERVED${NC}" else - echo -e " ${YELLOW}⚠${NC} Image : ${IMAGE_NAME} — ${YELLOW}STILL EXISTS (removal may have failed)${NC}" + echo -e " ${YELLOW}⚠${NC} Image : ${REMAINING_TAG} — ${YELLOW}STILL EXISTS (removal may have failed)${NC}" fi elif [ "$IMAGE_EXISTS" = true ]; then - echo -e " ${GREEN}✓${NC} Image : ${IMAGE_NAME} — removed" + echo -e " ${GREEN}✓${NC} Image : ${IMAGE_TAGS[*]} — removed" else - echo -e " ${GREEN}○${NC} Image : ${IMAGE_NAME} — not found (nothing to remove)" + echo -e " ${GREEN}○${NC} Image : ${IMAGE_TAGS[*]} — not found (nothing to remove)" fi fi diff --git a/watchdog-config.yaml.example b/watchdog-config.yaml.example index be1da9a..ad27b58 100644 --- a/watchdog-config.yaml.example +++ b/watchdog-config.yaml.example @@ -32,6 +32,15 @@ excluded_containers: [] log_level: DEBUG # DEBUG | INFO | WARNING | ERROR log_file: /var/log/watchdog/watchdog.log # bind-mounted to ./watchdog/watchdog.log on host +# Alert state is persisted here so recovery alerts still fire after the host or +# the watchdog restarts. Set to "" to disable persistence. +state_file: /var/log/watchdog/state.json +# After a detected host reboot, hold alerts this long while services start, then +# send one reboot-summary instead of a burst. Set to 0 to alert immediately. +# A detected reboot always produces exactly one summary, even when nothing failed, +# so a clean reboot is confirmed rather than passing silently. +boot_grace_seconds: 180 + # ── Syslog (remote / central log server) ───────────────────────────────────── syslog: enabled: true # set to true to enable diff --git a/watchdog.py b/watchdog.py index 4c08246..8fc6f2f 100644 --- a/watchdog.py +++ b/watchdog.py @@ -9,6 +9,7 @@ """ import argparse +import json import logging import logging.handlers import os @@ -50,20 +51,25 @@ def setup_logging(level: str, log_file: str | None, syslog_cfg: dict | None = No proto = syslog_cfg.get("protocol", "udp").lower() socktype = socket.SOCK_DGRAM if proto == "udp" else socket.SOCK_STREAM facility_name = syslog_cfg.get("facility", "local0").upper() - facility = getattr( - logging.handlers.SysLogHandler, - f"LOG_{facility_name}", - logging.handlers.SysLogHandler.LOG_LOCAL0, - ) - syslog_handler = logging.handlers.SysLogHandler( - address=(host, port), - facility=facility, - socktype=socktype, - ) - syslog_handler.setFormatter(logging.Formatter(SYSLOG_FORMAT)) - handlers.append(syslog_handler) - # Use a plain formatter for other handlers, syslog gets its own - log.info("Syslog handler added: %s:%d (%s)", host, port, proto) + facility = getattr(logging.handlers.SysLogHandler, f"LOG_{facility_name}", None) + if facility is None: + log.warning("Unknown syslog facility %r — falling back to local0", facility_name) + facility = logging.handlers.SysLogHandler.LOG_LOCAL0 + # SysLogHandler resolves (and for TCP connects to) the destination in its + # constructor. An unreachable log sink must not stop container monitoring. + try: + syslog_handler = logging.handlers.SysLogHandler( + address=(host, port), + facility=facility, + socktype=socktype, + ) + except Exception as exc: + log.warning("Syslog disabled — cannot reach %s:%d (%s): %s", host, port, proto, exc) + else: + syslog_handler.setFormatter(logging.Formatter(SYSLOG_FORMAT)) + handlers.append(syslog_handler) + # Use a plain formatter for other handlers, syslog gets its own + log.info("Syslog handler added: %s:%d (%s)", host, port, proto) logging.basicConfig(level=numeric, format=LOG_FORMAT, handlers=handlers, force=True) # Suppress noisy third-party debug chatter (urllib3 Docker socket calls, etc.) for noisy in ("urllib3", "urllib3.connectionpool", "docker", "requests"): @@ -93,6 +99,10 @@ def setup_logging(level: str, log_file: str | None, syslog_cfg: dict | None = No ], "log_file": "/var/log/watchdog/watchdog.log", "log_level": "INFO", + # Alert state survives restarts so recovery alerts still fire after a reboot. + "state_file": "/var/log/watchdog/state.json", + # After a host reboot, hold alerts this long and emit one summary instead. + "boot_grace_seconds": 180, "syslog": { "enabled": False, "host": "localhost", @@ -127,21 +137,40 @@ def load_config(path: str) -> dict: # ── Container state tracking ────────────────────────────────────────────────── +_STATE_VERSION = 1 + + +def _host_boot_time() -> int: + """Host boot timestamp from /proc/stat btime — not namespaced, so this is the host's.""" + try: + with open("/proc/stat") as f: + for line in f: + if line.startswith("btime "): + return int(line.split()[1]) + except Exception as exc: + log.debug("Could not read host boot time: %s", exc) + return 0 + + class ContainerState: - __slots__ = ("name", "unhealthy_cycles", "restart_times", "last_alert_time", "alerted_for") + __slots__ = ("name", "lock", "unhealthy_cycles", "restart_times", "last_alert_time", + "alerted_for", "last_restart_count") def __init__(self, name: str): self.name = name + self.lock = threading.Lock() # event thread and poll thread both mutate this self.unhealthy_cycles: int = 0 self.restart_times: list[float] = [] self.last_alert_time: float = 0.0 self.alerted_for: str = "" + self.last_restart_count: int = -1 # -1 = no baseline taken yet class WatchdogState: - def __init__(self): + def __init__(self, state_file: str = ""): self._lock = threading.Lock() self._states: dict[str, ContainerState] = {} + self._state_file = state_file def get(self, name: str) -> ContainerState: with self._lock: @@ -149,14 +178,65 @@ def get(self, name: str) -> ContainerState: self._states[name] = ContainerState(name) return self._states[name] - def cleanup_stale(self, active_names: set) -> None: + def cleanup_stale(self, active_names: set, retain_open: bool = False) -> None: with self._lock: - stale = [n for n in list(self._states) if n not in active_names] + stale = [ + n for n, st in list(self._states.items()) + if n not in active_names and not (retain_open and st.alerted_for) + ] for n in stale: self._states.pop(n, None) for n in stale: log.debug("Removing stale state for container: %s", n) + # ── Persistence ─────────────────────────────────────────────────────────── + def load(self) -> tuple[dict, int]: + """Return (persisted alert entries, host boot time recorded at last save).""" + if not self._state_file or not os.path.exists(self._state_file): + return {}, 0 + try: + with open(self._state_file) as f: + data = json.load(f) + if data.get("version") != _STATE_VERSION: + log.warning("State file %s has version %s (expected %d) — ignoring", + self._state_file, data.get("version"), _STATE_VERSION) + return {}, 0 + return data.get("containers") or {}, int(data.get("boot_time") or 0) + except Exception as exc: + # A corrupt state file must never stop monitoring. + log.warning("Could not read state file %s (%s) — starting with empty state", + self._state_file, exc) + return {}, 0 + + def save(self, boot_time: int) -> None: + if not self._state_file: + return + with self._lock: + containers = { + st.name: { + "alerted_for": st.alerted_for, + "last_alert_time": st.last_alert_time, + } + for st in self._states.values() + if st.alerted_for + } + record = { + "version": _STATE_VERSION, + "boot_time": boot_time, + "saved_at": time.time(), + "containers": containers, + } + tmp = f"{self._state_file}.tmp" + try: + parent = os.path.dirname(self._state_file) + if parent: + os.makedirs(parent, exist_ok=True) + with open(tmp, "w") as f: + json.dump(record, f) + os.replace(tmp, self._state_file) # atomic: never leave a half-written file + except Exception as exc: + log.warning("Could not write state file %s: %s", self._state_file, exc) + # ── Alert payload ───────────────────────────────────────────────────────────── def _probe_type_label(detail: str) -> str: @@ -167,6 +247,8 @@ def _probe_type_label(detail: str) -> str: return "TCP connect" if detail.startswith("Docker health probe failed"): return "Docker HEALTHCHECK" + if detail.startswith("Exec probe failed"): + return "Exec command" if detail.startswith("/proc alive check failed"): return "/proc alive check" if detail.startswith("Crash probe failed"): @@ -889,6 +971,10 @@ def _suppress_invalid_state(loop, context): "INFO", "Container is back to normal operation. No action required.", ), + "reboot-summary": ( + "INFO", + "Post-reboot status summary. No action required unless containers are listed as still failing.", + ), } @@ -1116,7 +1202,8 @@ def record_alert(state: ContainerState, failure_type: str) -> None: class Watchdog: def __init__(self, cfg: dict): self.cfg = cfg - self.state = WatchdogState() + self._state_file = cfg.get("state_file") or "" + self.state = WatchdogState(self._state_file) self.host = os.environ.get("WATCHDOG_HOST", socket.gethostname()) self.runbook_base = cfg.get( "runbook_base_url", "https://wiki.radware.internal/runbooks" @@ -1130,22 +1217,137 @@ def __init__(self, cfg: dict): self._discovered_checks: dict[str, dict | None] = {} # auto-discovery cache self._oom_containers: set[str] = set() # suppress die alert when oom already fired self._auto_cfg: dict = cfg.get("auto_health_check", {}) + _manual = cfg.get("container_health_checks") or {} + self._manual_checks: dict = _manual if isinstance(_manual, dict) else {} self._session = requests.Session() # reuse connections for health probes self.client = docker.from_env() self._stop_event = threading.Event() + self._boot_time = _host_boot_time() + self._boot_grace_secs = max(0, int(cfg.get("boot_grace_seconds", 180) or 0)) + self._grace_until: float = 0.0 + self._folded: list[AlertPayload] = [] # alerts held during the post-reboot window + self._restore_state() + + # ── State restore ───────────────────────────────────────────────────────── + def _restore_state(self) -> None: + """Reload alert state so recovery alerts still fire after a watchdog or host restart.""" + persisted, saved_boot_time = self.state.load() + for name, entry in persisted.items(): + if not isinstance(entry, dict): + continue + failure_type = str(entry.get("alerted_for") or "") + if not failure_type: + continue + st = self.state.get(str(name)) + st.alerted_for = failure_type + # unhealthy_cycles and restart_times stay at zero on purpose: they are + # rolling observation windows, and a stale count would alert instantly. + try: + st.last_alert_time = float(entry.get("last_alert_time") or 0.0) + except (TypeError, ValueError): + st.last_alert_time = 0.0 + log.info("Restored open '%s' alert for %s", failure_type, name) + + rebooted = bool(saved_boot_time and self._boot_time and saved_boot_time != self._boot_time) + if rebooted: + # Armed even when the grace window is 0: the window only controls how + # long alerts are held, not whether the reboot is reported. + self._grace_until = time.time() + self._boot_grace_secs + log.info( + "Host rebooted since last run (boot time %d → %d) — holding alerts " + "for %ds and reporting one summary", + saved_boot_time, self._boot_time, self._boot_grace_secs, + ) + elif persisted: + log.info("Watchdog restarted without a host reboot — resuming %d open alert(s)", + len(persisted)) + + # ── Alert gating ────────────────────────────────────────────────────────── + def _claim_alert(self, state: ContainerState, failure_type: str) -> bool: + """Atomically take the alert slot so both threads cannot fire the same alert.""" + with state.lock: + if not should_alert(state, failure_type, self._cooldown_minutes): + return False + record_alert(state, failure_type) + self.state.save(self._boot_time) + return True + + def _dispatch(self, payload: AlertPayload) -> None: + """Send an alert, or hold it for the summary during the post-reboot window.""" + if self._grace_until and time.time() < self._grace_until: + self._folded.append(payload) + log.info("%s: %s held for post-reboot summary", + payload.container_name, payload.failure_type) + return + dispatch_alert(payload, self.cfg) + + def _container_tally(self) -> tuple[int, int]: + """(running, not-running) counts of monitored containers, for the reboot summary.""" + running = not_running = 0 + try: + for c in self.client.containers.list(all=True): + if c.name in self._excluded: + continue + if c.status == "running": + running += 1 + else: + not_running += 1 + except Exception as exc: + log.debug("Could not tally containers for the reboot summary: %s", exc) + return running, not_running + + def _flush_boot_summary(self) -> None: + """Emit the single post-reboot status alert once the grace window closes.""" + if not self._grace_until or time.time() < self._grace_until: + return + self._grace_until = 0.0 + recovered = sorted({p.container_name for p in self._folded if p.failure_type == "recovered"}) + still_down = sorted({p.container_name for p in self._folded if p.failure_type != "recovered"}) + self._folded.clear() + # Sent even when nothing was held: after a reboot, silence is + # indistinguishable from the watchdog itself having failed to come back. + if not recovered and not still_down: + log.info("Post-reboot grace window closed — no alerts were held") + booted_at = ( + datetime.fromtimestamp(self._boot_time, timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ") + if self._boot_time else "unknown" + ) + running, not_running = self._container_tally() + detail = ( + f"Host booted {booted_at}. " + f"Containers: {running} running, {not_running} not running. " + f"Recovered since last alert: {', '.join(recovered) or 'none'}. " + f"Still failing: {', '.join(still_down) or 'none'}." + ) + payload = AlertPayload( + severity="HIGH" if still_down else "INFO", + container_name="(post-reboot summary)", + container_id="", + host=self.host, + failure_type="reboot-summary", + exit_code=None, + log_tail="", + recommended_action=ACTIONS["reboot-summary"][1], + runbook_url=self.runbook_base, + probe_detail=detail, + ) + dispatch_alert(payload, self.cfg) # ── Recovery alert ───────────────────────────────────────────────────────── def _maybe_alert_recovery(self, container, state: ContainerState) -> None: """Fire a one-time RECOVERED alert when a previously-alarmed container is healthy again.""" if not self.cfg.get("alert_on_recovery", True): return - if not state.alerted_for or state.alerted_for == "recovered": - return + with state.lock: + previous = state.alerted_for + if not previous or previous == "recovered": + return + state.alerted_for = "" + state.last_alert_time = time.time() + self.state.save(self._boot_time) payload = build_payload(container, "recovered", self.host, self.runbook_base, - probe_detail=f"Recovered from '{state.alerted_for}'") - dispatch_alert(payload, self.cfg) - state.alerted_for = "" - state.last_alert_time = time.time() + probe_detail=f"Recovered from '{previous}'") + self._dispatch(payload) # ── Event stream listener ───────────────────────────────────────────────── def _event_listener(self) -> None: @@ -1220,42 +1422,42 @@ def _handle_event(self, event: dict) -> None: log.warning("%s: exit 137 not suppressed — signals: %s", name, _mariadb_helper_signals(container, attributes)) failure_type = "crashed" - if should_alert(state, failure_type, self._cooldown_minutes): + if self._claim_alert(state, failure_type): payload = build_payload(container, failure_type, self.host, self.runbook_base, exit_code, probe_detail=f"Crash probe failed - container exited with exit code {exit_code}") - dispatch_alert(payload, self.cfg) - record_alert(state, failure_type) + self._dispatch(payload) elif action == "oom": failure_type = "oom" self._oom_containers.add(name) # mark so the imminent die event is suppressed - if should_alert(state, failure_type, self._cooldown_minutes): + if self._claim_alert(state, failure_type): payload = build_payload(container, failure_type, self.host, self.runbook_base, extra_context=get_memory_stats(container), probe_detail="OOM probe failed - container was OOM-killed by the kernel") - dispatch_alert(payload, self.cfg) - record_alert(state, failure_type) + self._dispatch(payload) elif action == "health_status: unhealthy": - state.unhealthy_cycles += 1 + with state.lock: + state.unhealthy_cycles += 1 + cycles = state.unhealthy_cycles threshold = self._unhealthy_threshold - log.debug("%s unhealthy cycle %d/%d", name, state.unhealthy_cycles, threshold) - if state.unhealthy_cycles >= threshold: + log.debug("%s unhealthy cycle %d/%d", name, cycles, threshold) + if cycles >= threshold: failure_type = "unhealthy" - if should_alert(state, failure_type, self._cooldown_minutes): + if self._claim_alert(state, failure_type): probe_detail = get_docker_health_log(container) or "Docker health probe failed - health_status: unhealthy event" payload = build_payload(container, failure_type, self.host, self.runbook_base, probe_detail=probe_detail) - dispatch_alert(payload, self.cfg) - record_alert(state, failure_type) + self._dispatch(payload) elif action == "health_status: healthy": - if state.unhealthy_cycles > 0: - log.info("%s recovered — resetting unhealthy counter", name) - state.unhealthy_cycles = 0 + with state.lock: + if state.unhealthy_cycles > 0: + log.info("%s recovered — resetting unhealthy counter", name) + state.unhealthy_cycles = 0 self._maybe_alert_recovery(container, state) # ── Poll loop ───────────────────────────────────────────────────────────── @@ -1270,6 +1472,7 @@ def _poll_loop(self) -> None: self._stop_event.wait(interval) def _poll_all_containers(self) -> None: + self._flush_boot_summary() excluded = self._excluded restart_threshold = self._restart_threshold restart_window = self._restart_window_secs @@ -1292,41 +1495,61 @@ def _poll_all_containers(self) -> None: health = self._get_health(container) # healthy, unhealthy, none state = self.state.get(name) - # Track restarts within rolling window - state.restart_times = [t for t in state.restart_times - if now - t < restart_window] + # Restart tracking reads Docker's RestartCount rather than counting + # polls that observe status == "restarting": one slow restart seen on + # several polls used to look like a loop, while restarts that began + # and finished between two polls were never counted at all. + try: + restart_count = int(container.attrs.get("RestartCount", 0) or 0) + except (TypeError, ValueError): + restart_count = 0 + + with state.lock: + state.restart_times = [t for t in state.restart_times + if now - t < restart_window] + if state.last_restart_count < 0 or restart_count < state.last_restart_count: + # First sighting, or the container was recreated (count resets): + # take a baseline instead of alerting on history we never saw. + state.last_restart_count = restart_count + elif restart_count > state.last_restart_count: + new_restarts = restart_count - state.last_restart_count + state.restart_times.extend([now] * min(new_restarts, restart_threshold)) + state.last_restart_count = restart_count + restarts = len(state.restart_times) + + if restarts >= restart_threshold: + failure_type = "restart-loop" + if self._claim_alert(state, failure_type): + payload = build_payload(container, failure_type, + self.host, self.runbook_base, + probe_detail=f"Restart-loop probe failed - {restarts} restarts in {int(self._restart_window_secs // 60)}min window") + self._dispatch(payload) - # Detect restart loop via poll (supplement to event stream) if status == "restarting": log.info("CONTAINER %-30s status=%-12s health=%s", name, status, health) - state.restart_times.append(now) - if len(state.restart_times) >= restart_threshold: - failure_type = "restart-loop" - if should_alert(state, failure_type, self._cooldown_minutes): - payload = build_payload(container, failure_type, - self.host, self.runbook_base, - probe_detail=f"Restart-loop probe failed - {len(state.restart_times)} restarts in {int(self._restart_window_secs // 60)}min window") - dispatch_alert(payload, self.cfg) - record_alert(state, failure_type) # Detect stuck containers missed by event stream elif status == "running" and health == "unhealthy": log.info("CONTAINER %-30s status=%-12s health=%s", name, status, health) - state.unhealthy_cycles += 1 - if state.unhealthy_cycles >= unhealthy_threshold: + with state.lock: + state.unhealthy_cycles += 1 + cycles = state.unhealthy_cycles + if cycles >= unhealthy_threshold: failure_type = "unhealthy" - if should_alert(state, failure_type, self._cooldown_minutes): + if self._claim_alert(state, failure_type): probe_detail = get_docker_health_log(container) or "Docker health probe failed - health_status: unhealthy (poll-detected)" payload = build_payload(container, failure_type, self.host, self.runbook_base, probe_detail=probe_detail) - dispatch_alert(payload, self.cfg) - record_alert(state, failure_type) + self._dispatch(payload) elif status == "running" and health in ("healthy", "none"): # Determine which health check to run and get result first probe_detail = "" - if health == "none" and self.cfg.get("auto_health_check", {}).get("enabled"): + manual_spec = self._manual_checks.get(name) if health == "none" else None + if isinstance(manual_spec, dict): + is_healthy, probe_detail = self._manual_health_check(container, manual_spec) + elif health == "none" and self._auto_cfg.get("enabled"): is_healthy, probe_detail = self._auto_discover_health_check(container) else: is_healthy = True # Docker-native healthy, or no check configured @@ -1336,33 +1559,45 @@ def _poll_all_containers(self) -> None: log.info("CONTAINER %-30s status=%-12s health=%s", name, status, display_health) if not is_healthy: - state.unhealthy_cycles += 1 + with state.lock: + state.unhealthy_cycles += 1 + cycles = state.unhealthy_cycles log.debug( "%s: health check failed — cycle %d/%d", - name, state.unhealthy_cycles, unhealthy_threshold, + name, cycles, unhealthy_threshold, ) - if state.unhealthy_cycles >= unhealthy_threshold: + if cycles >= unhealthy_threshold: failure_type = "unhealthy" - if should_alert(state, failure_type, self._cooldown_minutes): + if self._claim_alert(state, failure_type): payload = build_payload(container, failure_type, self.host, self.runbook_base, probe_detail=probe_detail) - dispatch_alert(payload, self.cfg) - record_alert(state, failure_type) + self._dispatch(payload) else: - if state.unhealthy_cycles > 0: - log.info("%s: check recovered — resetting counters", name) - state.unhealthy_cycles = 0 - state.restart_times = [] - self._maybe_alert_recovery(container, state) + with state.lock: + if state.unhealthy_cycles > 0: + log.info("%s: check recovered — resetting counters", name) + state.unhealthy_cycles = 0 + # restart_times ages out on its own. Clearing it here would + # declare recovery for a container merely caught healthy + # between two crashes of an ongoing restart loop. + still_looping = (state.alerted_for == "restart-loop" + and len(state.restart_times) >= restart_threshold) + if not still_looping: + self._maybe_alert_recovery(container, state) else: log.info("CONTAINER %-30s status=%-12s health=%s", name, status, health) - self.state.cleanup_stale(seen_names) + # During the post-reboot window a restored container may not exist yet — + # keep its open alert so the recovery still fires once it comes back. + self.state.cleanup_stale(seen_names, retain_open=bool(self._grace_until)) # Clean up discovery cache for containers no longer running for _name in self._discovered_checks.keys() - seen_names: self._discovered_checks.pop(_name, None) + # Refresh the stored boot time every cycle: otherwise a host that never + # alerted has no baseline, and an unclean reboot goes unreported. + self.state.save(self._boot_time) @staticmethod def _get_health(container) -> str: @@ -1435,6 +1670,57 @@ def _get_container_ports(container) -> list[int]: pass return sorted(ports) + def _manual_health_check(self, container, spec: dict) -> tuple[bool, str]: + """ + Run an operator-defined check from container_health_checks, which takes + precedence over auto-discovery for containers with health=none. + type: http — GET http://:, healthy = status < 500. + type: exec — run `command` inside the container, healthy = exit code 0. + Returns (healthy, probe_detail). + """ + name = container.name + check_type = str(spec.get("type", "http")).lower() + timeout = max(1, int(spec.get("timeout_seconds", 5))) + + if check_type == "exec": + command = spec.get("command") + if not command: + log.warning("%s: manual exec check has no 'command' — skipping", name) + return True, "" + try: + result = container.exec_run(command) + except Exception as exc: + log.debug("%s: manual exec check error: %s", name, exc) + return True, "" # can't run the probe — don't false-alarm + if result.exit_code == 0: + return True, "" + output = (result.output or b"").decode(errors="replace").strip() + return False, (f"Exec probe failed - '{command}' exited {result.exit_code}: " + f"{output or '(no output)'}") + + if check_type != "http": + log.warning("%s: unknown manual check type %r — skipping", name, check_type) + return True, "" + + port = spec.get("internal_port") + if not port: + log.warning("%s: manual http check has no 'internal_port' — skipping", name) + return True, "" + ip = self._get_container_ip(container) + if not ip: + log.debug("%s: manual http check — no container IP available, skipping", name) + return True, "" + url = f"http://{ip}:{port}{spec.get('path', '/')}" + try: + r = self._session.get(url, timeout=timeout) + except Exception as exc: + # An explicitly configured endpoint that cannot be reached is a failure. + return False, f"HTTP probe failed - {exc}, URI {url}" + if r.status_code < 500: + return True, "" + return False, (f"HTTP probe failed - error response {r.status_code} " + f"{r.reason or 'FAIL'}, URI {url}") + def _auto_discover_health_check(self, container) -> tuple[bool, str]: """ Auto-discover a working health endpoint for a container with health=none. @@ -1598,6 +1884,7 @@ def start(self) -> None: def stop(self) -> None: log.info("Watchdog shutting down") self._stop_event.set() + self.state.save(self._boot_time) # ── Entry point ─────────────────────────────────────────────────────────────── From e7141577993bbfd0b6949d272a590a340c157548 Mon Sep 17 00:00:00 2001 From: Egor Egorov Date: Thu, 1 Oct 2026 20:00:33 +0000 Subject: [PATCH 6/7] Fix watchdog reliability and deployment validation --- .github/workflows/ci.yml | 15 ++- DEPLOYMENT.md | 4 +- README.md | 2 +- docker-compose.build.yaml | 8 +- install.sh | 79 +++++++++------ tests/test_watchdog.py | 113 ++++++++++++++++++++++ uninstall.sh | 54 ++++++----- watchdog.py | 195 +++++++++++++++++++++++++++----------- 8 files changed, 350 insertions(+), 120 deletions(-) create mode 100644 tests/test_watchdog.py diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index f8e65ee..950e08f 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -22,6 +22,14 @@ jobs: run: | python3 -m py_compile watchdog.py + - name: Install test dependencies + run: | + python3 -m pip install docker==7.1.0 requests==2.32.3 PyYAML==6.0.2 + + - name: Run regression tests + run: | + python3 -m unittest discover -s tests -v + - name: Validate VERSION file run: | test -f VERSION @@ -33,8 +41,13 @@ jobs: - name: Prepare CI environment run: | cp .env.example .env + echo "WATCHDOG_VERSION=$(cat VERSION)" >> .env - name: Validate Docker Compose run: | docker compose -f docker-compose.yaml config -q - docker compose -f docker-compose.build.yaml config -q + WATCHDOG_VERSION="$(cat VERSION)" docker compose -f docker-compose.build.yaml config -q + + - name: Build watchdog image + run: | + WATCHDOG_VERSION="$(cat VERSION)" docker compose -f docker-compose.build.yaml build diff --git a/DEPLOYMENT.md b/DEPLOYMENT.md index dec6bf5..899164f 100644 --- a/DEPLOYMENT.md +++ b/DEPLOYMENT.md @@ -346,7 +346,7 @@ Use this path when you need full control over configuration files before startin *Note* - If internet access is available, build the image and skip to [Start the container](#start-the-container) below. ```bash -docker compose -f docker-compose.build.yaml build +WATCHDOG_VERSION="$(cat VERSION)" docker compose -f docker-compose.build.yaml build ``` #### Offline Installation @@ -491,7 +491,7 @@ docker compose up -d ```bash # Rebuild the image (requires internet) -docker compose -f docker-compose.build.yaml build +WATCHDOG_VERSION="$(cat VERSION)" docker compose -f docker-compose.build.yaml build # Redeploy using the production runtime file docker compose up -d diff --git a/README.md b/README.md index 502995f..30a825f 100644 --- a/README.md +++ b/README.md @@ -438,7 +438,7 @@ docker compose logs docker-container-watchdog Common causes: - **Missing `.env` file** — run `cp .env.example .env` and fill in credentials - **Docker socket not accessible** — ensure `/var/run/docker.sock` exists and the container has read access -- **Image not loaded** — run `docker images watchdog`; if empty, build with `docker compose -f docker-compose.build.yaml build` (or `docker build -t watchdog:latest .` directly — see [DEPLOYMENT.md](DEPLOYMENT.md) for details) +- **Image not loaded** — run `docker images watchdog`; if empty, set `WATCHDOG_VERSION=$(cat VERSION)` and build with `WATCHDOG_VERSION="$WATCHDOG_VERSION" docker compose -f docker-compose.build.yaml build` (see [DEPLOYMENT.md](DEPLOYMENT.md) for details) ### Alert Notifications Not Received diff --git a/docker-compose.build.yaml b/docker-compose.build.yaml index 5b9116d..c2fe83e 100644 --- a/docker-compose.build.yaml +++ b/docker-compose.build.yaml @@ -2,14 +2,14 @@ # Use this file when you have internet access and want to build a fresh image. # # Usage: +# export WATCHDOG_VERSION=$(cat VERSION) # docker compose -f docker-compose.build.yaml build # docker compose -f docker-compose.build.yaml up -d # # After a successful build, export the image for offline customers — always # ship the versioned tag so install.sh can pin the deployment to it: -# VERSION=$(cat VERSION) -# docker tag watchdog:"$VERSION" watchdog:latest -# docker save watchdog:"$VERSION" watchdog:latest -o watchdog.tar +# docker tag watchdog:"$WATCHDOG_VERSION" watchdog:latest +# docker save watchdog:"$WATCHDOG_VERSION" watchdog:latest -o watchdog.tar # ───────────────────────────────────────────────────────────────────────────── services: @@ -17,7 +17,7 @@ services: docker-container-watchdog: build: context: . - image: watchdog:${WATCHDOG_VERSION:-latest} + image: watchdog:${WATCHDOG_VERSION:?Set WATCHDOG_VERSION from the VERSION file before building} container_name: docker-container-watchdog restart: always network_mode: host diff --git a/install.sh b/install.sh index 4f161a0..3228326 100644 --- a/install.sh +++ b/install.sh @@ -198,23 +198,24 @@ update_env_vars() { } # watchdog-config.yaml is bind-mounted read-only into a container running as -# RUN_UID:RUN_GID. A config left 600 root:root by an older install makes the -# agent exit on startup, so normalise ownership/mode on every run. +# RUN_UID:RUN_GID. Keep it root-owned and group-readable: the runtime can read +# it, but UID 1000 on the host cannot rewrite probe commands and indirectly +# control Docker exec operations. harden_config_permissions() { local file="$1" [ -f "$file" ] || return 0 - if ! _run_priv chown "${RUN_UID}:${RUN_GID}" "$file"; then - print_error "Failed to set ownership of ${file} to UID ${RUN_UID}:GID ${RUN_GID}" + if ! _run_priv chown "0:${RUN_GID}" "$file"; then + print_error "Failed to set ownership of ${file} to root:GID ${RUN_GID}" echo "" - echo " The container runs as ${RUN_UID}:${RUN_GID} and could not read it. Fix manually:" - echo " sudo chown ${RUN_UID}:${RUN_GID} ${file} && sudo chmod 640 ${file}" + echo " Fix manually:" + echo " sudo chown root:${RUN_GID} ${file} && sudo chmod 640 ${file}" return 1 fi if ! _run_priv chmod 640 "$file"; then print_error "Failed to set permissions 640 on ${file}" return 1 fi - print_success "Config readable by the container runtime user (${RUN_UID}:${RUN_GID}, mode 640)" + print_success "Config secured as root:${RUN_GID}, mode 640 (runtime read-only)" return 0 } @@ -338,23 +339,26 @@ load_docker_image() { fi echo "$load_output" - # The archive must carry the release tag. Promoting whatever it happened - # to contain would silently relabel an older build as this version, and - # the post-install image check would then confirm the wrong release. - if ! docker image inspect "$IMAGE_NAME" &>/dev/null; then - print_error "${IMAGE_ARCHIVE} does not contain ${IMAGE_NAME}" + # The archive itself must carry the release tag. Merely inspecting the + # local tag after docker load is insufficient because a stale tag may + # have existed before this installer ran. + if ! printf '%s\n' "$load_output" | grep -Fxq "Loaded image: ${IMAGE_NAME}"; then + print_error "${IMAGE_ARCHIVE} did not load the required tag ${IMAGE_NAME}" echo "" - echo " The archive loaded:" - printf '%s\n' "$load_output" \ - | sed -n -e 's/^Loaded image: */ /p' -e 's/^Loaded image ID: */ /p' + echo " docker load reported:" + printf ' %s\n' "$load_output" echo "" echo " This package is version ${VERSION} (from ${VERSION_FILE})." - echo " Obtain a watchdog.tar built for ${VERSION}, or rebuild it:" + echo " Rebuild the archive with the versioned tag:" echo " docker build -t ${IMAGE_NAME} ." echo " docker tag ${IMAGE_NAME} ${IMAGE_LATEST}" echo " docker save ${IMAGE_NAME} ${IMAGE_LATEST} -o watchdog.tar" exit 1 fi + if ! docker image inspect "$IMAGE_NAME" &>/dev/null; then + print_error "Required image ${IMAGE_NAME} is not available after docker load" + exit 1 + fi else print_warning "Image archive not found: ${IMAGE_ARCHIVE}" @@ -464,11 +468,11 @@ configure_all() { if [ -f "$ENV_FILE" ]; then local CURRENT_GID MIGRATED_GID CURRENT_GID=$(_read_file_priv "$ENV_FILE" | grep -E '^DOCKER_GID=' | tail -n1 | cut -d= -f2 | tr -d '[:space:]') - if [[ "$CURRENT_GID" =~ ^[0-9]+$ ]]; then - MIGRATED_GID="$CURRENT_GID" - else - print_warning "Existing .env predates non-root hardening — detecting and adding it" - MIGRATED_GID=$(detect_docker_gid) || exit 1 + MIGRATED_GID=$(detect_docker_gid) || exit 1 + if [[ "$CURRENT_GID" =~ ^[0-9]+$ ]] && [ "$CURRENT_GID" != "$MIGRATED_GID" ]; then + print_warning "Docker socket GID changed: ${CURRENT_GID} -> ${MIGRATED_GID}; updating .env" + elif ! [[ "$CURRENT_GID" =~ ^[0-9]+$ ]]; then + print_warning "Existing .env predates non-root hardening — adding detected Docker socket GID" fi if ! update_env_vars "$ENV_FILE" \ "WATCHDOG_UID=${RUN_UID}" \ @@ -720,9 +724,10 @@ configure_all() { printf " timeout_seconds: 2\n" printf "\ncontainer_health_checks: {}\n" } > "$CONFIG_TMP" - # Owned by the container's runtime user: it is bind-mounted read-only and - # the agent cannot start if it is unreadable at UID/GID 1000. - if ! _publish_file "$CONFIG_TMP" "$CONFIG_FILE" 640 "${RUN_UID}:${RUN_GID}"; then + # Root-owned, runtime-group-readable. This prevents host UID 1000 from + # rewriting exec health-check commands while still allowing the non-root + # container to read the bind-mounted file. + if ! _publish_file "$CONFIG_TMP" "$CONFIG_FILE" 640 "0:${RUN_GID}"; then rm -f "$CONFIG_TMP" exit 1 fi @@ -769,13 +774,25 @@ start_watchdog() { verify_deployment() { print_section "Verifying Deployment" - sleep 2 # give the container a moment to initialise + local state health status attempts=0 + while [ "$attempts" -lt 18 ]; do + state=$(docker inspect --format '{{.State.Status}}' "$CONTAINER_NAME" 2>/dev/null || true) + health=$(docker inspect --format '{{if .State.Health}}{{.State.Health.Status}}{{else}}none{{end}}' "$CONTAINER_NAME" 2>/dev/null || true) + status=$(docker ps -a --filter "name=^${CONTAINER_NAME}$" --format '{{.Status}}' 2>/dev/null || true) - STATUS=$(docker ps --filter "name=^${CONTAINER_NAME}$" --format '{{.Status}}' 2>/dev/null || echo "") - if [ -n "$STATUS" ]; then - print_success "Container is running: ${STATUS}" - else - print_error "Container ${CONTAINER_NAME} is not running — the installation did not succeed" + if [ "$state" = "running" ] && { [ "$health" = "healthy" ] || [ "$health" = "none" ]; }; then + print_success "Container is running and healthy: ${status}" + break + fi + if [ "$state" = "exited" ] || [ "$state" = "dead" ]; then + break + fi + attempts=$((attempts + 1)) + sleep 5 + done + + if [ "$state" != "running" ] || { [ "$health" != "healthy" ] && [ "$health" != "none" ]; }; then + print_error "Container ${CONTAINER_NAME} did not become healthy (state=${state:-unknown}, health=${health:-unknown})" echo "" echo " Check the logs for the reason:" echo " $DC -f ${INSTALL_DIR}/docker-compose.yaml logs" @@ -830,7 +847,7 @@ display_usage_instructions() { echo " docker logs -f ${CONTAINER_NAME}" echo "" echo -e "${GREEN}Edit alert channels / thresholds:${NC}" - echo " nano ${INSTALL_DIR}/watchdog-config.yaml" + echo " sudoedit ${INSTALL_DIR}/watchdog-config.yaml" echo " $DC -f ${INSTALL_DIR}/docker-compose.yaml restart" echo "" echo -e "${GREEN}Edit secrets (Slack / SMTP):${NC}" diff --git a/tests/test_watchdog.py b/tests/test_watchdog.py new file mode 100644 index 0000000..1cf26bd --- /dev/null +++ b/tests/test_watchdog.py @@ -0,0 +1,113 @@ +import json +import tempfile +import threading +import time +import unittest +from types import SimpleNamespace +from unittest.mock import patch + +import watchdog + + +class FakeExecContainer: + name = "svc" + id = "abc123" + + def __init__(self, delay=0.0, exit_code=0): + self.delay = delay + self.exit_code = exit_code + + def exec_run(self, command): + time.sleep(self.delay) + return SimpleNamespace(exit_code=self.exit_code, output=b"probe output") + + +class WatchdogRegressionTests(unittest.TestCase): + def make_watchdog_shell(self): + w = object.__new__(watchdog.Watchdog) + w._manual_exec_lock = threading.Lock() + w._manual_exec_inflight = set() + return w + def test_manual_exec_timeout_does_not_block_poll_thread(self): + w = self.make_watchdog_shell() + c = FakeExecContainer(delay=0.15) + started = time.monotonic() + healthy, detail = w._manual_exec_health_check(c, ["check"], 0.02) + elapsed = time.monotonic() - started + self.assertFalse(healthy) + self.assertIn("timed out", detail) + self.assertLess(elapsed, 0.10) + + healthy, detail = w._manual_exec_health_check(c, ["check"], 0.02) + self.assertFalse(healthy) + self.assertIn("previous probe", detail) + time.sleep(0.18) + + def test_invalid_manual_timeout_falls_back_without_raising(self): + w = self.make_watchdog_shell() + c = FakeExecContainer() + healthy, detail = w._manual_health_check( + c, {"type": "exec", "command": ["true"], "timeout_seconds": "bad"} + ) + self.assertTrue(healthy) + self.assertEqual(detail, "") + + def test_recovery_disabled_still_clears_open_state(self): + w = object.__new__(watchdog.Watchdog) + w.cfg = {"alert_on_recovery": False} + w.state = watchdog.WatchdogState("") + w._boot_time = 123 + state = w.state.get("svc") + state.alerted_for = "crashed" + w._maybe_alert_recovery(SimpleNamespace(name="svc"), state) + self.assertEqual(state.alerted_for, "") + def test_reboot_summary_uses_current_state_not_folded_failures(self): + w = object.__new__(watchdog.Watchdog) + w.cfg = {} + w.host = "host" + w.runbook_base = "runbook" + w._boot_time = 1 + w._grace_until = time.time() - 1 + w._folded_lock = threading.Lock() + w._folded = [ + watchdog.AlertPayload("HIGH", "fixed", "", "host", "crashed", + None, "", "", ""), + watchdog.AlertPayload("INFO", "fixed", "", "host", "recovered", + None, "", "", ""), + ] + w.state = watchdog.WatchdogState("") + still = w.state.get("still") + still.alerted_for = "unhealthy" + w._container_tally = lambda: (2, 1, ["down"]) + + with patch("watchdog.dispatch_alert") as dispatch: + w._flush_boot_summary() + + payload = dispatch.call_args.args[0] + self.assertEqual(payload.severity, "HIGH") + self.assertIn("Recovered since last alert: fixed", payload.probe_detail) + self.assertIn("Still failing: down, still", payload.probe_detail) + self.assertNotIn("Still failing: down, fixed", payload.probe_detail) + + def test_state_save_produces_valid_json_under_concurrency(self): + with tempfile.TemporaryDirectory() as td: + path = f"{td}/state.json" + state = watchdog.WatchdogState(path) + st = state.get("svc") + st.alerted_for = "crashed" + st.last_alert_time = 42.0 + threads = [ + threading.Thread(target=lambda: [state.save(123) for _ in range(20)]) + for _ in range(4) + ] + for thread in threads: + thread.start() + for thread in threads: + thread.join() + with open(path) as f: + record = json.load(f) + self.assertEqual(record["containers"]["svc"]["alerted_for"], "crashed") + + +if __name__ == "__main__": + unittest.main() diff --git a/uninstall.sh b/uninstall.sh index 7e8d1d2..8b71076 100644 --- a/uninstall.sh +++ b/uninstall.sh @@ -355,34 +355,38 @@ remove_docker_image() { fi } -# Internal helper — stops/removes ALL containers that reference the image, then removes the image. +# Internal helper — removes only this deployment's image tags. +# Never stop/remove unrelated containers that happen to use a watchdog image. _do_remove_image() { - print_info "Removing Docker image: ${IMAGE_TAGS[*]}..." - - # Stop and remove every container (running or stopped) that uses this image. - # We iterate rather than rely on --filter ancestor because Docker 20.10 combos are unreliable. - for TAG in "${IMAGE_TAGS[@]}"; do - ALL_CTRS=$(docker ps -aq --filter "ancestor=${TAG}" 2>/dev/null || true) - [ -n "$ALL_CTRS" ] || continue - for CID in $ALL_CTRS; do - CSTATUS=$(docker inspect --format '{{.State.Status}}' "$CID" 2>/dev/null || true) - CNAME=$(docker inspect --format '{{.Name}}' "$CID" 2>/dev/null | sed 's|^/||' || true) - if [ "$CSTATUS" = "running" ] || [ "$CSTATUS" = "paused" ]; then - print_info " Stopping running container ${CNAME:-$CID} (${CSTATUS})..." - docker stop "$CID" 2>/dev/null || true - fi - print_info " Removing container ${CNAME:-$CID}..." - docker rm "$CID" 2>/dev/null || true - done - done + print_info "Removing Docker image for deployed version: ${IMAGE_NAME}..." + + local VERSION_ID LATEST_ID TAG + local SAFE_TAGS=() + + VERSION_ID=$(docker image inspect "${IMAGE_NAME}" --format '{{.Id}}' 2>/dev/null || true) + if [ -n "$VERSION_ID" ]; then + SAFE_TAGS+=("${IMAGE_NAME}") + fi + + if [ "${IMAGE_NAME}" != "${IMAGE_LATEST}" ]; then + LATEST_ID=$(docker image inspect "${IMAGE_LATEST}" --format '{{.Id}}' 2>/dev/null || true) + if [ -n "$VERSION_ID" ] && [ "$LATEST_ID" = "$VERSION_ID" ]; then + SAFE_TAGS+=("${IMAGE_LATEST}") + elif [ -n "$LATEST_ID" ]; then + print_info "Preserving ${IMAGE_LATEST}: it points to a different image" + fi + fi + + if [ "${#SAFE_TAGS[@]}" -eq 0 ]; then + print_info "No matching watchdog image tags remain" + return 0 + fi - # Remove every tag — leaving the :latest alias behind keeps the layers on disk. - for TAG in "${IMAGE_TAGS[@]}"; do - docker image inspect "${TAG}" &>/dev/null || continue - if docker rmi "${TAG}" 2>&1; then - print_success "Docker image removed: ${TAG}" + for TAG in "${SAFE_TAGS[@]}"; do + if docker rmi "$TAG" 2>&1; then + print_success "Docker image tag removed: $TAG" else - print_warning "Could not remove image — check for containers using it: docker ps -a --filter ancestor=${TAG}" + print_warning "Could not remove $TAG; another container may still reference the image" fi done } diff --git a/watchdog.py b/watchdog.py index 8fc6f2f..aab65fa 100644 --- a/watchdog.py +++ b/watchdog.py @@ -169,6 +169,7 @@ def __init__(self, name: str): class WatchdogState: def __init__(self, state_file: str = ""): self._lock = threading.Lock() + self._save_lock = threading.Lock() self._states: dict[str, ContainerState] = {} self._state_file = state_file @@ -189,6 +190,17 @@ def cleanup_stale(self, active_names: set, retain_open: bool = False) -> None: for n in stale: log.debug("Removing stale state for container: %s", n) + def open_alert_names(self) -> list[str]: + """Return container names with an alert that is still open.""" + with self._lock: + states = list(self._states.values()) + names: list[str] = [] + for st in states: + with st.lock: + if st.alerted_for and st.alerted_for != "recovered": + names.append(st.name) + return sorted(names) + # ── Persistence ─────────────────────────────────────────────────────────── def load(self) -> tuple[dict, int]: """Return (persisted alert entries, host boot time recorded at last save).""" @@ -211,31 +223,35 @@ def load(self) -> tuple[dict, int]: def save(self, boot_time: int) -> None: if not self._state_file: return - with self._lock: - containers = { - st.name: { - "alerted_for": st.alerted_for, - "last_alert_time": st.last_alert_time, - } - for st in self._states.values() - if st.alerted_for + # Event handling and polling can persist concurrently. Serialize the + # complete snapshot/write/replace sequence so writers cannot race. + with self._save_lock: + with self._lock: + states = list(self._states.values()) + containers: dict[str, dict] = {} + for st in states: + with st.lock: + if st.alerted_for: + containers[st.name] = { + "alerted_for": st.alerted_for, + "last_alert_time": st.last_alert_time, + } + record = { + "version": _STATE_VERSION, + "boot_time": boot_time, + "saved_at": time.time(), + "containers": containers, } - record = { - "version": _STATE_VERSION, - "boot_time": boot_time, - "saved_at": time.time(), - "containers": containers, - } - tmp = f"{self._state_file}.tmp" - try: - parent = os.path.dirname(self._state_file) - if parent: - os.makedirs(parent, exist_ok=True) - with open(tmp, "w") as f: - json.dump(record, f) - os.replace(tmp, self._state_file) # atomic: never leave a half-written file - except Exception as exc: - log.warning("Could not write state file %s: %s", self._state_file, exc) + tmp = f"{self._state_file}.tmp" + try: + parent = os.path.dirname(self._state_file) + if parent: + os.makedirs(parent, exist_ok=True) + with open(tmp, "w") as f: + json.dump(record, f) + os.replace(tmp, self._state_file) + except Exception as exc: + log.warning("Could not write state file %s: %s", self._state_file, exc) # ── Alert payload ───────────────────────────────────────────────────────────── @@ -1226,6 +1242,9 @@ def __init__(self, cfg: dict): self._boot_grace_secs = max(0, int(cfg.get("boot_grace_seconds", 180) or 0)) self._grace_until: float = 0.0 self._folded: list[AlertPayload] = [] # alerts held during the post-reboot window + self._folded_lock = threading.Lock() + self._manual_exec_lock = threading.Lock() + self._manual_exec_inflight: set[str] = set() self._restore_state() # ── State restore ───────────────────────────────────────────────────────── @@ -1274,16 +1293,21 @@ def _claim_alert(self, state: ContainerState, failure_type: str) -> bool: def _dispatch(self, payload: AlertPayload) -> None: """Send an alert, or hold it for the summary during the post-reboot window.""" - if self._grace_until and time.time() < self._grace_until: - self._folded.append(payload) + held = False + with self._folded_lock: + if self._grace_until and time.time() < self._grace_until: + self._folded.append(payload) + held = True + if held: log.info("%s: %s held for post-reboot summary", payload.container_name, payload.failure_type) return dispatch_alert(payload, self.cfg) - def _container_tally(self) -> tuple[int, int]: - """(running, not-running) counts of monitored containers, for the reboot summary.""" + def _container_tally(self) -> tuple[int, int, list[str]]: + """Return running/not-running counts and current non-running names.""" running = not_running = 0 + non_running_names: list[str] = [] try: for c in self.client.containers.list(all=True): if c.name in self._excluded: @@ -1292,27 +1316,33 @@ def _container_tally(self) -> tuple[int, int]: running += 1 else: not_running += 1 + non_running_names.append(c.name) except Exception as exc: log.debug("Could not tally containers for the reboot summary: %s", exc) - return running, not_running + return running, not_running, sorted(non_running_names) def _flush_boot_summary(self) -> None: - """Emit the single post-reboot status alert once the grace window closes.""" - if not self._grace_until or time.time() < self._grace_until: - return - self._grace_until = 0.0 - recovered = sorted({p.container_name for p in self._folded if p.failure_type == "recovered"}) - still_down = sorted({p.container_name for p in self._folded if p.failure_type != "recovered"}) - self._folded.clear() - # Sent even when nothing was held: after a reboot, silence is - # indistinguishable from the watchdog itself having failed to come back. + """Emit one post-reboot summary based on state at the end of a full poll.""" + with self._folded_lock: + if not self._grace_until or time.time() < self._grace_until: + return + self._grace_until = 0.0 + folded = list(self._folded) + self._folded.clear() + + running, not_running, non_running_names = self._container_tally() + still_down_set = set(self.state.open_alert_names()) | set(non_running_names) + recovered = sorted({ + p.container_name for p in folded if p.failure_type == "recovered" + } - still_down_set) + still_down = sorted(still_down_set) + if not recovered and not still_down: - log.info("Post-reboot grace window closed — no alerts were held") + log.info("Post-reboot grace window closed — no failures remain") booted_at = ( datetime.fromtimestamp(self._boot_time, timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ") if self._boot_time else "unknown" ) - running, not_running = self._container_tally() detail = ( f"Host booted {booted_at}. " f"Containers: {running} running, {not_running} not running. " @@ -1335,9 +1365,7 @@ def _flush_boot_summary(self) -> None: # ── Recovery alert ───────────────────────────────────────────────────────── def _maybe_alert_recovery(self, container, state: ContainerState) -> None: - """Fire a one-time RECOVERED alert when a previously-alarmed container is healthy again.""" - if not self.cfg.get("alert_on_recovery", True): - return + """Clear resolved state and optionally fire a one-time RECOVERED alert.""" with state.lock: previous = state.alerted_for if not previous or previous == "recovered": @@ -1345,6 +1373,10 @@ def _maybe_alert_recovery(self, container, state: ContainerState) -> None: state.alerted_for = "" state.last_alert_time = time.time() self.state.save(self._boot_time) + if not self.cfg.get("alert_on_recovery", True): + log.info("%s recovered from %s — recovery notification disabled", + state.name, previous) + return payload = build_payload(container, "recovered", self.host, self.runbook_base, probe_detail=f"Recovered from '{previous}'") self._dispatch(payload) @@ -1472,7 +1504,6 @@ def _poll_loop(self) -> None: self._stop_event.wait(interval) def _poll_all_containers(self) -> None: - self._flush_boot_summary() excluded = self._excluded restart_threshold = self._restart_threshold restart_window = self._restart_window_secs @@ -1598,6 +1629,8 @@ def _poll_all_containers(self) -> None: # Refresh the stored boot time every cycle: otherwise a host that never # alerted has no baseline, and an unclean reboot goes unreported. self.state.save(self._boot_time) + # Evaluate the reboot summary only after a complete poll. + self._flush_boot_summary() @staticmethod def _get_health(container) -> str: @@ -1670,33 +1703,83 @@ def _get_container_ports(container) -> list[int]: pass return sorted(ports) + def _manual_exec_health_check(self, container, command, timeout: int) -> tuple[bool, str]: + """Run an exec probe without allowing a hung command to stall the poll loop.""" + name = container.name + key = getattr(container, "id", None) or name + with self._manual_exec_lock: + if key in self._manual_exec_inflight: + return False, ( + f"Exec probe failed - previous probe for {name} is still running " + f"after its {timeout}s timeout" + ) + self._manual_exec_inflight.add(key) + + done = threading.Event() + holder: dict[str, object] = {} + + def _worker() -> None: + try: + holder["result"] = container.exec_run(command) + except Exception as exc: + holder["error"] = exc + finally: + with self._manual_exec_lock: + self._manual_exec_inflight.discard(key) + done.set() + + threading.Thread( + target=_worker, name=f"manual-exec-{name}", daemon=True + ).start() + + if not done.wait(timeout): + return False, f"Exec probe failed - timed out after {timeout}s running {command!r}" + + error = holder.get("error") + if error is not None: + log.debug("%s: manual exec check error: %s", name, error) + return True, "" + + result = holder.get("result") + if result is None: + log.debug("%s: manual exec check returned no result", name) + return True, "" + if result.exit_code == 0: + return True, "" + raw_output = result.output or b"" + output = ( + raw_output.decode(errors="replace").strip() + if isinstance(raw_output, (bytes, bytearray)) + else str(raw_output).strip() + ) + return False, ( + f"Exec probe failed - {command!r} exited {result.exit_code}: " + f"{output or '(no output)'}" + ) + def _manual_health_check(self, container, spec: dict) -> tuple[bool, str]: """ Run an operator-defined check from container_health_checks, which takes precedence over auto-discovery for containers with health=none. type: http — GET http://:, healthy = status < 500. - type: exec — run `command` inside the container, healthy = exit code 0. + type: exec — run command inside the container, healthy = exit code 0. Returns (healthy, probe_detail). """ name = container.name check_type = str(spec.get("type", "http")).lower() - timeout = max(1, int(spec.get("timeout_seconds", 5))) + try: + timeout = max(1, int(spec.get("timeout_seconds", 5))) + except (TypeError, ValueError): + log.warning("%s: invalid timeout_seconds=%r — using 5s", name, + spec.get("timeout_seconds")) + timeout = 5 if check_type == "exec": command = spec.get("command") if not command: log.warning("%s: manual exec check has no 'command' — skipping", name) return True, "" - try: - result = container.exec_run(command) - except Exception as exc: - log.debug("%s: manual exec check error: %s", name, exc) - return True, "" # can't run the probe — don't false-alarm - if result.exit_code == 0: - return True, "" - output = (result.output or b"").decode(errors="replace").strip() - return False, (f"Exec probe failed - '{command}' exited {result.exit_code}: " - f"{output or '(no output)'}") + return self._manual_exec_health_check(container, command, timeout) if check_type != "http": log.warning("%s: unknown manual check type %r — skipping", name, check_type) From d3e8c6646fdcf061269ee0942f9e556afefbe88a Mon Sep 17 00:00:00 2001 From: Egor Egorov Date: Thu, 1 Oct 2026 20:20:27 +0000 Subject: [PATCH 7/7] Mark installer scripts executable --- install.sh | 0 uninstall.sh | 0 2 files changed, 0 insertions(+), 0 deletions(-) mode change 100644 => 100755 install.sh mode change 100644 => 100755 uninstall.sh diff --git a/install.sh b/install.sh old mode 100644 new mode 100755 diff --git a/uninstall.sh b/uninstall.sh old mode 100644 new mode 100755