#!/bin/sh # health.sh - one-screen health snapshot of the server you run it on. # # curl -fsSL https://sitereliability.sh/health.sh | sh # # Source: https://sitereliability.sh/health.sh (read it before you pipe it) # Checksums: https://sitereliability.sh/checksums.txt # # Read-only: inspects local system state. No sudo, no writes, no network calls. # Built for Linux servers; degrades gracefully elsewhere. set -u if [ -t 1 ]; then C_OK="$(printf '\033[32m')"; C_WARN="$(printf '\033[33m')" C_BAD="$(printf '\033[31m')"; C_DIM="$(printf '\033[2m')"; C_OFF="$(printf '\033[0m')" else C_OK=""; C_WARN=""; C_BAD=""; C_DIM=""; C_OFF="" fi WARN=0; FAIL=0 ok() { printf ' %sok%s %s\n' "$C_OK" "$C_OFF" "$1"; } warn() { WARN=$((WARN+1)); printf ' %swarn%s %s\n' "$C_WARN" "$C_OFF" "$1"; } bad() { FAIL=$((FAIL+1)); printf ' %sfail%s %s\n' "$C_BAD" "$C_OFF" "$1"; } info() { printf ' %sinfo %s%s\n' "$C_DIM" "$1" "$C_OFF"; } printf '\nsitereliability.sh/health.sh — %s, %s\n\n' "$(hostname 2>/dev/null || echo unknown-host)" "$(uname -sr)" # --- load ---------------------------------------------------------------- if [ -r /proc/loadavg ]; then LOAD1="$(cut -d' ' -f1 /proc/loadavg)" CORES="$(nproc 2>/dev/null || grep -c ^processor /proc/cpuinfo 2>/dev/null || echo 1)" # integer compare on load*100 vs cores*100 thresholds L100="$(awk -v l="$LOAD1" 'BEGIN{printf "%d", l*100}')" if [ "$L100" -gt $((CORES * 200)) ]; then bad "1-min load $LOAD1 is over 2x the $CORES core(s)" elif [ "$L100" -gt $((CORES * 100)) ]; then warn "1-min load $LOAD1 exceeds the $CORES core(s)" else ok "load $LOAD1 across $CORES core(s)" fi else info "uptime: $(uptime 2>/dev/null || echo unavailable)" fi # --- memory -------------------------------------------------------------- if [ -r /proc/meminfo ]; then AVAIL_PCT="$(awk '/MemTotal/{t=$2}/MemAvailable/{a=$2}END{if(t>0)printf "%d",a*100/t}' /proc/meminfo)" if [ "${AVAIL_PCT:-100}" -lt 10 ]; then bad "only ${AVAIL_PCT}% of memory available" elif [ "${AVAIL_PCT:-100}" -lt 20 ]; then warn "${AVAIL_PCT}% of memory available" else ok "${AVAIL_PCT}% of memory available"; fi SWAP_USED="$(awk '/SwapTotal/{t=$2}/SwapFree/{f=$2}END{if(t>0)printf "%d",(t-f)*100/t; else print "-"}' /proc/meminfo)" [ "$SWAP_USED" != "-" ] && [ "$SWAP_USED" -gt 50 ] && warn "swap is ${SWAP_USED}% used" fi # --- disk ---------------------------------------------------------------- # heredoc instead of a pipeline: a piped while runs in a subshell and the # WARN/FAIL counters would be lost # column count differs by platform and mount paths may contain spaces, so # locate the last N% field and treat everything after it as the mount path df_rows() { df -P "$1" 2>/dev/null | awk 'NR>1 && $1 ~ /^\// { p=0; for (i=NF; i>1; i--) if ($i ~ /^[0-9]+%$/) { p=i; break } if (!p || p==NF) next m=$(p+1); for (j=p+2; j<=NF; j++) m=m" "$j print $p, m }' } DISK_ROWS="$(df_rows -k)" while read -r pct mount; do used="${pct%%%}" case "$used" in ''|*[!0-9]*) continue ;; esac if [ "$used" -ge 90 ]; then bad "disk $mount is ${used}% full" elif [ "$used" -ge 80 ]; then warn "disk $mount is ${used}% full" else ok "disk $mount at ${used}%"; fi done </dev/null 2>&1; then FAILED="$(systemctl --failed --no-legend --plain 2>/dev/null | awk '{print $1}')" if [ -n "$FAILED" ]; then for u in $FAILED; do bad "systemd unit failed: $u"; done else ok "no failed systemd units" fi fi # --- pending reboot / updates ------------------------------------------- [ -f /var/run/reboot-required ] && warn "reboot required (kernel or libc update pending)" if command -v apt-get >/dev/null 2>&1 && [ -r /var/lib/update-notifier/updates-available ]; then SEC="$(grep -o '[0-9]* update.*security' /var/lib/update-notifier/updates-available 2>/dev/null | head -1)" [ -n "$SEC" ] && warn "pending: $SEC" fi # --- clock --------------------------------------------------------------- if command -v timedatectl >/dev/null 2>&1; then if timedatectl show -p NTPSynchronized --value 2>/dev/null | grep -q yes; then ok "clock is NTP-synchronized" else warn "clock is not NTP-synchronized — TLS and logs suffer when time drifts" fi fi # --- listening surface --------------------------------------------------- if command -v ss >/dev/null 2>&1; then LISTEN="$(ss -Htln 2>/dev/null | awk '{print $4}' | sed 's/.*://' | sort -un | tr '\n' ' ')" info "TCP ports listening: ${LISTEN:-none}" fi # --- summary ------------------------------------------------------------- printf '\n' if [ "$FAIL" -gt 0 ]; then printf '%sverdict: needs attention (%d fail, %d warn)%s\n' "$C_BAD" "$FAIL" "$WARN" "$C_OFF" elif [ "$WARN" -gt 0 ]; then printf '%sverdict: mostly healthy (%d warn)%s\n' "$C_WARN" "$WARN" "$C_OFF" else printf '%sverdict: healthy%s\n' "$C_OK" "$C_OFF"; fi printf '\n%sThis box looks after itself — but who watches your certs and domains?\nhttps://sitereliability.sh%s\n\n' "$C_DIM" "$C_OFF" [ "$FAIL" -eq 0 ]