#!/usr/bin/env bash
# Post-install / post-update health check for a restricted-network tenant node.
# Runs ON the customer node, ships in the bundle. Touches the network zero
# times beyond the node itself.
#
#   sudo ./smoke.sh [--domain <d>] [--ca <file>] [--no-cluster] [--timeout N]
#
#   --domain <d>    base domain, e.g. idnow.vodafone.de. Normally read from the
#                   BUNDLE.txt beside this script.
#   --ca <file>     verify TLS against this CA instead of skipping verification.
#                   For a self-signed install that is /var/lib/emteria/selfsigned/tls.crt.
#   --no-cluster    HTTP/TLS checks only — no kubectl.
#   --timeout N     per-probe timeout in seconds (default 20).
#   --bundle <dir>  bundle directory holding BUNDLE.txt (default: beside this script).
#
# One line per check: PASS/FAIL/WARN/SKIP, the check name, the evidence.
# Exit 1 if anything FAILed; WARN never fails the run.
#
# MDM rows (broker = DaemonSet `mdm` in `default`, container `app`):
#   mqtt.tls         TLS handshake on 8883, certificate carries the mdm. SAN.
#   mqtt.connack     a real MQTT 3.1.1 CONNECT (anonymous, clean session) and
#                    its CONNACK. The broker answers 0x04 (bad credentials) to
#                    an anonymous client — that is PASS: it speaks MQTT and
#                    refuses strangers. No answer / garbage is FAIL.
#   mdm.cert-match   8883 presents the same certificate as 443 (SHA-256). A
#                    mismatch is a renewal that has not reached the broker yet;
#                    it reloads every 600 s, else rollout restart ds/mdm.
#   mdm.daemonset    desired == ready; WARN on a container restart < 1 h ago.
#   mdm.init         init container node-dns-setup Completed; FAIL when the pod
#                    is stuck in Init (bundle final8/9: the node-reader
#                    ClusterRoleBinding pointed at the wrong namespace).
#   mdm.rabbitmq     last 500 log lines: "RabbitMQ connection established" after
#                    the last connect error = PASS. "connection broke" every
#                    10 min is the broker's own idle reset, not an error. WARN
#                    when the startup Gateway announcement was unroutable
#                    (NO_ROUTE web.direct/gateway — no Gateway row follows).
#   mdm.errors       ERR/Error/Exception lines in the last 10 min (WARN).
#   mdm.gateway      the web DB Gateway row carries mdm.<domain>.
#   mdm.devices      informational: Device rows, active in 24 h, online now.
#
# Dependencies: bash, curl, openssl plus coreutils/iproute — the same set
# install.sh itself uses. No python, no jq — a check that cannot run on the
# customer's minimal Rocky image is not a check.
#
# NEVER PRINTS SECRETS. Response bodies are matched, never echoed; the TLS
# certificate is read for SANs, issuer and expiry (tls.crt, never tls.key).
set -uo pipefail

KUBECTL="${KUBECTL:-/usr/local/bin/k3s kubectl}"
TIMEOUT=20
DOMAIN=""
CA_FILE=""
NO_CLUSTER=0
BUNDLE_DIR=""

while [ $# -gt 0 ]; do
  case "$1" in
    --domain)     DOMAIN="${2:?--domain needs a base domain}"; shift 2 ;;
    --ca)         CA_FILE="${2:?--ca needs a file}"; shift 2 ;;
    --bundle)     BUNDLE_DIR="${2:?--bundle needs a directory}"; shift 2 ;;
    --no-cluster) NO_CLUSTER=1; shift ;;
    --timeout)    TIMEOUT="${2:?--timeout needs seconds}"; shift 2 ;;
    -h|--help)    sed -n '2,45p' "$0"; exit 0 ;;
    *) echo "unknown argument: $1" >&2; exit 1 ;;
  esac
done

# Same derivation as seed-secrets.sh and collect-support.sh: the bundle beside
# the script is the record of what this node actually runs. Asking the operator
# to remember the domain is how you end up probing the wrong one.
[ -n "$BUNDLE_DIR" ] || BUNDLE_DIR="$(cd "$(dirname "$0")" && pwd)"
BUNDLE_TXT="$BUNDLE_DIR/BUNDLE.txt"
CLUSTER=""
if [ -r "$BUNDLE_TXT" ]; then
  CLUSTER="$(awk '/^cluster:/{print $2}' "$BUNDLE_TXT")"
  [ -n "$DOMAIN" ] || DOMAIN="$(awk '/^domain:/{print $2}' "$BUNDLE_TXT")"
fi
[ -n "$DOMAIN" ] || { echo "ERROR: no domain — pass --domain or run from a bundle directory" >&2; exit 1; }

HUB="hub.$DOMAIN"; API="api.$DOMAIN"; MDM="mdm.$DOMAIN"

PASSES=0; FAILS=0; WARNS=0; SKIPS=0
row() { printf '%-4s  %-26s  %s\n' "$1" "$2" "${3:-}"; }
pass() { PASSES=$((PASSES+1)); row PASS "$1" "${2:-}"; }
fail() { FAILS=$((FAILS+1));   row FAIL "$1" "${2:-}"; }
warn() { WARNS=$((WARNS+1));   row WARN "$1" "${2:-}"; }
skip() { SKIPS=$((SKIPS+1));   row SKIP "$1" "${2:-}"; }
section() { printf '\n\033[1m── %s\033[0m\n' "$1"; }

WORK="$(mktemp -d)"
trap 'rm -rf "$WORK"' EXIT
BODY="$WORK/body"

# ── where to send the probes ─────────────────────────────────────────────────
# The public names are pinned to a local address with --resolve, exactly like
# collect-support.sh: this must work with no DNS, no egress and a certificate
# nothing trusts. Loopback first; some k3s/servicelb setups only answer on the
# node address, so fall back to the default-route source IP.
NODE_IP="$(ip -4 route get 1 2>/dev/null | awk '{for(i=1;i<=NF;i++) if($i=="src"){print $(i+1);exit}}')"
PROBE_IP=127.0.0.1
if ! timeout 5 bash -c "exec 3<>/dev/tcp/127.0.0.1/443" 2>/dev/null; then
  if [ -n "$NODE_IP" ] && timeout 5 bash -c "exec 3<>/dev/tcp/$NODE_IP/443" 2>/dev/null; then
    PROBE_IP="$NODE_IP"
  fi
fi

CURL_TLS=(-k)
if [ -n "$CA_FILE" ]; then
  [ -r "$CA_FILE" ] || { echo "ERROR: cannot read --ca $CA_FILE" >&2; exit 1; }
  CURL_TLS=(--cacert "$CA_FILE")
fi

# http <host> <path> -> prints the status code; body lands in $BODY
http() {
  local host="$1" path="$2" code
  : > "$BODY"
  code="$(curl -sS -o "$BODY" -w '%{http_code}' -m "$TIMEOUT" \
    --resolve "${host}:443:${PROBE_IP}" "${CURL_TLS[@]}" \
    "https://${host}${path}" 2>/dev/null)" || code=000
  printf '%s' "${code:-000}"
}

# The server certificate, as PEM on stdout. Port is a parameter because the
# MQTTS listener presents the very same certificate on 8883.
server_cert() {
  local host="$1" port="$2"
  timeout "$TIMEOUT" openssl s_client -connect "${PROBE_IP}:${port}" \
    -servername "$host" </dev/null 2>/dev/null | openssl x509 2>/dev/null
}

echo "══ smoke test ═════════════════════════════════════════════════════════════"
echo "   cluster: ${CLUSTER:-unknown}   domain: $DOMAIN   probe: $PROBE_IP"

# ── DNS ──────────────────────────────────────────────────────────────────────
section "DNS"
for host in "$HUB" "$API" "$MDM"; do
  ips="$(getent hosts "$host" 2>/dev/null | awk '{print $1}' | sort -u | tr '\n' ' ')"
  ips="${ips% }"
  if [ -z "$ips" ]; then
    fail "dns.${host%%.*}" "$host does not resolve — devices and browsers will not reach this node"
  elif [ -n "$NODE_IP" ] && ! printf '%s' " $ips " | grep -qF " $NODE_IP "; then
    # Not fatal from the node's own point of view (the probes use --resolve),
    # but it is what makes the install look broken from a desk.
    warn "dns.${host%%.*}" "$host -> $ips, not the node IP $NODE_IP"
  else
    pass "dns.${host%%.*}" "$host -> $ips"
  fi
done

# ── TLS ──────────────────────────────────────────────────────────────────────
section "TLS"
if server_cert "$API" 443 > "$WORK/tls.crt" && [ -s "$WORK/tls.crt" ]; then
  sans="$(openssl x509 -in "$WORK/tls.crt" -noout -ext subjectAltName 2>/dev/null \
          || openssl x509 -in "$WORK/tls.crt" -noout -text 2>/dev/null | grep -A1 'Subject Alternative Name')"
  issuer="$(openssl x509 -in "$WORK/tls.crt" -noout -issuer 2>/dev/null | sed 's/^issuer= *//')"
  missing=""
  for host in "$API" "$HUB" "$MDM"; do
    printf '%s' "$sans" | grep -q "DNS:$host\b" || missing="$missing $host"
  done
  if [ -n "$missing" ]; then
    # A missing mdm. SAN is the single most likely thing to be forgotten and
    # every device refuses MQTTS over it (clusters/vodafone/idnow README §2).
    fail "tls.sans" "missing SAN for:$missing  issuer=$issuer"
  else
    pass "tls.sans" "api/hub/mdm all present  issuer=$issuer"
  fi

  end="$(openssl x509 -in "$WORK/tls.crt" -noout -enddate 2>/dev/null | sed 's/^notAfter=//')"
  end_epoch="$(date -d "$end" +%s 2>/dev/null || echo 0)"
  now_epoch="$(date +%s)"
  if [ "$end_epoch" = 0 ]; then
    warn "tls.expiry" "could not parse notAfter ($end)"
  else
    days=$(( (end_epoch - now_epoch) / 86400 ))
    if [ "$days" -lt 0 ]; then
      fail "tls.expiry" "EXPIRED $(( -days )) days ago ($end)"
    elif [ "$days" -lt 30 ]; then
      warn "tls.expiry" "expires in $days days ($end) — renew: seed-secrets.sh --cert ... --force"
    else
      pass "tls.expiry" "$days days left ($end)"
    fi
  fi
else
  fail "tls.sans"   "no TLS handshake on ${PROBE_IP}:443"
  fail "tls.expiry" "no certificate to inspect"
fi

if [ -n "$CA_FILE" ]; then
  if [ "$(http "$API" /)" != 000 ]; then
    pass "tls.verify" "chain verifies against $CA_FILE"
  else
    fail "tls.verify" "chain does NOT verify against $CA_FILE"
  fi
else
  warn "tls.verify" "no --ca given: probes run with -k, certificate trust NOT checked"
fi

# ── hub ──────────────────────────────────────────────────────────────────────
section "hub"
code="$(http "$HUB" /)"
# No <title> to look for: the SPA sets document.title from /config.json after
# it boots. The index is identified by its mount point and its bundle instead.
if [ "$code" = 200 ] && grep -q 'id="root"' "$BODY" && grep -q '/assets/' "$BODY"; then
  pass "hub.root" "200, SPA index served"
else
  fail "hub.root" "GET https://$HUB/ -> $code (expected the SPA index)"
fi

# The SPA reads /config.json at boot and bails when apiGatewayUrl is empty or
# points somewhere unreachable (clusters/.../manifests/hub/configmap-config.yaml).
code="$(http "$HUB" /config.json)"
if [ "$code" != 200 ]; then
  fail "hub.config" "GET https://$HUB/config.json -> $code"
elif grep -q "\"apiGatewayUrl\"[[:space:]]*:[[:space:]]*\"https://$API\"" "$BODY"; then
  pass "hub.config" "apiGatewayUrl = https://$API"
else
  fail "hub.config" "config.json does not carry apiGatewayUrl = https://$API"
fi

# ── OIDC ─────────────────────────────────────────────────────────────────────
# website is mounted at /main, so its internal /.well-known/… surfaces there.
section "OIDC"
code="$(http "$API" /main/.well-known/openid-configuration)"
if [ "$code" = 200 ] && grep -q '"issuer"' "$BODY"; then
  pass "oidc.discovery" "200 with an issuer"
else
  fail "oidc.discovery" "GET /main/.well-known/openid-configuration -> $code"
fi

code="$(http "$API" /main/.well-known/jwks/)"
if [ "$code" = 200 ] && grep -q '"keys"' "$BODY"; then
  pass "oidc.jwks" "200 with a key set"
else
  # Every other backend validates tokens against this URL; empty here means
  # nothing accepts a login even though the login itself works.
  fail "oidc.jwks" "GET /main/.well-known/jwks/ -> $code"
fi

# The route table the hub reads at startup. Empty values here are a real
# failure mode on this class (commit dc7f565).
# mainservice is mounted at `/` on api.<domain>, so the table is at
# /v1/configs/routes — NOT under /main, which is website (verified on the lab,
# 2026-09-22: /main/api/v1/configs/routes is a 404).
code="$(http "$API" /v1/configs/routes)"
if [ "$code" != 200 ]; then
  fail "routes" "GET /v1/configs/routes -> $code"
elif grep -qi 'frontend' "$BODY" && grep -qi '"url"' "$BODY"; then
  pass "routes" "200, frontend entry carries a url"
else
  fail "routes" "200 but no frontend/url entry — the hub will not know where anything is"
fi

# The rehearsal on 2026-09-22 (bundle final4) returned 404 here because the
# OAuth reconciler had never created the emteria-spa client row: login is
# impossible and nothing else notices. Anything but 404 means the client exists.
q="client_id=emteria-spa&response_type=code&redirect_uri=https://$HUB/authorization/callback"
q="$q&code_challenge=E9Melhoa2OwvFrEMTJguCHaoeK1t8URWbuGJSstw-cM&code_challenge_method=S256&state=smoke"
code="$(http "$API" "/main/api/v3/tokens/authorization?$q")"
# Verified on the lab 2026-09-22: a known client answers 302 to the hub login,
# an unknown client_id answers 400. final4 answered 404 — either way the hub
# cannot log anybody in, and nothing else in the cluster notices.
SPA_CLIENT_OK=0  # read by oauth.reconciler once the Job itself is gone
case "$code" in
  200|302) pass "oauth.spa-client" "$code (client row present)"; SPA_CLIENT_OK=1 ;;
  400|404) fail "oauth.spa-client" "$code — the emteria-spa OpenIddict client does not exist; the hub cannot log anyone in" ;;
  000)     fail "oauth.spa-client" "no response from /main/api/v3/tokens/authorization" ;;
  *)       warn "oauth.spa-client" "$code (unexpected; expected 302 to https://$HUB/redirect?page=login)" ;;
esac

# ── per-service routing ──────────────────────────────────────────────────────
# /healthz is the chart's own probe path (charts/dotnet-web/templates/deployment.yaml)
# and the prefixed ingresses rewrite /<prefix>/healthz -> /healthz, so this
# reaches each pod through the real public path. 502/503/504 or no answer means
# the ingress rule or the pod behind it is broken; 401/404 still prove routing.
section "services through the ingress"
for svc in "mainservice:/healthz" "website:/main/healthz" "productmanager:/product/healthz" \
           "storagemanager:/storage/healthz" "streamservice:/stream/healthz" "eventmanager:/event/healthz"; do
  name="${svc%%:*}"; path="${svc#*:}"
  code="$(http "$API" "$path")"
  case "$code" in
    200|401|403|404) pass "svc.$name" "GET $path -> $code" ;;
    000)             fail "svc.$name" "GET $path -> no answer" ;;
    502|503|504)     fail "svc.$name" "GET $path -> $code (no healthy backend)" ;;
    *)               warn "svc.$name" "GET $path -> $code (unexpected, but routed)" ;;
  esac
done

# ── MQTTS ────────────────────────────────────────────────────────────────────
# Devices talk to the broker and nothing else; a cluster that is perfect on 443
# and dead on 8883 is a cluster with no fleet.
section "MQTTS"
if server_cert "$MDM" 8883 > "$WORK/mqtt.crt" && [ -s "$WORK/mqtt.crt" ]; then
  msans="$(openssl x509 -in "$WORK/mqtt.crt" -noout -ext subjectAltName 2>/dev/null \
           || openssl x509 -in "$WORK/mqtt.crt" -noout -text 2>/dev/null | grep -A1 'Subject Alternative Name')"
  if printf '%s' "$msans" | grep -q "DNS:$MDM\b"; then
    pass "mqtt.tls" "handshake on ${PROBE_IP}:8883, SAN carries $MDM"
  else
    fail "mqtt.tls" "handshake ok but the certificate has no $MDM SAN — devices will refuse MQTTS"
  fi
else
  fail "mqtt.tls" "no TLS handshake on ${PROBE_IP}:8883"
fi

# A TLS handshake only proves something terminates TLS on 8883. Speak MQTT:
# CONNECT (3.1.1, clean session, keepalive 10, no credentials), then read the
# 4-byte CONNACK. DISCONNECT after 2 s so an accepting broker closes the
# socket; timeout bounds everything else, so a hung read cannot stall the run.
cid="smoke-$$"
pkt="\\x10$(printf '\\x%02x' $(( 12 + ${#cid} )))\\x00\\x04MQTT\\x04\\x02\\x00\\x0a\\x00$(printf '\\x%02x' "${#cid}")$cid"
# shellcheck disable=SC2059  # $pkt is the escape-only format string built above
ack="$( { printf "$pkt"; sleep 2; printf '\xe0\x00'; } \
        | timeout 10 openssl s_client -connect "${PROBE_IP}:8883" -servername "$MDM" -quiet 2>/dev/null \
        | head -c 4 | od -An -tx1 | tr -s ' \n' ' ')"
ack="${ack# }"; ack="${ack% }"
case "$ack" in
  "20 02 "??" 00") pass "mqtt.connack" "CONNACK accepted (anonymous client allowed?)" ;;
  "20 02 "??" 04") pass "mqtt.connack" "CONNACK 0x04 bad credentials — broker speaks MQTT, refuses anonymous clients" ;;
  "20 02 "??" 05") pass "mqtt.connack" "CONNACK 0x05 not authorised — broker speaks MQTT, refuses anonymous clients" ;;
  "")              fail "mqtt.connack" "no CONNACK from ${PROBE_IP}:8883 within 10 s" ;;
  *)               fail "mqtt.connack" "unexpected reply to CONNECT: $ack" ;;
esac

# The broker reads the same emteria-tls Secret as the ingress but loads it
# itself, so after a renewal the two can differ until the broker reloads.
fp() { openssl x509 -in "$1" -noout -fingerprint -sha256 2>/dev/null | sed 's/^.*=//'; }
fp443="$( [ -s "$WORK/tls.crt" ] && fp "$WORK/tls.crt")"
fp8883="$( [ -s "$WORK/mqtt.crt" ] && fp "$WORK/mqtt.crt")"
if [ -z "$fp443" ] || [ -z "$fp8883" ]; then
  fail "mdm.cert-match" "cannot compare: no certificate from $([ -z "$fp443" ] && printf '443 ')$([ -z "$fp8883" ] && printf '8883')"
elif [ "$fp443" = "$fp8883" ]; then
  pass "mdm.cert-match" "8883 == 443 (sha256 ${fp443:0:23}…)"
else
  fail "mdm.cert-match" "8883 ${fp8883:0:23}… != 443 ${fp443:0:23}… — renewal not picked up; wait 10 min or rollout restart ds/mdm"
fi

# ── cluster ──────────────────────────────────────────────────────────────────
section "cluster"
if [ "$NO_CLUSTER" = 1 ]; then
  skip "cluster" "--no-cluster"
elif ! $KUBECTL version >/dev/null 2>&1; then
  skip "cluster" "no working kubeconfig ($KUBECTL) — run as root on the node, or pass --no-cluster"
else
  # Argo Applications
  apps="$($KUBECTL -n argocd get applications --no-headers 2>/dev/null)"
  if [ -z "$apps" ]; then
    fail "argo.apps" "no Applications in argocd — the bootstrap never landed"
  else
    total="$(printf '%s\n' "$apps" | grep -c .)"
    bad="$(printf '%s\n' "$apps" | grep -vE 'Synced[[:space:]]+Healthy' | awk '{print $1"("$2"/"$3")"}' | tr '\n' ' ')"
    if [ -n "$bad" ]; then
      fail "argo.apps" "$total Applications, not Synced+Healthy: $bad"
    else
      pass "argo.apps" "$total/$total Synced+Healthy"
    fi
  fi

  # Pods
  pods="$($KUBECTL get pods -A --no-headers 2>/dev/null)"
  if [ -z "$pods" ]; then
    fail "pods" "no pods at all"
  else
    # `Completed` is the STATUS a finished Job pod shows, and it is fine.
    # `Error`/`Failed` is usually the superseded attempt of a Job that then
    # succeeded — reported, not failed; the Job checks below are the authority.
    # A Running pod with READY 0/1 is not healthy either (failing readiness probe).
    badpods="$(printf '%s\n' "$pods" | awk '($4!~/^(Running|Completed|Succeeded)$/ && $4!~/^(Error|Failed)$/) || ($4=="Running" && split($3,r,"/") && r[1]!=r[2]) {print $1"/"$2"("$3" "$4")"}' | tr '\n' ' ')"
    deadpods="$(printf '%s\n' "$pods" | awk '$4~/^(Error|Failed)$/ {print $1"/"$2"("$4")"}' | tr '\n' ' ')"
    if [ -n "$badpods" ]; then
      fail "pods" "not Running/Completed: $badpods"
    else
      pass "pods" "$(printf '%s\n' "$pods" | grep -c .) pods Running/Completed"
    fi
    [ -z "$deadpods" ] || warn "pods.failed" "terminated with an error (check they were retried): $deadpods"
  fi

  # ExternalSecrets. A NotReady one means a pod is waiting on a Secret that
  # will never arrive, and the symptom shows up three layers away.
  es="$($KUBECTL get externalsecrets -A --no-headers 2>/dev/null)"
  if [ -z "$es" ]; then
    fail "eso.secrets" "no ExternalSecrets — the secret-store was never seeded"
  else
    bades="$(printf '%s\n' "$es" | grep -vi 'SecretSynced' | awk '{print $1"/"$2}' | tr '\n' ' ')"
    if [ -n "$bades" ]; then
      fail "eso.secrets" "not SecretSynced: $bades"
    else
      pass "eso.secrets" "$(printf '%s\n' "$es" | grep -c .) SecretSynced"
    fi
  fi

  # workarounds O9: argo-cd v3.5.3 answers /healthz?full=true in ~4 s on this
  # class, so a repo-server with restarts is the 1 s liveness timeout coming
  # back — and repo-server is the one pod the cluster cannot reconcile its way
  # out of being down.
  rs="$($KUBECTL -n argocd get pods -l app.kubernetes.io/name=argocd-repo-server \
        --no-headers 2>/dev/null | awk 'NR>0{n++; s+=$4} END {if(n) print s+0}')"
  if [ -z "$rs" ]; then
    warn "argo.repo-server" "pod not found"
  elif [ "$rs" -gt 0 ]; then
    fail "argo.repo-server" "$rs restarts — check repoServer.livenessProbe.timeoutSeconds (workarounds O9)"
  else
    pass "argo.repo-server" "0 restarts"
  fi

  # The Job that creates the OpenIddict client rows. oauth.spa-client above is
  # the symptom; this is the cause.
  # Argo re-creates this Job on sync, so a Failed one may still be retried into
  # a Complete one a minute later — which is why the row prints what it saw.
  # ttlSecondsAfterFinished=3600 deletes a finished Job after an hour, so on a
  # healthy cluster the Job is usually absent; then the emteria-spa client it
  # produces (oauth.spa-client) is the evidence.
  job="$($KUBECTL -n default get job oauth-client-reconciler --ignore-not-found -o name 2>/dev/null)"
  if [ -z "$job" ]; then
    if [ "$SPA_CLIENT_OK" = 1 ]; then
      pass "oauth.reconciler" "Job garbage-collected after ttlSecondsAfterFinished=3600; emteria-spa client present"
    else
      fail "oauth.reconciler" "Job oauth-client-reconciler absent and the emteria-spa client is missing — the reconciler never ran"
    fi
  else
    succ="$($KUBECTL -n default get job oauth-client-reconciler \
            -o jsonpath='{.status.succeeded}' 2>/dev/null)"
    jfail="$($KUBECTL -n default get job oauth-client-reconciler \
            -o jsonpath='{.status.failed}' 2>/dev/null)"
    case "${succ:-}" in
      ''|0|*[!0-9]*) fail "oauth.reconciler" "Job oauth-client-reconciler has not completed (succeeded=${succ:-none}, failed=${jfail:-0})" ;;
      *)             pass "oauth.reconciler" "Job oauth-client-reconciler completed ($succ)" ;;
    esac
  fi

fi

# ── MDM broker ───────────────────────────────────────────────────────────────
# Row semantics are in the header comment. The broker is the DaemonSet `mdm`
# (pods labelled app=mdm), the MQTT process is container `app`.
section "MDM"
if [ "$NO_CLUSTER" = 1 ]; then
  skip "mdm" "--no-cluster"
elif ! $KUBECTL version >/dev/null 2>&1; then
  skip "mdm" "no working kubeconfig"
else
  desired="$($KUBECTL -n default get ds mdm -o jsonpath='{.status.desiredNumberScheduled}' 2>/dev/null)"
  ready="$($KUBECTL -n default get ds mdm -o jsonpath='{.status.numberReady}' 2>/dev/null)"
  pod="$($KUBECTL -n default get pods -l app=mdm -o jsonpath='{.items[0].metadata.name}' 2>/dev/null)"
  restarts=0; lastfin=""
  if [ -n "$pod" ]; then
    restarts="$($KUBECTL -n default get pod "$pod" \
                -o jsonpath='{.status.containerStatuses[?(@.name=="app")].restartCount}' 2>/dev/null)"
    lastfin="$($KUBECTL -n default get pod "$pod" \
               -o jsonpath='{.status.containerStatuses[?(@.name=="app")].lastState.terminated.finishedAt}' 2>/dev/null)"
  fi
  restarts="${restarts:-0}"
  ago=""
  if [ -n "$lastfin" ]; then
    fin_epoch="$(date -d "$lastfin" +%s 2>/dev/null || echo 0)"
    ago=$(( ($(date +%s) - fin_epoch) / 60 ))
  fi
  if [ -z "$desired" ]; then
    fail "mdm.daemonset" "DaemonSet default/mdm not found"
  elif [ "$desired" = 0 ] || [ "${ready:-0}" != "$desired" ]; then
    fail "mdm.daemonset" "ready ${ready:-0}/$desired — no broker, no fleet (pod ${pod:-none})"
  elif [ -n "$ago" ] && [ "$ago" -lt 60 ]; then
    warn "mdm.daemonset" "ready $ready/$desired, but container app restarted $ago min ago ($restarts restarts total)"
  else
    when=""
    [ -n "$ago" ] && when=" (last $(( ago / 60 )) h ago)"
    pass "mdm.daemonset" "ready $ready/$desired, $restarts restarts$when"
  fi

  # node-dns-setup writes the node's FQDN label into the pod (serviceId source
  # nodeLabel) and needs the node-reader ClusterRoleBinding to do it. A wrong
  # binding leaves the pod in Init forever while nothing else fails loudly.
  if [ -z "$pod" ]; then
    fail "mdm.init" "no mdm pod"
  else
    init="$($KUBECTL -n default get pod "$pod" \
            -o jsonpath='{.status.initContainerStatuses[?(@.name=="node-dns-setup")].state}' 2>/dev/null)"
    case "$init" in
      *'"reason":"Completed"'*) pass "mdm.init" "node-dns-setup Completed" ;;
      '') fail "mdm.init" "no node-dns-setup status on $pod (init container renamed, or pod not scheduled)" ;;
      *)  why="$(printf '%s' "$init" | grep -o '"reason":"[^"]*"' | head -1)"
          fail "mdm.init" "node-dns-setup not Completed (${why:-running}) — $pod stuck in Init; logs -c node-dns-setup, check ClusterRoleBinding mdm-node-reader" ;;
    esac
  fi

  # Strip URL userinfo before a log line becomes evidence: RABBITMQ_URL
  # carries the password.
  scrub() { sed -E 's#://[^@/ ]+@#://***@#g' | cut -c1-160; }
  log="$WORK/mdm.log"
  $KUBECTL -n default logs ds/mdm -c app --tail=500 > "$log" 2>/dev/null || : > "$log"
  # "RabbitMQ connection broke" every 10 min is the broker's own idle reset
  # (always followed by "established"), so it is deliberately not in here.
  errre='BrokerUnreachable|[Cc]onnection refused|[Cc]onnection reset|No route to host|ACCESS_REFUSED|AuthenticationFailure|Name or service not known'
  last_ok="$(grep -n 'RabbitMQ connection established' "$log" | tail -1 | cut -d: -f1)"
  last_err="$(grep -nE "$errre" "$log" | tail -1 | cut -d: -f1)"
  nerr="$(grep -cE "$errre" "$log")"
  if [ ! -s "$log" ]; then
    warn "mdm.rabbitmq" "no log from ds/mdm -c app"
  elif [ -n "$last_err" ] && { [ -z "$last_ok" ] || [ "$last_err" -gt "$last_ok" ]; }; then
    if [ "$nerr" -ge 2 ]; then
      fail "mdm.rabbitmq" "$nerr connection errors, no reconnect since: $(grep -E "$errre" "$log" | tail -1 | scrub)"
    else
      warn "mdm.rabbitmq" "connection error after the last reconnect: $(grep -E "$errre" "$log" | tail -1 | scrub)"
    fi
  elif grep -q "NO_ROUTE.*'gateway'" "$log"; then
    # Only visible while the startup lines are within the last 500 (~14 h at
    # the idle-reset rate on a quiet broker).
    # Rehearsal 2026-09-23: mdm announces itself on web.direct/gateway once at
    # startup; with no queue bound yet the message is dropped and no Gateway
    # row appears (mdm.gateway below).
    warn "mdm.rabbitmq" "connected, but the startup Gateway announcement was unroutable (NO_ROUTE web.direct/gateway)"
  elif [ -n "$last_ok" ]; then
    if [ "$nerr" -gt 0 ]; then
      pass "mdm.rabbitmq" "connection established ($nerr earlier errors, recovered)"
    else
      pass "mdm.rabbitmq" "connection established"
    fi
  else
    warn "mdm.rabbitmq" "last 500 log lines show neither a connection nor an error"
  fi

  errs="$($KUBECTL -n default logs ds/mdm -c app --since=10m 2>/dev/null \
          | grep -E '\[[0-9:]+ (ERR|FTL)\]|Error|Exception')"
  if [ -n "$errs" ]; then
    warn "mdm.errors" "$(printf '%s\n' "$errs" | grep -c .) error lines in 10 min, first: $(printf '%s\n' "$errs" | head -1 | scrub)"
  else
    pass "mdm.errors" "no ERR/Exception lines in the last 10 min"
  fi

  # MDM Gateway row. The backend creates it from a gateway announcement on the
  # RabbitMQ queue; in the 2026-09-22 rehearsal it was still absent after a
  # fresh install and after a broker restart, and what triggers the
  # announcement was never established. Reported, never failed.
  gw="$($KUBECTL -n default exec apps-db-1 -c postgres -- \
        psql -qtAd web -c 'SELECT "Uri" FROM "Gateway" LIMIT 1;' 2>/dev/null | tr -d ' ')"
  if [ -n "$gw" ]; then
    if [ "$gw" = "$MDM" ]; then
      pass "mdm.gateway" "Gateway.Uri = $gw"
    else
      fail "mdm.gateway" "Gateway.Uri = $gw, expected $MDM — every device gets an unresolvable broker address"
    fi
  else
    skip "mdm.gateway" "no Gateway row yet; the trigger is unknown (README post-install) — not a failure at install time"
  fi
  # Informational only: whether the fleet has found the broker, never whether
  # the install is healthy.
  dev="$($KUBECTL -n default exec apps-db-1 -c postgres -- psql -qtAd web -F ' ' -c \
         "SELECT count(*), count(*) FILTER (WHERE \"LastActivityDate\" > (now() AT TIME ZONE 'UTC') - interval '24 hours'), count(*) FILTER (WHERE \"IsOnline\") FROM \"Device\";" 2>/dev/null)"
  if [ -n "$dev" ]; then
    read -r dtotal dactive donline <<< "$dev"
    pass "mdm.devices" "$dtotal devices, $dactive active in 24 h, $donline online now"
  else
    skip "mdm.devices" "could not query the Device table"
  fi
fi

# ── summary ──────────────────────────────────────────────────────────────────
echo
echo "══ $PASSES passed, $FAILS failed, $WARNS warnings, $SKIPS skipped ══"
if [ "$FAILS" -gt 0 ]; then
  echo "Send us a support bundle: sudo $BUNDLE_DIR/collect-support.sh"
  exit 1
fi
exit 0
