#!/usr/bin/env bash
# Installs two operator commands on the node: emteria-status (quick health check) and emteria-restart.
#   curl ... install-tools.sh | bash
set -euo pipefail
install -d /usr/local/bin
cat > /usr/local/bin/emteria-status <<'EOS'
#!/usr/bin/env bash
# Quick health check of the on-prem emteria stack (~10 s). Exit 0 = all green.
K="/usr/local/bin/k3s kubectl"; D=idnow.vodafone.de; rc=0
ok()   { printf 'PASS  %-22s %s\n' "$1" "$2"; }
bad()  { printf 'FAIL  %-22s %s\n' "$1" "$2"; rc=1; }
systemctl is-active --quiet k3s && ok k3s "active since $(systemctl show k3s -p ActiveEnterTimestamp --value | cut -d' ' -f2-3)" || bad k3s "not active"
n=$($K get nodes --no-headers 2>/dev/null | awk '{print $2}'); [ "$n" = Ready ] && ok node Ready || bad node "$n"
c=$($K -n kube-system get ds cilium -o jsonpath='{.metadata.labels.app\.kubernetes\.io/version}' 2>/dev/null); [ -n "$c" ] || c=$($K -n kube-system exec ds/cilium -c cilium-agent -- cilium-dbg version 2>/dev/null | grep -oE 'Client: [0-9.]+' | cut -d' ' -f2); r=$($K -n kube-system get ds cilium -o jsonpath='{.status.numberReady}/{.status.desiredNumberScheduled}' 2>/dev/null); [ "$r" = "1/1" ] && ok cilium "$c ready $r" || bad cilium "$c ready $r"
a=$($K -n argocd get applications --no-headers 2>/dev/null | grep -cE 'Synced\s+Healthy'); t=$($K -n argocd get applications --no-headers 2>/dev/null | wc -l); [ "$a" = "$t" ] && [ "$t" -gt 0 ] && ok argo.apps "$a/$t Synced+Healthy" || { bad argo.apps "$a/$t"; $K -n argocd get applications --no-headers | grep -vE 'Synced\s+Healthy' | sed 's/^/        /'; }
p=$($K get pods -A --no-headers 2>/dev/null | grep -vE 'Running|Completed' | grep -v 'db-migrate'); [ -z "$p" ] && ok pods "all Running/Completed" || { bad pods "not running:"; echo "$p" | awk '{printf "        %s/%s %s\n",$1,$2,$4}'; }
e=$($K get externalsecrets -A --no-headers 2>/dev/null | grep -vc SecretSynced); [ "$e" = 0 ] && ok eso "all ExternalSecrets synced" || bad eso "$e not synced"
CA=$(mktemp); $K -n default get secret customer-ca-bundle -o jsonpath='{.data.ca-certificates\.crt}' 2>/dev/null | base64 -d > "$CA"
for h in api:/main/healthz hub:/; do host=${h%%:*}; path=${h#*:}; code=$(curl -s -o /dev/null -w '%{http_code}' --cacert "$CA" --resolve "$host.$D:443:127.0.0.1" "https://$host.$D$path" --max-time 8); [ "$code" = 200 ] && ok "https.$host" "$path -> 200" || bad "https.$host" "$path -> $code"; done
exp=$(openssl s_client -connect 127.0.0.1:443 -servername api.$D </dev/null 2>/dev/null | openssl x509 -noout -enddate 2>/dev/null | cut -d= -f2); days=$(( ( $(date -d "$exp" +%s 2>/dev/null || echo 0) - $(date +%s) ) / 86400 )); [ "$days" -gt 30 ] && ok tls.expiry "$days days left" || bad tls.expiry "$days days left ($exp)"
m=$(openssl s_client -connect 127.0.0.1:8883 -servername mdm.$D -CAfile "$CA" </dev/null 2>/dev/null | grep -E 'Verify return code' | sed 's/^ *//'); echo "$m" | grep -q ': 0 ' && ok mqtts "8883 handshake, $m" || bad mqtts "8883: ${m:-no handshake}"
rm -f "$CA"
rs=$($K -n argocd get pods -l app.kubernetes.io/name=argocd-repo-server --no-headers 2>/dev/null | awk '{print $4}'); [ "${rs:-0}" = 0 ] && ok argo.repo-server "0 restarts" || bad argo.repo-server "$rs restarts"
df -h -x overlay -x tmpfs -x devtmpfs -x shm -x nsfs 2>/dev/null | awk 'NR>1 && ($6=="/"||$6=="/data") {printf "INFO  %-22s %s used, %s free\n","disk "$6,$5,$4}'
[ $rc = 0 ] && echo "== all green ==" || echo "== problems above; full test: <bundle dir>/smoke.sh --ca /tmp/ca-bundle.pem =="
exit $rc
EOS
cat > /usr/local/bin/emteria-restart <<'EOS'
#!/usr/bin/env bash
# Restart the on-prem emteria stack. Levels:
#   emteria-restart apps   rolling restart of the application workloads in namespace default (services, website,
#                          mdm broker, seaweedfs, rabbitmq). Database (CNPG) and platform stay up. ~2 min, brief API outage.
#   emteria-restart k3s    systemctl restart k3s: control plane + kubelet restart, containers keep running. ~1-2 min API outage.
#   emteria-restart all    k3s-killall.sh (stops EVERY container) + start k3s: the full cold restart. ~3-5 min outage.
set -euo pipefail
K="/usr/local/bin/k3s kubectl"; L="${1:-}"
[ "$(id -u)" = 0 ] || { echo "run as root"; exit 1; }
case "$L" in
  apps)
    echo "rolling restart of workloads in namespace default"
    $K -n default get deploy,sts,ds -o name | sed 's/^/   /'
    read -r -p "proceed? [y/N] " a </dev/tty; [ "$a" = y ] || exit 1
    $K -n default rollout restart deploy,sts,ds
    for r in $($K -n default get deploy,sts,ds -o name); do $K -n default rollout status "$r" --timeout=300s; done ;;
  k3s)
    echo "systemctl restart k3s (containers keep running; API unavailable for a moment)"
    read -r -p "proceed? [y/N] " a </dev/tty; [ "$a" = y ] || exit 1
    systemctl restart k3s; sleep 20; until $K get nodes >/dev/null 2>&1; do sleep 5; done ;;
  all)
    echo "FULL restart: k3s-killall.sh stops every container, then k3s starts and re-creates them all"
    read -r -p "proceed? [y/N] " a </dev/tty; [ "$a" = y ] || exit 1
    /usr/local/bin/k3s-killall.sh; systemctl start k3s; sleep 30; until $K get nodes >/dev/null 2>&1; do sleep 5; done ;;
  *) sed -n '2,7p' "$0"; exit 1 ;;
esac
echo "waiting for the stack to settle..."; for i in $(seq 1 60); do a=$($K -n argocd get applications --no-headers 2>/dev/null | grep -cE 'Synced\s+Healthy' || true); [ "$a" -ge 14 ] && break; sleep 10; done
/usr/local/bin/emteria-status
EOS
chmod 755 /usr/local/bin/emteria-status /usr/local/bin/emteria-restart
# root shells on this box lack /usr/local/bin in PATH: add it for every user (login shells) instead of symlinking.
rm -f /usr/bin/emteria-status /usr/bin/emteria-restart
cat > /etc/profile.d/emteria-path.sh <<'EOP'
case ":$PATH:" in *:/usr/local/bin:*) ;; *) PATH="$PATH:/usr/local/bin" ;; esac
export PATH
EOP
chmod 644 /etc/profile.d/emteria-path.sh
echo "installed: emteria-status, emteria-restart (apps|k3s|all); PATH via /etc/profile.d/emteria-path.sh (new login shells; now: source /etc/profile.d/emteria-path.sh)"; /usr/local/bin/emteria-status
