#!/usr/bin/env bash set -euo pipefail # init_monitoring.sh # Purpose: # - Configure k3d environment for monitoring (Prometheus and Grafana) # - Setup kube-prometheus-stack and CNPG prometheus rules SCRIPT_DIR=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd) PROLE_ROOT=$(cd "$SCRIPT_DIR/.." && pwd) # Load environment and config via prole_cfg.sh # shellcheck disable=SC1090 source "$SCRIPT_DIR/prole_cfg.sh" if [[ "${1:-}" == "--mode" || "${1:-}" == "-m" ]]; then prole_set_mode "${2:-}" shift 2 elif [[ "${1:-}" == --mode=* || "${1:-}" == -m=* ]]; then prole_set_mode "${1#*=}" shift fi if [[ -z "${PROLE_SERVICE:-}" ]]; then echo "ERROR: PROLE_SERVICE is not defined. Provide PROLE_HOME/env.sh or ~/.prole/env.sh" >&2 exit 1 fi GRAFANA_RELEASE="grafana" LEGACY_GRAFANA_RELEASE="grafana-prole" log() { echo "==> $*" } err() { echo "ERROR: $*" >&2 } ensure_tools() { for t in helm kubectl curl jq; do command -v "$t" >/dev/null || { err "Missing required tool: $t"; exit 1; } done } ensure_namespace() { local ns="$1" if ! kubectl get namespace "$ns" >/dev/null 2>&1; then log "Creating namespace '$ns' ..." kubectl create namespace "$ns" >/dev/null 2>&1 || true fi } storage_class_exists() { local sc="$1" [[ -n "$sc" ]] || return 1 kubectl get storageclass "$sc" >/dev/null 2>&1 } default_storage_class() { kubectl get storageclass -o jsonpath='{range .items[?(@.metadata.annotations.storageclass\.kubernetes\.io/is-default-class=="true")]}{.metadata.name}{"\n"}{end}' 2>/dev/null | head -n1 } choose_monitoring_storage_class() { if [[ -n "${MONITORING_STORAGE_CLASS:-}" ]] && storage_class_exists "$MONITORING_STORAGE_CLASS"; then echo "$MONITORING_STORAGE_CLASS" return 0 fi if storage_class_exists "pi-local-iscsi"; then echo "pi-local-iscsi" return 0 fi local default_sc default_sc=$(default_storage_class) if [[ -n "$default_sc" ]]; then echo "$default_sc" return 0 fi if storage_class_exists "local-path"; then echo "local-path" return 0 fi echo "" } monitoring_nodes_available() { kubectl get nodes -l "prole.org/role=monitoring" -o name 2>/dev/null | grep -q . } render_node_selector() { local indent="$1" if [[ "${MONITORING_HAS_NODE_LABEL:-0}" == "1" ]]; then cat </dev/null 2>&1 || true continue fi if [[ -n "$expected_sc" && -n "$pvc_sc" && "$pvc_sc" != "$expected_sc" ]]; then log "Deleting Pending PVC '$pvc_name' with storageClass '$pvc_sc' (expected '$expected_sc') ..." kubectl -n "$ns" delete pvc "$pvc_name" >/dev/null 2>&1 || true fi done <<< "$pending" } apply_grafana_dashboard() { local ns="$1" local dashboard_path="" if [[ -n "${PROLE_GRAFANA_DASHBOARD_PATH:-}" ]]; then dashboard_path="$PROLE_GRAFANA_DASHBOARD_PATH" elif [[ -n "${PROLE_SERVICE:-}" && -f "$PROLE_SERVICE/prole-db/grafana-dashboard.json" ]]; then dashboard_path="$PROLE_SERVICE/prole-db/grafana-dashboard.json" elif [[ -f "$PROLE_ROOT/prole-db/grafana-dashboard.json" ]]; then dashboard_path="$PROLE_ROOT/prole-db/grafana-dashboard.json" fi if [[ -z "$dashboard_path" || ! -f "$dashboard_path" ]]; then log "Grafana dashboard not found; skipping." return 0 fi local tmp tmp=$(mktemp) sed 's/\\${DS_PROMETHEUS}/prometheus/g' "$dashboard_path" > "$tmp" kubectl -n "$ns" delete configmap prole-db-grafana-dashboard >/dev/null 2>&1 || true kubectl -n "$ns" create configmap prole-db-grafana-dashboard --from-file=prole-db.json="$tmp" >/dev/null kubectl -n "$ns" label configmap prole-db-grafana-dashboard grafana_dashboard=1 --overwrite >/dev/null rm -f "$tmp" } helm_release_status() { local release="$1" local ns="$2" helm status "$release" -n "$ns" -o json 2>/dev/null | jq -r '.info.status' 2>/dev/null || true } wait_for_helm_release() { local release="$1" local ns="$2" local timeout="${HELM_WAIT_TIMEOUT:-300}" local interval="${HELM_WAIT_INTERVAL:-5}" local start start=$(date +%s) while true; do local status status=$(helm_release_status "$release" "$ns") if [[ -z "$status" || "$status" == "null" ]]; then return 0 fi case "$status" in pending-*) if (( $(date +%s) - start > timeout )); then err "Timed out waiting for Helm release '$release' in '$ns' (status=$status)." return 1 fi log "Helm release '$release' is $status; waiting..." sleep "$interval" ;; *) return 0 ;; esac done } helm_upgrade_with_retry() { local release="$1" local ns="$2" local chart="$3" shift 3 local attempts="${HELM_UPGRADE_RETRIES:-5}" local delay="${HELM_RETRY_DELAY:-5}" local attempt out rc for ((attempt=1; attempt<=attempts; attempt++)); do wait_for_helm_release "$release" "$ns" || true set +e out=$(helm upgrade --install "$release" "$chart" --namespace "$ns" "$@" 2>&1) rc=$? set -e if [[ $rc -eq 0 ]]; then printf '%s\n' "$out" return 0 fi if echo "$out" | grep -q "another operation (install/upgrade/rollback) is in progress"; then log "Helm release '$release' is busy; retrying ($attempt/$attempts)..." wait_for_helm_release "$release" "$ns" || true sleep "$delay" continue fi echo "$out" >&2 return "$rc" done err "Helm upgrade failed after $attempts attempts for release '$release' in '$ns'." return 1 } openbao_url() { if [[ -n "${PROLE_OPENBAO_URL:-}" ]]; then echo "$PROLE_OPENBAO_URL" return 0 fi if prole_is_in_cluster; then echo "http://openbao.${SERVICE_NAMESPACE:-${NAMESPACE:-default}}.svc.cluster.local:8200" return 0 fi if curl -sS "http://127.0.0.1:8200/v1/sys/health" >/dev/null 2>&1; then echo "http://127.0.0.1:8200" return 0 elif curl -sS "http://127.0.0.1:18200/v1/sys/health" >/dev/null 2>&1; then echo "http://127.0.0.1:18200" return 0 else echo "" return 0 fi } openbao_token() { if [[ -f "$PROLE_SERVICE/secrets/openbao-root-token" ]]; then cat "$PROLE_SERVICE/secrets/openbao-root-token" else echo "${OPENBAO_ROOT_TOKEN:-}" fi } fetch_openbao_secret() { local path="$1" local key="$2" local token url token=$(openbao_token) url=$(openbao_url) if [[ -z "$token" || -z "$url" ]]; then echo "" return 0 fi curl -sS -H "X-Vault-Token: $token" "$url/v1/kv/data/$path" | jq -r ".data.data.\"$key\"" || echo "" } resolve_grafana_password() { if [[ -z "${GRAFANA_ADMIN_PASSWORD:-}" || "${GRAFANA_ADMIN_PASSWORD}" == '${OPENBAO:'* || "${GRAFANA_ADMIN_PASSWORD}" == '${PROLE_SECRET:'* ]]; then local fetched fetched=$(fetch_openbao_secret "prole/${NAMESPACE:-default}/monitoring" "grafana_admin_password") if [[ -n "$fetched" && "$fetched" != "null" ]]; then GRAFANA_ADMIN_PASSWORD="$fetched" fi fi if [[ -z "${GRAFANA_ADMIN_PASSWORD:-}" || "${GRAFANA_ADMIN_PASSWORD}" == '${OPENBAO:'* || "${GRAFANA_ADMIN_PASSWORD}" == '${PROLE_SECRET:'* ]]; then local db_pw="${DB_PASSWORD:-}" if [[ -z "$db_pw" || "$db_pw" == '${OPENBAO:'* || "$db_pw" == '${PROLE_SECRET:'* ]]; then local fetched_db fetched_db=$(fetch_openbao_secret "prole/${NAMESPACE:-default}/db" "password") if [[ -n "$fetched_db" && "$fetched_db" != "null" ]]; then db_pw="$fetched_db" fi fi if [[ -n "$db_pw" ]]; then GRAFANA_ADMIN_PASSWORD="$db_pw" fi fi } write_grafana_password_to_openbao() { local token url token=$(openbao_token) url=$(openbao_url) if [[ -z "$token" || -z "$url" || -z "${GRAFANA_ADMIN_PASSWORD:-}" ]]; then return 0 fi curl -sS -H "X-Vault-Token: $token" -H 'Content-Type: application/json' \ -X POST "$url/v1/kv/data/prole/${NAMESPACE:-default}/monitoring" \ -d "{\"data\":{\"grafana_admin_password\":\"$GRAFANA_ADMIN_PASSWORD\"}}" >/dev/null || true } cleanup_grafana_rbac_conflicts() { local release="$GRAFANA_RELEASE" local ns="$NAMESPACE" local cr="${release}-clusterrole" local crb="${release}-clusterrolebinding" local rel_ns rel_name if kubectl get clusterrole "$cr" >/dev/null 2>&1; then rel_ns=$(kubectl get clusterrole "$cr" -o jsonpath='{.metadata.annotations.meta\.helm\.sh/release-namespace}' 2>/dev/null || true) rel_name=$(kubectl get clusterrole "$cr" -o jsonpath='{.metadata.annotations.meta\.helm\.sh/release-name}' 2>/dev/null || true) if [[ -n "$rel_ns" && "$rel_ns" != "$ns" ]]; then log "Detected existing ClusterRole '$cr' owned by release '${rel_name:-unknown}' in namespace '$rel_ns'." if [[ -n "$rel_name" ]] && helm status "$rel_name" -n "$rel_ns" >/dev/null 2>&1; then log "Uninstalling old Grafana release '$rel_name' from '$rel_ns' ..." helm uninstall "$rel_name" -n "$rel_ns" || true fi if kubectl get clusterrole "$cr" >/dev/null 2>&1; then log "Deleting orphaned ClusterRole '$cr' ..." kubectl delete clusterrole "$cr" || true fi fi fi if kubectl get clusterrolebinding "$crb" >/dev/null 2>&1; then rel_ns=$(kubectl get clusterrolebinding "$crb" -o jsonpath='{.metadata.annotations.meta\.helm\.sh/release-namespace}' 2>/dev/null || true) rel_name=$(kubectl get clusterrolebinding "$crb" -o jsonpath='{.metadata.annotations.meta\.helm\.sh/release-name}' 2>/dev/null || true) if [[ -n "$rel_ns" && "$rel_ns" != "$ns" ]]; then log "Detected existing ClusterRoleBinding '$crb' owned by release '${rel_name:-unknown}' in namespace '$rel_ns'." if [[ -n "$rel_name" ]] && helm status "$rel_name" -n "$rel_ns" >/dev/null 2>&1; then log "Uninstalling old Grafana release '$rel_name' from '$rel_ns' ..." helm uninstall "$rel_name" -n "$rel_ns" || true fi if kubectl get clusterrolebinding "$crb" >/dev/null 2>&1; then log "Deleting orphaned ClusterRoleBinding '$crb' ..." kubectl delete clusterrolebinding "$crb" || true fi fi fi } cleanup_legacy_grafana_release() { local ns="$NAMESPACE" if helm status "$LEGACY_GRAFANA_RELEASE" -n "$ns" >/dev/null 2>&1; then log "Uninstalling legacy Grafana release '$LEGACY_GRAFANA_RELEASE' from '$ns' ..." helm uninstall "$LEGACY_GRAFANA_RELEASE" -n "$ns" || true fi } install_monitoring() { ensure_tools local monitoring_ns="monitoring" ensure_namespace "$monitoring_ns" MONITORING_HAS_NODE_LABEL=0 if monitoring_nodes_available; then MONITORING_HAS_NODE_LABEL=1 else log "No nodes labeled prole.org/role=monitoring; scheduling without nodeSelector/tolerations." fi MONITORING_STORAGE_CLASS_SELECTED=$(choose_monitoring_storage_class) if [[ -n "$MONITORING_STORAGE_CLASS_SELECTED" ]]; then log "Using storageClass '$MONITORING_STORAGE_CLASS_SELECTED' for monitoring PVCs." else log "No storageClass detected; disabling persistence for Grafana and skipping Prometheus/Alertmanager storage." fi cleanup_pending_pvcs "$monitoring_ns" "$MONITORING_STORAGE_CLASS_SELECTED" log "Installing kube-prometheus-stack in namespace '$monitoring_ns' ..." helm repo add prometheus-community https://prometheus-community.github.io/helm-charts || true helm repo update prometheus-community || true resolve_grafana_password if [[ -z "${GRAFANA_ADMIN_PASSWORD:-}" ]]; then GRAFANA_ADMIN_PASSWORD="admin" # Fallback fi local values_file values_file=$(mktemp) cat > "$values_file" </dev/null || true kubectl rollout status deployment/kps-grafana -n "$monitoring_ns" --timeout=60s 2>/dev/null || true log "Applying myrddin-node-exporter resources in namespace '$monitoring_ns'..." cat <