prole/etc/init_monitoring.sh
chrisfu da2f6600ba feat: infrastructure and installer updates for k3s, OpenTofu, and prole-db
- Add k3s start/stop Ansible playbooks and roles.

- Implement OpenTofu initialization scripts and k8s manifests.

- Update ncurses installer with OpenTofu support and improved k3s integration.

- Add mode support (--mode) to etc/ initialization scripts.

- Update prole-db with recovery, barman objectstore, and SSH OpenBao support.

- Refine k8s manifests for OpenBao and prole-db.
2026-02-05 21:27:18 -08:00

243 lines
8.9 KiB
Bash
Executable File

#!/usr/bin/env bash
set -euo pipefail
# init_monitoring.sh
# Purpose:
# - Configure k3d environment for monitoring (Prometheus and Grafana)
# - Setup kube-prometheus-stack and CNPG prometheus rules
SCRIPT_DIR=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)
# Load environment and config via prole_cfg.sh
# shellcheck disable=SC1090
source "$SCRIPT_DIR/prole_cfg.sh"
if [[ "${1:-}" == "--mode" || "${1:-}" == "-m" ]]; then
prole_set_mode "${2:-}"
shift 2
elif [[ "${1:-}" == --mode=* || "${1:-}" == -m=* ]]; then
prole_set_mode "${1#*=}"
shift
fi
if [[ -z "${PROLE_SERVICE:-}" ]]; then
echo "ERROR: PROLE_SERVICE is not defined. Provide PROLE_HOME/env.sh or ~/.prole/env.sh" >&2
exit 1
fi
log() {
echo "==> $*"
}
err() {
echo "ERROR: $*" >&2
}
ensure_tools() {
for t in helm kubectl curl jq; do
command -v "$t" >/dev/null || { err "Missing required tool: $t"; exit 1; }
done
}
ensure_namespace() {
local ns="$1"
if ! kubectl get namespace "$ns" >/dev/null 2>&1; then
log "Creating namespace '$ns' ..."
kubectl create namespace "$ns" >/dev/null 2>&1 || true
fi
}
openbao_url() {
if [[ -n "${PROLE_OPENBAO_URL:-}" ]]; then
echo "$PROLE_OPENBAO_URL"
elif curl -sS "http://127.0.0.1:18200/v1/sys/health" >/dev/null 2>&1; then
echo "http://127.0.0.1:18200"
else
echo "http://openbao.${NAMESPACE}.svc.cluster.local:8200"
fi
}
openbao_token() {
if [[ -f "$PROLE_SERVICE/secrets/openbao-root-token" ]]; then
cat "$PROLE_SERVICE/secrets/openbao-root-token"
else
echo "${OPENBAO_ROOT_TOKEN:-}"
fi
}
fetch_openbao_secret() {
local path="$1"
local key="$2"
local token url
token=$(openbao_token)
url=$(openbao_url)
if [[ -z "$token" ]]; then
echo ""
return 0
fi
curl -sS -H "X-Vault-Token: $token" "$url/v1/kv/data/$path" | jq -r ".data.data.\"$key\"" || echo ""
}
resolve_grafana_password() {
if [[ -z "${GRAFANA_ADMIN_PASSWORD:-}" || "${GRAFANA_ADMIN_PASSWORD}" == '${OPENBAO:'* || "${GRAFANA_ADMIN_PASSWORD}" == '${PROLE_SECRET:'* ]]; then
local fetched
fetched=$(fetch_openbao_secret "prole/${NAMESPACE:-default}/monitoring" "grafana_admin_password")
if [[ -n "$fetched" && "$fetched" != "null" ]]; then
GRAFANA_ADMIN_PASSWORD="$fetched"
fi
fi
if [[ -z "${GRAFANA_ADMIN_PASSWORD:-}" ]]; then
local db_pw="${DB_PASSWORD:-}"
if [[ -z "$db_pw" || "$db_pw" == '${OPENBAO:'* || "$db_pw" == '${PROLE_SECRET:'* ]]; then
local fetched_db
fetched_db=$(fetch_openbao_secret "prole/${NAMESPACE:-default}/db" "password")
if [[ -n "$fetched_db" && "$fetched_db" != "null" ]]; then
db_pw="$fetched_db"
fi
fi
if [[ -n "$db_pw" ]]; then
GRAFANA_ADMIN_PASSWORD="$db_pw"
fi
fi
}
write_grafana_password_to_openbao() {
local token url
token=$(openbao_token)
url=$(openbao_url)
if [[ -z "$token" || -z "${GRAFANA_ADMIN_PASSWORD:-}" ]]; then
return 0
fi
curl -sS -H "X-Vault-Token: $token" -H 'Content-Type: application/json' \
-X POST "$url/v1/kv/data/prole/${NAMESPACE:-default}/monitoring" \
-d "{\"data\":{\"grafana_admin_password\":\"$GRAFANA_ADMIN_PASSWORD\"}}" >/dev/null || true
}
cleanup_grafana_rbac_conflicts() {
local release="grafana-prole"
local ns="$NAMESPACE"
local cr="${release}-clusterrole"
local crb="${release}-clusterrolebinding"
local rel_ns rel_name
if kubectl get clusterrole "$cr" >/dev/null 2>&1; then
rel_ns=$(kubectl get clusterrole "$cr" -o jsonpath='{.metadata.annotations.meta\.helm\.sh/release-namespace}' 2>/dev/null || true)
rel_name=$(kubectl get clusterrole "$cr" -o jsonpath='{.metadata.annotations.meta\.helm\.sh/release-name}' 2>/dev/null || true)
if [[ -n "$rel_ns" && "$rel_ns" != "$ns" ]]; then
log "Detected existing ClusterRole '$cr' owned by release '${rel_name:-unknown}' in namespace '$rel_ns'."
if [[ -n "$rel_name" ]] && helm status "$rel_name" -n "$rel_ns" >/dev/null 2>&1; then
log "Uninstalling old Grafana release '$rel_name' from '$rel_ns' ..."
helm uninstall "$rel_name" -n "$rel_ns" || true
fi
if kubectl get clusterrole "$cr" >/dev/null 2>&1; then
log "Deleting orphaned ClusterRole '$cr' ..."
kubectl delete clusterrole "$cr" || true
fi
fi
fi
if kubectl get clusterrolebinding "$crb" >/dev/null 2>&1; then
rel_ns=$(kubectl get clusterrolebinding "$crb" -o jsonpath='{.metadata.annotations.meta\.helm\.sh/release-namespace}' 2>/dev/null || true)
rel_name=$(kubectl get clusterrolebinding "$crb" -o jsonpath='{.metadata.annotations.meta\.helm\.sh/release-name}' 2>/dev/null || true)
if [[ -n "$rel_ns" && "$rel_ns" != "$ns" ]]; then
log "Detected existing ClusterRoleBinding '$crb' owned by release '${rel_name:-unknown}' in namespace '$rel_ns'."
if [[ -n "$rel_name" ]] && helm status "$rel_name" -n "$rel_ns" >/dev/null 2>&1; then
log "Uninstalling old Grafana release '$rel_name' from '$rel_ns' ..."
helm uninstall "$rel_name" -n "$rel_ns" || true
fi
if kubectl get clusterrolebinding "$crb" >/dev/null 2>&1; then
log "Deleting orphaned ClusterRoleBinding '$crb' ..."
kubectl delete clusterrolebinding "$crb" || true
fi
fi
fi
}
install_monitoring() {
ensure_tools
local monitoring_ns="default"
ensure_namespace "$monitoring_ns"
ensure_namespace "$NAMESPACE"
log "Installing kube-prometheus-stack in namespace '$monitoring_ns' ..."
helm repo add prometheus-community https://prometheus-community.github.io/helm-charts || true
helm repo update prometheus-community || true
# Install Prometheus stack in default namespace, but disable Grafana there
# Also set grafana.enabled=false explicitly to avoid conflicts if it was previously enabled.
# Use --force-conflicts with Server-Side Apply (SSA) to handle webhook conflicts.
# SSA is more robust for managing shared resources like webhooks.
helm upgrade --install \
--namespace "$monitoring_ns" \
--set grafana.enabled=false \
--force-conflicts \
--server-side=true \
-f https://raw.githubusercontent.com/cloudnative-pg/cloudnative-pg/main/docs/src/samples/monitoring/kube-stack-config.yaml \
prometheus-community prometheus-community/kube-prometheus-stack
log "Installing Grafana in namespace '$NAMESPACE' ..."
# We use the same chart but only for Grafana, or we could use the standalone grafana chart.
# Using the same chart ensures we can use the same config if needed, but we must avoid ClusterRole conflicts.
# Actually, standalone grafana chart is cleaner if we only want Grafana.
helm repo add grafana https://grafana.github.io/helm-charts || true
helm repo update grafana || true
local prometheus_svc="http://prometheus-community-kube-prometheus.$monitoring_ns.svc.cluster.local:9090"
cleanup_grafana_rbac_conflicts
resolve_grafana_password
if [[ -z "${GRAFANA_ADMIN_PASSWORD:-}" ]]; then
err "GRAFANA_ADMIN_PASSWORD is empty. Set it or ensure DB_PASSWORD is available."
exit 1
fi
write_grafana_password_to_openbao
# Note: kube-prometheus-stack may have already been installed with Grafana enabled in 'default'.
# If we want to move Grafana to $NAMESPACE, we install it there.
# We use the grafana/grafana chart for the per-namespace instance.
# Use --force-conflicts with Server-Side Apply (SSA) to handle potential conflicts during upgrade.
helm upgrade --install \
--namespace "$NAMESPACE" \
--force-conflicts \
--server-side=true \
--set "rbac.namespaced=true" \
--set "persistence.enabled=true" \
--set "persistence.size=5Gi" \
--set "datasources.datasources\.yaml.apiVersion=1" \
--set "datasources.datasources\.yaml.datasources[0].name=Prometheus" \
--set "datasources.datasources\.yaml.datasources[0].type=prometheus" \
--set "datasources.datasources\.yaml.datasources[0].url=$prometheus_svc" \
--set "datasources.datasources\.yaml.datasources[0].access=proxy" \
--set "datasources.datasources\.yaml.datasources[0].isDefault=true" \
--set "adminPassword=$GRAFANA_ADMIN_PASSWORD" \
grafana-prole grafana/grafana
log "Applying CNPG prometheus rules in namespace '$NAMESPACE'..."
kubectl apply --namespace "$NAMESPACE" -f https://raw.githubusercontent.com/cloudnative-pg/cloudnative-pg/main/docs/src/samples/monitoring/prometheusrule.yaml
log "Retrieving Grafana admin password from namespace '$NAMESPACE'..."
# Secret name is 'grafana-prole' from the helm release name
GRAFANA_PASSWORD=$(kubectl --namespace "$NAMESPACE" get secrets grafana-prole -o jsonpath="{.data.admin-password}" 2>/dev/null | base64 -d || true)
if [[ -n "$GRAFANA_PASSWORD" ]]; then
log "Grafana installation password: $GRAFANA_PASSWORD"
# We will save this to prole.cfg via the installer, but also output it here for logs
echo "GRAFANA_ADMIN_PASSWORD=$GRAFANA_PASSWORD"
else
err "Failed to retrieve Grafana admin password."
fi
}
case "${1:-}" in
initialize)
install_monitoring
;;
*)
echo "Usage: $0 initialize"
exit 1
;;
esac