prole/etc/ensure_default_storage_class.sh
chrisfu e44ff8e476 fix(gitlab): unblock standalone reconcile + OIDC + object storage
Enables iterating on GitLab reconciles without a full deploy.sh run and
unblocks the end-to-end login flow:

- etc/init_gitlab.sh
  * Pre-scan $@ for --config before sourcing prole_cfg.sh so
    PROLE_DEPLOY_CFG is set for standalone invocations; previously
    prole_cfg.sh auto-picked conf/k3d.cfg and leaked KUBECONTEXT=dev,
    failing with `error: context "dev" does not exist`.
  * _init_gitlab_resolve_secretref: bash mirror of
    knoe/core/actions.py:_resolve_secretref_value so
    secretref://google-oidc-client-* resolves from etc/secrets/ files
    when run outside deploy.sh Python env-injection.
  * Garage bucket array: drop -storage suffix from uploads/artifacts/
    lfs/packages/dependency-proxy. Chart defaults for these have no
    suffix, and the object_store block in the CR does not override
    per-object bucket names; the prior mismatch caused first-login 500s
    (NoSuchBucket on avatar PUT).
- deploy/gcp/gke/gitlab-google-oidc-secret.example.yaml
  * Add discovery: true so omniauth-openid_connect fetches Google
    .well-known/openid-configuration; fixes "Could not authenticate
    from OpenIDConnect: No host info" on the callback.
- conf/gke.cfg
  * GITLAB_WEBSERVICE_LIMITS_MEMORY 1800M -> 3Gi, REQUESTS_MEMORY
    900M -> 2Gi, REQUESTS_CPU 200m -> 500m. Live pod was sitting at
    1706Mi/1800M (99%) in OOMKilled loop.
- conf/port-mapping.cfg: add supabase + gitea forwards, fix postgres
  namespace knoe-db -> knoe-db-0 for split-cluster CNPG layout.

Adds read-only diagnostics used to chase the above:
- etc/diag_gitlab_boot.sh, diag_gitlab_webservice_oom.sh, diag_gke_storage.sh
- etc/ensure_default_storage_class.sh + k8s/prole/storageclass-gcp-standard-hdd.yaml
  (preflight + HDD-default SC manifest for SSD-quota-constrained GKE projects).

Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>
2026-04-20 18:51:41 -07:00

256 lines
9.8 KiB
Bash
Executable File

#!/usr/bin/env bash
# ensure_default_storage_class.sh
#
# Preflight / remediator that asserts the cluster-wide default StorageClass is
# pd-standard (HDD), not pd-balanced / pd-ssd. GKE ships `standard-rwo` as the
# default, which provisions `pd-balanced` under the hood and draws from the
# SSD_TOTAL_GB quota. That has repeatedly wedged provisioning on low-SSD-quota
# projects (see init_cnpg_gke.sh step-down logic, which compensates on the
# consumer side).
#
# This script fixes the root cause cluster-side: it un-defaults any built-in
# SSD-backed class and promotes `standard-hdd` (pd-standard) as the default.
# Workloads that actually need SSD latency (CNPG) opt in explicitly via
# `storageClassName: premium-rwo` in their PVC template — they are unaffected
# by the default change.
#
# Usage
# -----
# ./etc/ensure_default_storage_class.sh [--check | --apply] [--kube-context CTX]
#
# --check (default) Report current state; exit 0 if compliant, 1 if drift
# detected. No cluster mutations.
# --apply Remediate: create standard-hdd if absent, clear the
# default annotation from any SSD-backed class, set standard-hdd
# as the default. Idempotent.
# --kube-context CTX Use the named kube-context (default: current).
#
# Exit codes
# ----------
# 0 Compliant (or successfully remediated)
# 1 Drift detected in --check mode
# 2 Remediation failed (in --apply mode)
# 3 Cluster not a GKE cluster (no pd.csi.storage.gke.io provisioner seen)
# 4 Invalid arguments / missing tooling
set -euo pipefail
SCRIPT_NAME="$(basename "${BASH_SOURCE[0]}")"
DEFAULT_SC_NAME="${PROLE_DEFAULT_SC_NAME:-standard-hdd}"
DEFAULT_SC_DISK_TYPE="pd-standard"
GKE_CSI_PROVISIONER="pd.csi.storage.gke.io"
MODE="check"
KUBE_CONTEXT=""
log() { printf '[%s] %s\n' "$SCRIPT_NAME" "$*"; }
warn() { printf '[%s] WARN: %s\n' "$SCRIPT_NAME" "$*" >&2; }
die() { printf '[%s] ERROR: %s\n' "$SCRIPT_NAME" "$*" >&2; exit "${2:-4}"; }
while [[ $# -gt 0 ]]; do
case "$1" in
--check) MODE="check"; shift ;;
--apply) MODE="apply"; shift ;;
--kube-context)
[[ -n "${2:-}" ]] || die "--kube-context requires an argument" 4
KUBE_CONTEXT="$2"; shift 2 ;;
-h|--help)
sed -n '1,40p' "${BASH_SOURCE[0]}" | sed 's/^# \{0,1\}//'
exit 0 ;;
*) die "unknown argument: $1" 4 ;;
esac
done
command -v kubectl >/dev/null || die "kubectl not found in PATH" 4
command -v python3 >/dev/null || die "python3 not found in PATH" 4
kctl() {
if [[ -n "$KUBE_CONTEXT" ]]; then
kubectl --context "$KUBE_CONTEXT" "$@"
else
kubectl "$@"
fi
}
# ── 1. Confirm we're talking to a GKE cluster ────────────────────────────────
sc_json="$(kctl get storageclass -o json 2>/dev/null || true)"
if [[ -z "$sc_json" ]] || ! printf '%s' "$sc_json" | python3 -c '
import json, sys
data = json.load(sys.stdin)
items = data.get("items") or []
sys.exit(0 if any(it.get("provisioner") == "pd.csi.storage.gke.io" for it in items) else 1)
'; then
die "no ${GKE_CSI_PROVISIONER} StorageClasses found; this does not look like a GKE cluster" 3
fi
# ── 2. Inspect current state ─────────────────────────────────────────────────
# Emits: <name>|<provisioner>|<type>|<is_default> for every StorageClass.
sc_summary="$(printf '%s' "$sc_json" | python3 - <<'PY'
import json, sys
data = json.load(sys.stdin)
for it in data.get("items") or []:
meta = it.get("metadata") or {}
name = meta.get("name", "")
ann = (meta.get("annotations") or {})
is_default = str(ann.get("storageclass.kubernetes.io/is-default-class", "") or "").lower() == "true"
prov = it.get("provisioner", "") or ""
disk_type = ((it.get("parameters") or {}).get("type", "")) or ""
print(f"{name}|{prov}|{disk_type}|{'true' if is_default else 'false'}")
PY
)"
log "Current StorageClasses:"
while IFS='|' read -r name prov disk_type is_default; do
[[ -n "$name" ]] || continue
if [[ "$is_default" == "true" ]]; then
log " * ${name} prov=${prov} type=${disk_type} (default)"
else
log " ${name} prov=${prov} type=${disk_type}"
fi
done <<< "$sc_summary"
# Identify the current default + its disk type.
current_default=""
current_default_type=""
ssd_backed_defaults=()
while IFS='|' read -r name prov disk_type is_default; do
[[ -n "$name" ]] || continue
if [[ "$is_default" == "true" ]]; then
current_default="$name"
current_default_type="$disk_type"
if [[ "$disk_type" == "pd-balanced" || "$disk_type" == "pd-ssd" ]]; then
ssd_backed_defaults+=("$name")
fi
fi
done <<< "$sc_summary"
# Does standard-hdd (or PROLE_DEFAULT_SC_NAME override) exist with the right shape?
target_exists="false"
target_disk_type=""
target_is_default="false"
while IFS='|' read -r name prov disk_type is_default; do
if [[ "$name" == "$DEFAULT_SC_NAME" ]]; then
target_exists="true"
target_disk_type="$disk_type"
target_is_default="$is_default"
fi
done <<< "$sc_summary"
compliant="true"
issues=()
if [[ "$target_exists" != "true" ]]; then
compliant="false"
issues+=("target StorageClass '${DEFAULT_SC_NAME}' does not exist")
elif [[ "$target_disk_type" != "$DEFAULT_SC_DISK_TYPE" ]]; then
compliant="false"
issues+=("'${DEFAULT_SC_NAME}' has type='${target_disk_type}', want '${DEFAULT_SC_DISK_TYPE}'")
fi
if [[ "${#ssd_backed_defaults[@]}" -gt 0 ]]; then
compliant="false"
issues+=("SSD-backed class(es) still marked default: ${ssd_backed_defaults[*]}")
fi
if [[ "$target_exists" == "true" && "$target_is_default" != "true" ]]; then
compliant="false"
issues+=("'${DEFAULT_SC_NAME}' is not annotated is-default-class=true")
fi
if [[ "$compliant" == "true" ]]; then
log "OK — default StorageClass is '${current_default}' (type=${current_default_type}). No action needed."
exit 0
fi
log "Drift detected:"
for issue in "${issues[@]}"; do
log " - ${issue}"
done
if [[ "$MODE" == "check" ]]; then
log "Run with --apply to remediate."
exit 1
fi
# ── 3. Remediate ─────────────────────────────────────────────────────────────
log "Applying remediation ..."
# 3a. Create/update standard-hdd — but NOT yet marked default so we never have
# a window with two defaults racing.
cat <<EOF | kctl apply -f -
apiVersion: storage.k8s.io/v1
kind: StorageClass
metadata:
name: ${DEFAULT_SC_NAME}
annotations:
storageclass.kubernetes.io/is-default-class: "false"
provisioner: ${GKE_CSI_PROVISIONER}
parameters:
type: ${DEFAULT_SC_DISK_TYPE}
reclaimPolicy: Retain
volumeBindingMode: WaitForFirstConsumer
allowVolumeExpansion: true
EOF
# Verify the live class has the right disk type. If `parameters` drifted on an
# existing class, the apply above would silently no-op (parameters are
# immutable). Catch that and hard-fail with a clear message.
live_type="$(kctl get storageclass "$DEFAULT_SC_NAME" -o jsonpath='{.parameters.type}' 2>/dev/null || true)"
if [[ "$live_type" != "$DEFAULT_SC_DISK_TYPE" ]]; then
warn "live '${DEFAULT_SC_NAME}' has type='${live_type}' (want '${DEFAULT_SC_DISK_TYPE}'). Recreating ..."
kctl delete storageclass "$DEFAULT_SC_NAME" --wait=true >/dev/null 2>&1 || die "failed to delete '${DEFAULT_SC_NAME}' for recreation" 2
cat <<EOF | kctl apply -f -
apiVersion: storage.k8s.io/v1
kind: StorageClass
metadata:
name: ${DEFAULT_SC_NAME}
annotations:
storageclass.kubernetes.io/is-default-class: "false"
provisioner: ${GKE_CSI_PROVISIONER}
parameters:
type: ${DEFAULT_SC_DISK_TYPE}
reclaimPolicy: Retain
volumeBindingMode: WaitForFirstConsumer
allowVolumeExpansion: true
EOF
live_type="$(kctl get storageclass "$DEFAULT_SC_NAME" -o jsonpath='{.parameters.type}' 2>/dev/null || true)"
[[ "$live_type" == "$DEFAULT_SC_DISK_TYPE" ]] || die "after recreate, '${DEFAULT_SC_NAME}' still has type='${live_type}'" 2
fi
# 3b. Clear the default annotation from any SSD-backed class that currently
# holds it. Patch to "false" (not remove) so we leave a clear audit trail.
if [[ "${#ssd_backed_defaults[@]}" -gt 0 ]]; then
for sc in "${ssd_backed_defaults[@]}"; do
log "Un-defaulting SSD-backed class: ${sc}"
kctl annotate storageclass "$sc" \
"storageclass.kubernetes.io/is-default-class=false" --overwrite >/dev/null \
|| die "failed to un-default '${sc}'" 2
done
fi
# 3c. Promote standard-hdd to default — atomic with respect to other defaults
# because we cleared them in 3b first.
kctl annotate storageclass "$DEFAULT_SC_NAME" \
"storageclass.kubernetes.io/is-default-class=true" --overwrite >/dev/null \
|| die "failed to mark '${DEFAULT_SC_NAME}' as default" 2
# ── 4. Re-verify ─────────────────────────────────────────────────────────────
final_default="$(kctl get storageclass \
-o jsonpath='{range .items[?(@.metadata.annotations.storageclass\.kubernetes\.io/is-default-class=="true")]}{.metadata.name}{"\n"}{end}' \
2>/dev/null | head -n1)"
if [[ "$final_default" != "$DEFAULT_SC_NAME" ]]; then
die "post-remediation: default is '${final_default:-<none>}', expected '${DEFAULT_SC_NAME}'" 2
fi
# Count defaults — having two is a worse state than having zero.
default_count="$(kctl get storageclass \
-o jsonpath='{range .items[?(@.metadata.annotations.storageclass\.kubernetes\.io/is-default-class=="true")]}{.metadata.name}{"\n"}{end}' \
2>/dev/null | grep -cv '^$' || true)"
if [[ "$default_count" != "1" ]]; then
die "post-remediation: ${default_count} default StorageClasses exist (want exactly 1)" 2
fi
log "Remediation OK. Default StorageClass is now '${DEFAULT_SC_NAME}' (${DEFAULT_SC_DISK_TYPE})."
log "Re-run 'kubectl get storageclass' to confirm."
exit 0