mirror of
https://github.com/dredx/prole.git
synced 2026-09-23 12:03:59 +00:00
- Updated `deploy.sh` and `init_kong.sh` to support GKE ManagedCertificate and FrontendConfig reconciliation in k8s mode with GCE ingress class. - Enhanced TLS path diagnostics to distinguish between missing, attached-but-not-serving, and actively serving configurations. - Added HTTPS probing for GKE-managed public ingress paths and validation for managed certificate statuses. - Updated `render_supabase.py` to generate and attach TLS annotations for Supabase API and Studio ingresses in k8s mode. - Added tests to validate TLS path handling, probing, and manifest generation for GKE deployments.
935 lines
38 KiB
Bash
Executable File
935 lines
38 KiB
Bash
Executable File
#!/usr/bin/env bash
|
|
set -euo pipefail
|
|
|
|
ROOT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
|
CONFIG_PATH="${PROLE_DEPLOY_CFG:-${ROOT_DIR}/conf/gke.cfg}"
|
|
|
|
if [[ -x "${ROOT_DIR}/.venv/bin/python3" ]]; then
|
|
PYTHON_BIN="${ROOT_DIR}/.venv/bin/python3"
|
|
elif [[ -x "${ROOT_DIR}/bin/python3" ]]; then
|
|
PYTHON_BIN="${ROOT_DIR}/bin/python3"
|
|
else
|
|
PYTHON_BIN="${PYTHON:-python3}"
|
|
fi
|
|
|
|
# Validate StorageClass on GKE before deployment
|
|
if [[ "$CONFIG_PATH" == *"gke.cfg"* ]]; then
|
|
echo "Validating Supabase StorageClass configuration..."
|
|
"${PYTHON_BIN}" - "${CONFIG_PATH}" <<'PY'
|
|
import sys
|
|
import subprocess
|
|
import json
|
|
import configparser
|
|
|
|
def die(msg):
|
|
print(f"ERROR: {msg}", file=sys.stderr)
|
|
sys.exit(1)
|
|
|
|
def warn(msg):
|
|
print(f"WARN: {msg}", file=sys.stderr)
|
|
|
|
c = configparser.ConfigParser()
|
|
c.read(sys.argv[1])
|
|
g = c["Global"] if "Global" in c else {}
|
|
e = c["env_setup"] if "env_setup" in c else {}
|
|
i = c["init_cluster"] if "init_cluster" in c else {}
|
|
|
|
ctx = g.get("APP_CLUSTER_KUBECONTEXT") or e.get("APP_CLUSTER_KUBECONTEXT") or i.get("app_cluster_kubecontext", "")
|
|
if not ctx:
|
|
sys.exit(0)
|
|
|
|
# Resolve StorageClass name
|
|
sc_name = g.get("SUPABASE_STORAGE_CLASS") or c.get("Supabase", "STORAGE_CLASS", fallback="supabase-standard")
|
|
|
|
try:
|
|
out = subprocess.check_output(["kubectl", "--context", ctx, "get", "storageclass", sc_name, "-o", "json"], stderr=subprocess.DEVNULL, text=True)
|
|
sc = json.loads(out)
|
|
provisioner = sc.get("provisioner", "")
|
|
params = sc.get("parameters", {})
|
|
disk_type = params.get("type", "")
|
|
|
|
if provisioner == "pd.csi.storage.gke.io" and disk_type == "pd-standard":
|
|
print(f"Verified live Supabase StorageClass '{sc_name}' (type: {disk_type}).", file=sys.stderr)
|
|
else:
|
|
warn(
|
|
f"storage-class preflight advisory: StorageClass '{sc_name}' is present but not the expected GKE pd-standard class "
|
|
f"(provisioner='{provisioner}', type='{disk_type}'). Supabase deploy will reconcile it."
|
|
)
|
|
except subprocess.CalledProcessError:
|
|
warn(
|
|
f"storage-class preflight advisory: StorageClass '{sc_name}' does not exist on cluster '{ctx}'. "
|
|
"Supabase deploy will create/reconcile it in this run."
|
|
)
|
|
PY
|
|
fi
|
|
|
|
"${PYTHON_BIN}" -m prole.deploy_pipeline --config "${CONFIG_PATH}" "$@"
|
|
|
|
# Post-deploy summary
|
|
echo ""
|
|
echo "Public Endpoints:"
|
|
ROOT_DIR="${ROOT_DIR}" "${PYTHON_BIN}" - "${CONFIG_PATH}" <<'PY'
|
|
import sys
|
|
import os
|
|
import subprocess
|
|
import json
|
|
import configparser
|
|
import time
|
|
import re
|
|
|
|
def _cfg_get(section, *keys):
|
|
if not section:
|
|
return ""
|
|
for key in keys:
|
|
if not key:
|
|
continue
|
|
for candidate in (key, key.lower(), key.upper()):
|
|
try:
|
|
value = section.get(candidate, "")
|
|
except Exception:
|
|
value = ""
|
|
if value is None:
|
|
continue
|
|
value = str(value).strip()
|
|
if value:
|
|
return value
|
|
return ""
|
|
|
|
|
|
def _cfg_first(g, e, i, *keys):
|
|
return _cfg_get(g, *keys) or _cfg_get(e, *keys) or _cfg_get(i, *keys)
|
|
|
|
|
|
def _as_bool(value, default=False):
|
|
if value is None:
|
|
return default
|
|
token = str(value).strip().lower()
|
|
if not token:
|
|
return default
|
|
if token in {"1", "true", "yes", "on", "y"}:
|
|
return True
|
|
if token in {"0", "false", "no", "off", "n"}:
|
|
return False
|
|
return default
|
|
|
|
|
|
def get_runtime_config(path):
|
|
c = configparser.ConfigParser()
|
|
c.read(path)
|
|
g = c["Global"] if "Global" in c else {}
|
|
e = c["env_setup"] if "env_setup" in c else {}
|
|
i = c["init_cluster"] if "init_cluster" in c else {}
|
|
|
|
mode = (_cfg_first(g, e, i, "DEPLOYMENT_MODE", "deployment_mode") or "").strip().lower()
|
|
if not mode:
|
|
mode = "k8s" if "gke" in str(path).lower() else "k3s"
|
|
|
|
app_ctx = _cfg_first(g, e, i, "APP_CLUSTER_KUBECONTEXT", "app_cluster_kubecontext")
|
|
db_ctx = _cfg_first(g, e, i, "DB_CLUSTER_KUBECONTEXT", "db_cluster_kubecontext")
|
|
gitlab_ns = _cfg_first(g, e, i, "GITLAB_NAMESPACE", "gitlab_namespace") or "gitlab"
|
|
gitlab_release = _cfg_first(g, e, i, "GITLAB_RELEASE", "gitlab_release") or "gitlab"
|
|
|
|
gitlab_public_hosts_raw = _cfg_first(g, e, i, "GITLAB_PUBLIC_HOSTS", "gitlab_public_hosts")
|
|
gitlab_domain = _cfg_first(g, e, i, "GITLAB_DOMAIN", "GITLAB_HOSTNAME", "gitlab_domain", "gitlab_hostname")
|
|
if not gitlab_domain:
|
|
gitlab_domain = "git.knoe.dev" if mode == "k8s" else "git.prole.org"
|
|
gitlab_public_hosts = [h.strip() for h in str(gitlab_public_hosts_raw or "").split(",") if h.strip()]
|
|
if not gitlab_public_hosts:
|
|
gitlab_public_hosts = [gitlab_domain]
|
|
gitlab_host = gitlab_public_hosts[0]
|
|
|
|
gitlab_frontdoor_owner = (_cfg_first(g, e, i, "GITLAB_FRONTDOOR_OWNER", "gitlab_frontdoor_owner") or "").strip().lower()
|
|
if not gitlab_frontdoor_owner:
|
|
gitlab_frontdoor_owner = "fallback" if mode == "k8s" else "operator"
|
|
gitlab_fallback_ingress_name = _cfg_first(g, e, i, "GITLAB_FALLBACK_INGRESS_NAME", "gitlab_fallback_ingress_name") or "gitlab-frontdoor-ingress"
|
|
|
|
auth_host = _cfg_first(g, e, i, "AUTH_HOSTNAME", "auth_hostname")
|
|
if not auth_host:
|
|
auth_host = "api.knoe.dev" if mode == "k8s" else "api.prole.org"
|
|
|
|
service_host = _cfg_first(g, e, i, "SERVICE_HOSTNAME", "service_hostname", "GRAFANA_HOSTNAME", "grafana_hostname")
|
|
if not service_host:
|
|
service_host = "svc.knoe.dev" if mode == "k8s" else "svc.prole.org"
|
|
|
|
service_ns = _cfg_first(g, e, i, "SERVICE_NAMESPACE", "service_namespace") or "knoe-system"
|
|
|
|
supabase_enabled = _as_bool(_cfg_first(g, e, i, "SUPABASE_ENABLED", "supabase_enabled"), default=True)
|
|
supabase_api_host = _cfg_first(g, e, i, "SUPABASE_API_HOSTNAME", "supabase_api_hostname") or "api.0.knoe.dev"
|
|
supabase_studio_host = _cfg_first(g, e, i, "SUPABASE_STUDIO_HOSTNAME", "SUPABASE_HOSTNAME", "supabase_studio_hostname", "supabase_hostname") or "db.0.knoe.dev"
|
|
|
|
return {
|
|
"mode": mode,
|
|
"app_ctx": app_ctx,
|
|
"db_ctx": db_ctx,
|
|
"gitlab_ns": gitlab_ns,
|
|
"gitlab_release": gitlab_release,
|
|
"gitlab_host": gitlab_host,
|
|
"gitlab_frontdoor_owner": gitlab_frontdoor_owner,
|
|
"gitlab_fallback_ingress_name": gitlab_fallback_ingress_name,
|
|
"gitlab_operator_ingress_name": f"{gitlab_release}-webservice-default",
|
|
"auth_host": auth_host,
|
|
"service_host": service_host,
|
|
"service_ns": service_ns,
|
|
"supabase_enabled": supabase_enabled,
|
|
"supabase_api_host": supabase_api_host,
|
|
"supabase_studio_host": supabase_studio_host,
|
|
}
|
|
|
|
|
|
def load_ingress_inventory(ctx):
|
|
if not ctx:
|
|
return []
|
|
try:
|
|
out = subprocess.check_output([
|
|
"kubectl", "--context", ctx, "get", "ingress", "-A", "-o", "json"
|
|
], stderr=subprocess.DEVNULL, text=True)
|
|
return json.loads(out).get("items", [])
|
|
except Exception:
|
|
return []
|
|
|
|
|
|
def _ingress_class(item):
|
|
spec = item.get("spec", {}) or {}
|
|
metadata = item.get("metadata", {}) or {}
|
|
annotations = metadata.get("annotations", {}) or {}
|
|
spec_class = str(spec.get("ingressClassName") or "").strip()
|
|
ann_class = str(annotations.get("kubernetes.io/ingress.class") or "").strip()
|
|
effective = spec_class or ann_class or "<unset>"
|
|
return effective, spec_class, ann_class
|
|
|
|
|
|
def _ingress_address(item):
|
|
ing = (item.get("status", {}) or {}).get("loadBalancer", {}).get("ingress", [])
|
|
if not ing:
|
|
return "pending"
|
|
first = ing[0] or {}
|
|
return first.get("ip") or first.get("hostname") or "pending"
|
|
|
|
|
|
def _ingress_backend_summary(item, host):
|
|
services = []
|
|
spec = item.get("spec", {}) or {}
|
|
for rule in spec.get("rules", []) or []:
|
|
if (rule or {}).get("host") != host:
|
|
continue
|
|
http = (rule or {}).get("http", {}) or {}
|
|
for path in http.get("paths", []) or []:
|
|
backend = (path or {}).get("backend", {}) or {}
|
|
service = backend.get("service", {}) or {}
|
|
name = service.get("name")
|
|
port = (service.get("port") or {}).get("number") or (service.get("port") or {}).get("name")
|
|
if name:
|
|
services.append(f"{name}:{port}" if port else str(name))
|
|
return ",".join(services) if services else "-"
|
|
|
|
|
|
def _ingress_tls_diagnostics(item, host):
|
|
metadata = item.get("metadata", {}) or {}
|
|
annotations = metadata.get("annotations", {}) or {}
|
|
spec = item.get("spec", {}) or {}
|
|
|
|
tls_entries = spec.get("tls", []) or []
|
|
tls_hosts = []
|
|
tls_secret_names = []
|
|
tls_host_match = False
|
|
for entry in tls_entries:
|
|
if not isinstance(entry, dict):
|
|
continue
|
|
hosts = [str(h).strip() for h in (entry.get("hosts") or []) if str(h).strip()]
|
|
if hosts:
|
|
tls_hosts.extend(hosts)
|
|
else:
|
|
tls_host_match = True
|
|
if host in hosts:
|
|
tls_host_match = True
|
|
secret_name = str(entry.get("secretName") or "").strip()
|
|
if secret_name:
|
|
tls_secret_names.append(secret_name)
|
|
|
|
managed_cert = str(annotations.get("networking.gke.io/managed-certificates") or "").strip()
|
|
pre_shared_cert = str(annotations.get("ingress.gcp.kubernetes.io/pre-shared-cert") or "").strip()
|
|
|
|
has_tls_path = bool(tls_host_match or managed_cert or pre_shared_cert)
|
|
tls_path_state = "TLS_PATH_ATTACHED" if has_tls_path else "MISSING_TLS_PATH"
|
|
|
|
return {
|
|
"has_tls_path": has_tls_path,
|
|
"tls_path_state": tls_path_state,
|
|
"tls_hosts": ",".join(sorted(set(tls_hosts))) if tls_hosts else "-",
|
|
"tls_secrets": ",".join(sorted(set(tls_secret_names))) if tls_secret_names else "-",
|
|
"managed_cert": managed_cert or "-",
|
|
"pre_shared_cert": pre_shared_cert or "-",
|
|
}
|
|
|
|
|
|
def get_ingresses_for_host(inventory, host):
|
|
result = []
|
|
for item in inventory:
|
|
spec = item.get("spec", {}) or {}
|
|
matched = False
|
|
for rule in spec.get("rules", []) or []:
|
|
if (rule or {}).get("host") == host:
|
|
matched = True
|
|
break
|
|
if not matched:
|
|
continue
|
|
|
|
metadata = item.get("metadata", {}) or {}
|
|
namespace = metadata.get("namespace") or "default"
|
|
name = metadata.get("name") or "<unknown>"
|
|
ingress_class, class_from_spec, class_from_annotation = _ingress_class(item)
|
|
tls_diag = _ingress_tls_diagnostics(item, host)
|
|
result.append({
|
|
"namespace": namespace,
|
|
"name": name,
|
|
"ingressClass": ingress_class,
|
|
"classFromSpec": class_from_spec or "-",
|
|
"classFromAnnotation": class_from_annotation or "-",
|
|
"address": _ingress_address(item),
|
|
"backend": _ingress_backend_summary(item, host),
|
|
**tls_diag,
|
|
})
|
|
return result
|
|
|
|
|
|
def _pick_authoritative_owner(owners, expected):
|
|
if not owners:
|
|
return None
|
|
if expected:
|
|
for owner in owners:
|
|
if owner.get("namespace") == expected.get("namespace") and owner.get("name") == expected.get("name"):
|
|
return owner
|
|
return owners[0]
|
|
|
|
|
|
def _managed_cert_status(ctx, namespace, cert_name):
|
|
if not ctx or not namespace or not cert_name:
|
|
return "Unknown"
|
|
try:
|
|
out = subprocess.check_output([
|
|
"kubectl", "--context", ctx, "get", "managedcertificate", cert_name,
|
|
"-n", namespace, "-o", "json"
|
|
], stderr=subprocess.DEVNULL, text=True)
|
|
item = json.loads(out)
|
|
status = (item.get("status") or {}) if isinstance(item, dict) else {}
|
|
cert_status = str(status.get("certificateStatus") or status.get("status") or "").strip()
|
|
return cert_status or "Unknown"
|
|
except Exception:
|
|
return "Unknown"
|
|
|
|
|
|
def _managed_cert_statuses(ctx, namespace, managed_cert_annotation):
|
|
cert_names = [token.strip() for token in str(managed_cert_annotation or "").split(",") if token.strip()]
|
|
statuses = {}
|
|
for cert_name in cert_names:
|
|
statuses[cert_name] = _managed_cert_status(ctx, namespace, cert_name)
|
|
return statuses
|
|
|
|
|
|
def _probe_https_endpoint(host, timeout_seconds=15):
|
|
url = f"https://{host}/"
|
|
try:
|
|
completed = subprocess.run([
|
|
"curl",
|
|
"--silent",
|
|
"--show-error",
|
|
"--location",
|
|
"--output", "/dev/null",
|
|
"--write-out", "%{http_code}",
|
|
"--connect-timeout", "5",
|
|
"--max-time", str(timeout_seconds),
|
|
url,
|
|
], check=False, capture_output=True, text=True)
|
|
http_code = str(completed.stdout or "").strip()
|
|
if completed.returncode == 0 and re.fullmatch(r"\d{3}", http_code):
|
|
return True, f"http={http_code}"
|
|
|
|
detail = str(completed.stderr or "").strip()
|
|
if not detail:
|
|
detail = str(completed.stdout or "").strip()
|
|
if detail:
|
|
detail = detail.splitlines()[-1]
|
|
if not detail:
|
|
detail = f"curl_exit={completed.returncode}"
|
|
return False, detail[:220]
|
|
except Exception as exc:
|
|
return False, str(exc)
|
|
|
|
def get_ingress_status(ctx, namespace, ingress_name):
|
|
status = {
|
|
"name": ingress_name,
|
|
"exists": False,
|
|
"address": "pending",
|
|
"hosts": [],
|
|
}
|
|
if not ctx:
|
|
return status
|
|
try:
|
|
out = subprocess.check_output([
|
|
"kubectl", "--context", ctx, "get", "ingress", ingress_name,
|
|
"-n", namespace, "-o", "json"
|
|
], stderr=subprocess.DEVNULL, text=True)
|
|
item = json.loads(out)
|
|
status["exists"] = True
|
|
status["hosts"] = [
|
|
rule.get("host")
|
|
for rule in item.get("spec", {}).get("rules", [])
|
|
if rule.get("host")
|
|
]
|
|
ing = item.get("status", {}).get("loadBalancer", {}).get("ingress", [])
|
|
if ing:
|
|
addr = ing[0].get("ip") or ing[0].get("hostname")
|
|
if addr:
|
|
status["address"] = addr
|
|
except subprocess.CalledProcessError:
|
|
pass
|
|
except Exception:
|
|
pass
|
|
return status
|
|
|
|
def check_supabase_pvcs(ctx):
|
|
if not ctx: return None
|
|
try:
|
|
from datetime import datetime, timezone
|
|
out = subprocess.check_output(["kubectl", "--context", ctx, "get", "events", "-n", "supabase", "-o", "json"], stderr=subprocess.DEVNULL, text=True)
|
|
events = json.loads(out).get("items", [])
|
|
|
|
# Fetch current PVC states to ignore Bound ones
|
|
out_pvc = subprocess.check_output(["kubectl", "--context", ctx, "get", "pvc", "-n", "supabase", "-o", "json"], stderr=subprocess.DEVNULL, text=True)
|
|
pvcs = json.loads(out_pvc).get("items", [])
|
|
pvc_status = {}
|
|
pvc_sc = {}
|
|
for p in pvcs:
|
|
name = (p.get("metadata") or {}).get("name")
|
|
if not name:
|
|
continue
|
|
pvc_status[name] = (p.get("status") or {}).get("phase")
|
|
pvc_sc[name] = (p.get("spec") or {}).get("storageClassName") or "<unset>"
|
|
|
|
sc_cache = {}
|
|
def sc_info(sc_name):
|
|
key = sc_name or "<unset>"
|
|
if key in sc_cache:
|
|
return sc_cache[key]
|
|
info = {
|
|
"storageClass": key,
|
|
"provisioner": "<unknown>",
|
|
"type": "<unknown>",
|
|
"volumeBindingMode": "<unknown>",
|
|
"reclaimPolicy": "<unknown>",
|
|
}
|
|
if sc_name and sc_name != "<unset>":
|
|
try:
|
|
out_sc = subprocess.check_output([
|
|
"kubectl", "--context", ctx, "get", "storageclass", sc_name, "-o", "json"
|
|
], stderr=subprocess.DEVNULL, text=True)
|
|
sc_obj = json.loads(out_sc)
|
|
info["provisioner"] = sc_obj.get("provisioner") or "<unknown>"
|
|
info["type"] = (sc_obj.get("parameters") or {}).get("type") or "<unknown>"
|
|
info["volumeBindingMode"] = sc_obj.get("volumeBindingMode") or "<unknown>"
|
|
info["reclaimPolicy"] = sc_obj.get("reclaimPolicy") or "<unknown>"
|
|
except Exception:
|
|
pass
|
|
sc_cache[key] = info
|
|
return info
|
|
|
|
def parse_ts(value):
|
|
if not value:
|
|
return None
|
|
try:
|
|
return datetime.fromisoformat(value.replace('Z', '+00:00'))
|
|
except Exception:
|
|
return None
|
|
|
|
blocking_pvcs = {}
|
|
for event in events:
|
|
obj = event.get("involvedObject", {})
|
|
if obj.get("kind") == "PersistentVolumeClaim":
|
|
pvc_name = obj.get("name")
|
|
if not pvc_name:
|
|
continue
|
|
|
|
# Skip Bound PVCs
|
|
if pvc_status.get(pvc_name) == "Bound":
|
|
continue
|
|
|
|
msg = event.get("message", "")
|
|
reason = event.get("reason", "")
|
|
msg_l = msg.lower()
|
|
reason_l = reason.lower()
|
|
if not (
|
|
"quota_exceeded" in msg_l
|
|
or "exceeded" in msg_l
|
|
or "quota" in msg_l
|
|
or "provisioningfailed" in reason_l
|
|
or "failedbinding" in reason_l
|
|
):
|
|
continue
|
|
|
|
ts = parse_ts(event.get("lastTimestamp") or event.get("eventTime") or event.get("firstTimestamp"))
|
|
sc = pvc_sc.get(pvc_name) or "<unset>"
|
|
sc_details = sc_info(sc)
|
|
candidate = {
|
|
"pvc": pvc_name,
|
|
"phase": pvc_status.get(pvc_name) or "Unknown",
|
|
"message": msg,
|
|
"reason": reason or "Unknown",
|
|
"eventTime": ts.isoformat() if ts else "",
|
|
"storageClass": sc_details["storageClass"],
|
|
"storageProvisioner": sc_details["provisioner"],
|
|
"storageType": sc_details["type"],
|
|
"volumeBindingMode": sc_details["volumeBindingMode"],
|
|
"reclaimPolicy": sc_details["reclaimPolicy"],
|
|
}
|
|
|
|
previous = blocking_pvcs.get(pvc_name)
|
|
if not previous:
|
|
blocking_pvcs[pvc_name] = candidate
|
|
continue
|
|
|
|
prev_ts = parse_ts(previous.get("eventTime"))
|
|
if prev_ts is None or (ts is not None and ts >= prev_ts):
|
|
blocking_pvcs[pvc_name] = candidate
|
|
|
|
stale_retained = []
|
|
out_pv = subprocess.check_output([
|
|
"kubectl", "--context", ctx, "get", "pv", "-o", "json"
|
|
], stderr=subprocess.DEVNULL, text=True)
|
|
pvs = json.loads(out_pv).get("items", [])
|
|
for pv in pvs:
|
|
spec = pv.get("spec", {})
|
|
status = pv.get("status", {})
|
|
claim = spec.get("claimRef") or {}
|
|
claim_ns = claim.get("namespace")
|
|
claim_name = claim.get("name")
|
|
if claim_ns != "supabase":
|
|
continue
|
|
if not str(claim_name or "").startswith("supabase-"):
|
|
continue
|
|
|
|
phase = status.get("phase", "")
|
|
reclaim = spec.get("persistentVolumeReclaimPolicy", "")
|
|
if reclaim != "Retain" or phase not in ("Released", "Failed"):
|
|
continue
|
|
|
|
csi = spec.get("csi") or {}
|
|
gce = spec.get("gcePersistentDisk") or {}
|
|
labels = (pv.get("metadata") or {}).get("labels") or {}
|
|
disk_handle = csi.get("volumeHandle") or gce.get("pdName") or "<unknown>"
|
|
zone = (
|
|
labels.get("topology.kubernetes.io/zone")
|
|
or labels.get("failure-domain.beta.kubernetes.io/zone")
|
|
or (csi.get("volumeAttributes") or {}).get("topology.gke.io/zone")
|
|
or ""
|
|
)
|
|
if not zone and isinstance(disk_handle, str):
|
|
m = re.search(r"/zones/([^/]+)/disks/[^/]+$", disk_handle)
|
|
if m:
|
|
zone = m.group(1)
|
|
stale_retained.append({
|
|
"pv": (pv.get("metadata") or {}).get("name", "?"),
|
|
"pvc": claim_name or "?",
|
|
"phase": phase or "Unknown",
|
|
"storageClass": spec.get("storageClassName") or "<unset>",
|
|
"diskHandle": disk_handle,
|
|
"zone": zone or "<unknown>",
|
|
})
|
|
|
|
if blocking_pvcs or stale_retained:
|
|
return {
|
|
"blockingPVCs": sorted(blocking_pvcs.values(), key=lambda x: x.get("pvc") or ""),
|
|
"staleRetainedPVs": stale_retained,
|
|
}
|
|
except: pass
|
|
return None
|
|
|
|
def get_deployment_status(ctx, namespace, name):
|
|
status = {
|
|
"exists": False,
|
|
"specReplicas": 0,
|
|
"readyReplicas": 0,
|
|
"availableReplicas": 0,
|
|
}
|
|
if not ctx:
|
|
return status
|
|
try:
|
|
out = subprocess.check_output([
|
|
"kubectl", "--context", ctx, "get", "deployment", name, "-n", namespace, "-o", "json"
|
|
], stderr=subprocess.DEVNULL, text=True)
|
|
dep = json.loads(out)
|
|
spec = dep.get("spec") or {}
|
|
st = dep.get("status") or {}
|
|
status["exists"] = True
|
|
status["specReplicas"] = int(spec.get("replicas") or 0)
|
|
status["readyReplicas"] = int(st.get("readyReplicas") or 0)
|
|
status["availableReplicas"] = int(st.get("availableReplicas") or 0)
|
|
except Exception:
|
|
pass
|
|
return status
|
|
|
|
def reconcile_gitlab(app_ctx, db_ctx, ns, config_path):
|
|
if not app_ctx: return
|
|
print(f"\nReconciling GitLab workloads in namespace '{ns}'...")
|
|
print(f" Using config: {config_path}")
|
|
print(f" Using APP context: {app_ctx}")
|
|
if db_ctx:
|
|
print(f" Using DB context: {db_ctx}")
|
|
|
|
try:
|
|
# Get all deployments in one go
|
|
out = subprocess.check_output(["kubectl", "--context", app_ctx, "-n", ns, "get", "deployments", "-o", "json"], text=True)
|
|
deploys = json.loads(out).get("items", [])
|
|
|
|
# Get all replicasets in one go
|
|
out = subprocess.check_output(["kubectl", "--context", app_ctx, "-n", ns, "get", "replicasets", "-o", "json"], text=True)
|
|
all_rss = json.loads(out).get("items", [])
|
|
except Exception as e:
|
|
print(f" Warning: Could not fetch GitLab workloads for reconciliation: {e}")
|
|
return
|
|
|
|
# Targets specifically mentioned in requirements or operator managed
|
|
gitlab_workloads = [
|
|
"gitlab-sidekiq-all-in-1-v2",
|
|
"gitlab-gitlab-shell",
|
|
"gitlab-kas",
|
|
"gitlab-registry",
|
|
"gitlab-webservice-default"
|
|
]
|
|
|
|
summary = []
|
|
over_deployed_detected = False
|
|
|
|
for dep in deploys:
|
|
name = dep['metadata']['name']
|
|
# Filter for operator-managed GitLab workloads only
|
|
is_gitlab = False
|
|
if name in gitlab_workloads:
|
|
is_gitlab = True
|
|
elif name.startswith("gitlab-") and not any(x in name for x in ["-migration", "-secrets", "-minio-create-buckets", "-db-config"]):
|
|
is_gitlab = True
|
|
|
|
if not is_gitlab:
|
|
continue
|
|
|
|
spec = dep.get('spec', {})
|
|
status = dep.get('status', {})
|
|
|
|
# Target should be 1 for these specific GitLab workloads (Requirement 11)
|
|
if name in gitlab_workloads:
|
|
desired = 1
|
|
else:
|
|
desired = spec.get('replicas', 0)
|
|
|
|
live = status.get('replicas', 0)
|
|
updated = status.get('updatedReplicas', 0)
|
|
|
|
# Enforce desired replica count (Requirement 3 & 4)
|
|
# Reconciliation compares:
|
|
# - desired replicas (from Helm values)
|
|
# - actual replicas (from Deployment.spec.replicas)
|
|
actual_spec_replicas = spec.get('replicas', 0)
|
|
if name in gitlab_workloads and actual_spec_replicas != desired:
|
|
print(f" [FIX] Deployment '{name}' spec.replicas ({actual_spec_replicas}) differs from desired ({desired}).")
|
|
over_deployed_detected = True
|
|
|
|
summary.append({
|
|
"name": name,
|
|
"desired": desired,
|
|
"live": live,
|
|
"updated": updated
|
|
})
|
|
|
|
# Check for Gitaly storage class drift on GKE (Requirement 3)
|
|
is_gke = "gke" in sys.argv[1].lower()
|
|
if is_gke:
|
|
try:
|
|
out = subprocess.check_output(["kubectl", "--context", app_ctx, "-n", ns, "get", "gitlab", "-o", "json"], text=True)
|
|
crs = json.loads(out).get("items", [])
|
|
for cr in crs:
|
|
spec = cr.get("spec", {})
|
|
live_sc = spec.get("chart", {}).get("values", {}).get("global", {}).get("persistence", {}).get("storageClass")
|
|
if not live_sc:
|
|
live_sc = spec.get("chart", {}).get("values", {}).get("gitlab", {}).get("gitaly", {}).get("persistence", {}).get("storageClass")
|
|
|
|
# Desired is 'standard' in this environment by default
|
|
desired_sc = "standard"
|
|
|
|
if live_sc:
|
|
if live_sc != desired_sc:
|
|
print(f" [FIX] GitLab CR '{cr['metadata']['name']}' storageClass '{live_sc}' differs from desired '{desired_sc}'.")
|
|
over_deployed_detected = True
|
|
except:
|
|
pass
|
|
|
|
# Print summary (Requirement 8)
|
|
print(f"\n{'GitLab Deployment':<35} {'Desired':<8} {'Live':<8} {'Updated':<8}")
|
|
print("-" * 70)
|
|
for s in summary:
|
|
print(f"{s['name']:<35} {s['desired']:<8} {s['live']:<8} {s['updated']:<8}")
|
|
|
|
if over_deployed_detected:
|
|
print("\nEnforcing desired configuration by re-applying GitLab Helm release/CR...")
|
|
# Re-apply the Helm release (Requirement 2 & 4)
|
|
try:
|
|
root_dir = os.environ.get("ROOT_DIR", ".")
|
|
env = os.environ.copy()
|
|
# Explicitly pass contexts and config path to avoid fallbacks
|
|
env["APP_CLUSTER_KUBECONTEXT"] = app_ctx
|
|
if db_ctx:
|
|
env["DB_CLUSTER_KUBECONTEXT"] = db_ctx
|
|
env["KUBECTL_CONTEXT"] = app_ctx
|
|
env["PROLE_DEPLOY_CFG"] = config_path
|
|
|
|
subprocess.run(["/usr/bin/env", "bash", f"{root_dir}/etc/init_gitlab.sh", "--config", config_path, "deploy"], env=env, check=True)
|
|
except Exception as e:
|
|
print(f" ERROR: Failed to re-apply GitLab configuration: {e}")
|
|
sys.exit(1)
|
|
|
|
# Final Convergence Check (Requirement 6)
|
|
still_over_deployed = False
|
|
for s in summary:
|
|
# If we detected drift, we re-applied the configuration.
|
|
# Now we check if it converged.
|
|
if over_deployed_detected:
|
|
# Re-fetch live status for the final check
|
|
try:
|
|
out = subprocess.check_output(["kubectl", "--context", app_ctx, "-n", ns, "get", "deployment", s['name'], "-o", "json"], text=True)
|
|
d = json.loads(out)
|
|
st = d.get('status', {})
|
|
# Compare against the intended 'desired' count (1), not necessarily the (potentially stale) spec.replicas
|
|
if st.get('replicas', 0) > s['desired'] or st.get('updatedReplicas', 0) != s['desired']:
|
|
print(f"ERROR: GitLab deployment '{s['name']}' failed to converge (live={st.get('replicas')}, desired={s['desired']}).")
|
|
still_over_deployed = True
|
|
except:
|
|
pass
|
|
elif s['live'] > s['desired'] or s['updated'] != s['desired']:
|
|
print(f"ERROR: GitLab deployment '{s['name']}' is not converged.")
|
|
still_over_deployed = True
|
|
|
|
if still_over_deployed:
|
|
sys.exit(1)
|
|
|
|
runtime = get_runtime_config(sys.argv[1])
|
|
app_ctx = runtime["app_ctx"]
|
|
db_ctx = runtime["db_ctx"]
|
|
gitlab_ns = runtime["gitlab_ns"]
|
|
config_path = sys.argv[1]
|
|
|
|
# Run GitLab reconciliation first
|
|
reconcile_gitlab(app_ctx, db_ctx, gitlab_ns, config_path)
|
|
|
|
# Supabase frontdoor ingresses that must converge together
|
|
required_supabase_ingresses = [
|
|
{"name": "supabase-kong", "endpoint": runtime["supabase_api_host"]},
|
|
{"name": "supabase-studio", "endpoint": runtime["supabase_studio_host"]},
|
|
]
|
|
timeout = 300 # 5 minutes
|
|
start = time.time()
|
|
supabase_ingress_state = {}
|
|
studio_deploy_status = get_deployment_status(app_ctx, "supabase", "supabase-studio")
|
|
|
|
if runtime["supabase_enabled"] and runtime["supabase_studio_host"]:
|
|
if not studio_deploy_status.get("exists"):
|
|
print("\nERROR: Supabase Studio ingress cannot reconcile because deployment/supabase-studio is absent.")
|
|
print("Reason: Studio ingress rendering depends on Studio workload enablement in rendered Supabase values.")
|
|
sys.exit(1)
|
|
if int(studio_deploy_status.get("specReplicas") or 0) < 1:
|
|
print("\nERROR: Supabase Studio ingress cannot reconcile because deployment/supabase-studio is scaled to 0.")
|
|
print("Reason: Studio ingress is enabled but Studio workload replicas are zero in the current rendered/apply state.")
|
|
sys.exit(1)
|
|
|
|
while time.time() - start < timeout:
|
|
pvc_state = check_supabase_pvcs(app_ctx)
|
|
if pvc_state:
|
|
stale = pvc_state.get("staleRetainedPVs") or []
|
|
if stale:
|
|
print("\nERROR: Supabase PVC provisioning is blocked by stale retained PV/GCE disk artifacts from previous runs.")
|
|
for item in stale:
|
|
print(f" - PV: {item['pv']}")
|
|
print(f" PVC: {item['pvc']}")
|
|
print(f" Phase/Reclaim: {item['phase']}/{item.get('storageClass','<unset>')} Retain")
|
|
print(f" Disk: {item['diskHandle']} (zone: {item['zone']})")
|
|
print("\nAction: stale Supabase Retain PVs should be auto-cleaned during deploy; delete the remaining backing GCE disks listed above, or enable SUPABASE_AUTO_CLEAN_RETAINED_GCE_DISKS=true with gcloud/project env configured.")
|
|
sys.exit(1)
|
|
|
|
blockers = pvc_state.get("blockingPVCs") or []
|
|
if blockers:
|
|
print("\nERROR: Supabase PVC provisioning failed (Storage Quota Exceeded).")
|
|
for item in blockers:
|
|
print(f" - PVC: {item.get('pvc')}")
|
|
print(f" Phase/Reason: {item.get('phase')}/{item.get('reason')}")
|
|
print(
|
|
f" StorageClass: {item.get('storageClass')} "
|
|
f"(provisioner={item.get('storageProvisioner')}, type={item.get('storageType')}, "
|
|
f"binding={item.get('volumeBindingMode')}, reclaim={item.get('reclaimPolicy')})"
|
|
)
|
|
print(f" GKE Error: {item.get('message')}")
|
|
sys.exit(1)
|
|
|
|
pending = False
|
|
supabase_ingress_state = {}
|
|
for req in required_supabase_ingresses:
|
|
ing = get_ingress_status(app_ctx, "supabase", req["name"])
|
|
supabase_ingress_state[req["name"]] = ing
|
|
if not ing.get("exists") or ing.get("address") == "pending":
|
|
pending = True
|
|
break
|
|
if not pending:
|
|
break
|
|
time.sleep(10)
|
|
|
|
if time.time() - start >= timeout:
|
|
print("\nERROR: Timeout waiting for Supabase ingress reconciliation (both supabase-kong and supabase-studio are required).")
|
|
for req in required_supabase_ingresses:
|
|
ing = supabase_ingress_state.get(req["name"]) or get_ingress_status(app_ctx, "supabase", req["name"])
|
|
print(f" - {req['name']} ({req['endpoint']}): exists={ing.get('exists')} address={ing.get('address')} hosts={','.join(ing.get('hosts') or []) or '-'}")
|
|
print("Ingress Controller Diagnosis (kubectl get events -n supabase):")
|
|
try:
|
|
subprocess.run(["kubectl", "--context", app_ctx, "get", "events", "-n", "supabase", "--sort-by=.lastTimestamp"], check=False)
|
|
print("\nIngress Resource Status:")
|
|
subprocess.run(["kubectl", "--context", app_ctx, "describe", "ingress", "supabase-kong", "-n", "supabase"], check=False)
|
|
subprocess.run(["kubectl", "--context", app_ctx, "describe", "ingress", "supabase-studio", "-n", "supabase"], check=False)
|
|
except:
|
|
pass
|
|
sys.exit(1)
|
|
|
|
kong_state = supabase_ingress_state.get("supabase-kong") or get_ingress_status(app_ctx, "supabase", "supabase-kong")
|
|
studio_state = supabase_ingress_state.get("supabase-studio") or get_ingress_status(app_ctx, "supabase", "supabase-studio")
|
|
if (not kong_state.get("exists") or kong_state.get("address") == "pending" or
|
|
not studio_state.get("exists") or studio_state.get("address") == "pending"):
|
|
print("\nERROR: Supabase frontdoor is not converged: both supabase-kong and supabase-studio ingresses must exist and have addresses.")
|
|
print(f" - supabase-kong ({runtime['supabase_api_host']}): exists={kong_state.get('exists')} address={kong_state.get('address')}")
|
|
print(f" - supabase-studio ({runtime['supabase_studio_host']}): exists={studio_state.get('exists')} address={studio_state.get('address')}")
|
|
sys.exit(1)
|
|
|
|
public_hosts = []
|
|
for candidate in (
|
|
runtime["gitlab_host"],
|
|
runtime["supabase_api_host"] if runtime["supabase_enabled"] else "",
|
|
runtime["supabase_studio_host"] if runtime["supabase_enabled"] else "",
|
|
runtime["auth_host"],
|
|
runtime["service_host"],
|
|
):
|
|
h = str(candidate or "").strip()
|
|
if not h or h in public_hosts:
|
|
continue
|
|
public_hosts.append(h)
|
|
|
|
expected_owner = {
|
|
runtime["gitlab_host"]: {
|
|
"namespace": gitlab_ns,
|
|
"name": runtime["gitlab_fallback_ingress_name"] if runtime["gitlab_frontdoor_owner"] == "fallback" else runtime["gitlab_operator_ingress_name"],
|
|
},
|
|
runtime["auth_host"]: {"namespace": runtime["service_ns"], "name": "svc-knoe-ingress"},
|
|
runtime["service_host"]: {"namespace": runtime["service_ns"], "name": "svc-knoe-ingress"},
|
|
}
|
|
if runtime["supabase_enabled"]:
|
|
expected_owner[runtime["supabase_api_host"]] = {"namespace": "supabase", "name": "supabase-kong"}
|
|
expected_owner[runtime["supabase_studio_host"]] = {"namespace": "supabase", "name": "supabase-studio"}
|
|
|
|
inventory = load_ingress_inventory(app_ctx)
|
|
broken_hosts = []
|
|
host_diagnostics = {}
|
|
|
|
print("\nIngress/TLS diagnostics by public hostname:")
|
|
for host in public_hosts:
|
|
owners = get_ingresses_for_host(inventory, host)
|
|
host_diagnostics[host] = owners
|
|
if not owners:
|
|
broken_hosts.append(f"{host}: no ingress claims this host")
|
|
print(
|
|
f" - host={host} ingress=(none) class=- address=pending tlsState=MISSING_TLS_PATH tlsHosts=- tlsSecrets=- managedCert=- managedCertStatus=- preSharedCert=- httpsProbe=not-run backend=-"
|
|
)
|
|
continue
|
|
|
|
expected = expected_owner.get(host)
|
|
owner = _pick_authoritative_owner(owners, expected)
|
|
if not owner:
|
|
broken_hosts.append(f"{host}: unable to determine authoritative ingress owner")
|
|
continue
|
|
|
|
if len(owners) > 1:
|
|
owner_desc = ", ".join(f"{o['namespace']}/{o['name']}[class={o['ingressClass']}]" for o in owners)
|
|
broken_hosts.append(f"{host}: multiple ingress owners found ({owner_desc})")
|
|
|
|
managed_cert_statuses = _managed_cert_statuses(app_ctx, owner["namespace"], owner["managed_cert"])
|
|
managed_cert_status_summary = (
|
|
",".join(f"{name}:{status}" for name, status in managed_cert_statuses.items()) if managed_cert_statuses else "-"
|
|
)
|
|
owner["managed_cert_status"] = managed_cert_status_summary
|
|
|
|
probe_ok = False
|
|
probe_result = "not-run"
|
|
if owner["has_tls_path"]:
|
|
probe_ok, probe_result = _probe_https_endpoint(host)
|
|
owner["tls_path_state"] = "TLS_PATH_SERVING" if probe_ok else "TLS_PATH_ATTACHED_BUT_NOT_SERVING"
|
|
else:
|
|
owner["tls_path_state"] = "MISSING_TLS_PATH"
|
|
owner["https_probe"] = probe_result
|
|
|
|
print(
|
|
" - host={host} ingress={ns}/{name} class={cls} address={addr} tlsState={tls_state} tlsHosts={tls_hosts} tlsSecrets={tls_secrets} managedCert={managed} managedCertStatus={managed_status} preSharedCert={pre_shared} httpsProbe={https_probe} backend={backend}".format(
|
|
host=host,
|
|
ns=owner["namespace"],
|
|
name=owner["name"],
|
|
cls=owner["ingressClass"],
|
|
addr=owner["address"],
|
|
tls_state=owner["tls_path_state"],
|
|
tls_hosts=owner["tls_hosts"],
|
|
tls_secrets=owner["tls_secrets"],
|
|
managed=owner["managed_cert"],
|
|
managed_status=owner.get("managed_cert_status", "-"),
|
|
pre_shared=owner["pre_shared_cert"],
|
|
https_probe=owner.get("https_probe", "not-run"),
|
|
backend=owner["backend"],
|
|
)
|
|
)
|
|
|
|
if expected and (owner["namespace"] != expected["namespace"] or owner["name"] != expected["name"]):
|
|
broken_hosts.append(
|
|
f"{host}: unexpected ingress owner {owner['namespace']}/{owner['name']} (expected {expected['namespace']}/{expected['name']})"
|
|
)
|
|
|
|
if owner["address"] == "pending":
|
|
broken_hosts.append(f"{host}: ingress {owner['namespace']}/{owner['name']} has no load-balancer address yet")
|
|
|
|
if owner["tls_path_state"] == "MISSING_TLS_PATH":
|
|
broken_hosts.append(
|
|
f"{host}: MISSING_TLS_PATH on ingress {owner['namespace']}/{owner['name']} (tlsHosts={owner['tls_hosts']}, managedCert={owner['managed_cert']}, preSharedCert={owner['pre_shared_cert']})"
|
|
)
|
|
|
|
if owner["tls_path_state"] == "TLS_PATH_ATTACHED_BUT_NOT_SERVING":
|
|
broken_hosts.append(
|
|
f"{host}: TLS_PATH_ATTACHED_BUT_NOT_SERVING on ingress {owner['namespace']}/{owner['name']} (httpsProbe={owner.get('https_probe', '-')})"
|
|
)
|
|
|
|
if host == runtime["gitlab_host"] and managed_cert_statuses:
|
|
non_active = [f"{name}:{status}" for name, status in managed_cert_statuses.items() if status != "Active"]
|
|
if non_active:
|
|
broken_hosts.append(
|
|
f"{host}: managed certificate status is not Active ({', '.join(non_active)})"
|
|
)
|
|
|
|
if broken_hosts:
|
|
print("\nERROR: Public HTTPS front-door validation failed.")
|
|
for issue in broken_hosts:
|
|
print(f" - {issue}")
|
|
sys.exit(1)
|
|
|
|
endpoint_labels = {
|
|
runtime["supabase_api_host"]: "Supabase Kong",
|
|
runtime["supabase_studio_host"]: "Supabase Studio",
|
|
runtime["gitlab_host"]: "GitLab",
|
|
runtime["auth_host"]: "Auth API",
|
|
runtime["service_host"]: "Service/Grafana",
|
|
}
|
|
|
|
print(f"\n{'hostname':<24} {'owner ingress':<34} {'class':<10} {'external IP':<22}")
|
|
print("-" * 100)
|
|
for host in public_hosts:
|
|
owners = host_diagnostics.get(host) or []
|
|
owner = owners[0] if owners else {}
|
|
owner_name = f"{owner.get('namespace','-')}/{owner.get('name','-')}"
|
|
print(f"{host:<24} {owner_name:<34} {owner.get('ingressClass','-'):<10} {owner.get('address','pending'):<22}")
|
|
PY
|