mirror of
https://github.com/dredx/prole.git
synced 2026-09-28 03:24:30 +00:00
## GCP / Cluster Environment Screen - Auto-populate Cloud tab from conf/prod/gcp.cfg on screen open (org_id, billing_account, billing_project, project_id) - gcloud auth validity checked on screen startup; friendly modal dialog streams gcloud auth login output live so user never leaves the app - Live GKE cluster browser: fetches clusters via gcloud container clusters list, displays with checkmark selector, auto-selects saved cluster - Selecting a cluster runs get-credentials, sets KUBECONFIG/KUBECONTEXT, and syncs the region dropdown to the selected cluster's location - Region dropdown populated live from gcloud compute regions list with checkmark on currently selected region; graceful fallback when offline - New 'GCP Storage' tab with workload->StorageClass mapping (CNPG->premium-rwo, Redis/Monitoring->standard-rwo, Garage->garage-hdd) and Fetch from Cluster - Provider readonly field styled correctly (no solid-black on macOS) - Stale prole.cfg/conf/prole.cfg symlinks removed; all config I/O now resolves env-specific paths via prole_conf.entrypoint_path() ## GKE Autopilot Compatibility (Common Services) - Synology iSCSI StorageClass and static PVs guarded behind PROLE_MODE!=k8s in init_openbao.sh (GKE Autopilot forbids hostPath/iSCSI volumes) - In-cluster Docker registry (hostPath) skipped in k8s mode; GCP Artifact Registry used instead - Kong renamed knoe-svc-kong in k8s mode; all health-check kubectl calls in init_common_services.sh and status_common_services.sh updated accordingly - DNS endpoints switched from *.prole.org to *.knoe.dev in k8s mode (api.knoe.dev, git.knoe.dev, svc.knoe.dev); ingress uses gce class - New GKE-clean Kong manifests under deploy/opentofu/k8s/manifests/prole/: no k3s node affinity, explicit Autopilot resource requests/limits ## Garage S3 Store (GKE) - New garage-statefulset-gcp.yaml targeting garage-hdd StorageClass (pd-standard, avoids SSD_TOTAL_GB quota exhaustion in us-west3) - New storageclass-gcp-hdd.yaml (pd-standard, Retain, WaitForFirstConsumer) - GCP StorageClass manifests skipped on re-runs (Autopilot built-ins are immutable; skip-if-exists guard added) - PVC deletion guard extended to cover any storageClass (not just synology) so stale claims are cleaned before StatefulSet recreation ## Topology (GKE Autopilot) - DaemonSet collector skipped in prod mode (forbidden in kube-system by GKE Warden); Kubernetes-only node facts path used instead - All ready GKE nodes assumed cnpg-eligible and monitoring-eligible without taint/synology-mount checks (skip_collector + assume_nodes_eligible flags) ## KUBECONFIG / kubectl (k8s mode) - actions.py: new elif mode==k8s branch sets KUBECONFIG=~/.kube/config and injects KUBECONTEXT from prole_cfg_data into script env - _build_kubectl_cmd falls back to Global.KUBECONTEXT when init_cluster.selected_kubectx is empty - _activate_selected_gke_cluster persists KUBECONFIG/KUBECONTEXT to prole_cfg_data and saves prole.cfg immediately after get-credentials ## Database Build Screen (GKE) - Registry display shows correct Artifact Registry URL (<region>-docker.pkg.dev/<project>/<namespace>/knoe-db) in green - Build+push: gcloud auth configure-docker, auto-creates AR repository named after SERVICE_NAMESPACE (e.g. knoe-system) if missing, then docker tag + push; falls back to gcr.io if region unavailable - GCP config loaded from conf/prod/gcp.cfg on every screen entry; keys normalised to lowercase so project_id lookup is always consistent ## Config / Namespace persistence - prole_conf.py activate_environment: symlink creation removed; sets CLUSTER_ENV env-var so all subsequent calls resolve correct env directory - knoe/ui/screens/__init__.py: startup config load uses entrypoint_path() instead of hardcoded conf/prole.cfg; seeds SERVICE_NAMESPACE=knoe-system for managed envs so Common Services never defaults to 'default' - cfg.py _save_prole_cfg: saves to env-specific path via entrypoint_path() - etc/prole_cfg.sh: removed all ln -snf symlink creation Co-authored-by: Junie <junie@jetbrains.com>
219 lines
6.9 KiB
Python
219 lines
6.9 KiB
Python
from __future__ import annotations
|
|
|
|
import os
|
|
import tempfile
|
|
|
|
from ._services_common import _LogFn, _detect_mode, _helm, _kubectl, _log, _namespace, _to_bool
|
|
|
|
|
|
def _monitoring_namespace(namespace: str | None, env: dict | None) -> str:
|
|
if env:
|
|
explicit = str(env.get("MONITORING_NAMESPACE") or "").strip()
|
|
if explicit:
|
|
return explicit
|
|
return _namespace(namespace, env, default="monitoring")
|
|
|
|
|
|
def _release_name(env: dict | None) -> str:
|
|
return str((env or {}).get("MONITORING_RELEASE") or "prometheus")
|
|
|
|
|
|
def initialize(
|
|
*,
|
|
namespace: str | None = None,
|
|
env: dict | None = None,
|
|
log: _LogFn | None = None,
|
|
mode: str | None = None,
|
|
) -> None:
|
|
update(namespace=namespace, env=env, log=log, mode=mode)
|
|
|
|
|
|
def start(
|
|
*,
|
|
namespace: str | None = None,
|
|
env: dict | None = None,
|
|
log: _LogFn | None = None,
|
|
mode: str | None = None,
|
|
) -> None:
|
|
update(namespace=namespace, env=env, log=log, mode=mode)
|
|
|
|
|
|
def update(
|
|
*,
|
|
namespace: str | None = None,
|
|
env: dict | None = None,
|
|
log: _LogFn | None = None,
|
|
mode: str | None = None,
|
|
) -> None:
|
|
_detect_mode(mode, env) # mode retained for parity with other owners
|
|
ns = _monitoring_namespace(namespace, env)
|
|
release = _release_name(env)
|
|
chart = str((env or {}).get("MONITORING_CHART") or "prometheus-community/kube-prometheus-stack")
|
|
grafana_password = str((env or {}).get("GRAFANA_ADMIN_PASSWORD") or "prole")
|
|
|
|
# Derive storage class names from PROLE_MONITORING_DATA_DIR (same convention as init_monitoring.sh)
|
|
data_dir = str((env or {}).get("PROLE_MONITORING_DATA_DIR") or "/synology/d004").rstrip("/")
|
|
volume_id = os.path.basename(data_dir) # e.g. "d004"
|
|
sc_prom = f"merlin-local-iscsi-{volume_id}-prometheus"
|
|
sc_alert = f"merlin-local-iscsi-{volume_id}-alertmanager"
|
|
sc_grafana = f"merlin-local-iscsi-{volume_id}-grafana"
|
|
|
|
# Primary monitoring node: prometheus and alertmanager must schedule here to bind local PVs
|
|
monitoring_node = str((env or {}).get("MONITORING_PRIMARY_NODE") or "merlin.prole.org")
|
|
|
|
_log(log, "[MONITORING] Ensuring helm repos")
|
|
_helm(["repo", "add", "prometheus-community", "https://prometheus-community.github.io/helm-charts"], env=env)
|
|
_helm(["repo", "update"], env=env)
|
|
|
|
_kubectl(["create", "namespace", ns], env=env, timeout=60)
|
|
|
|
# Nodes with broken kubelet (e.g. pi.prole.org returning 502) must be excluded
|
|
# from the node-exporter DaemonSet so helm --wait can succeed.
|
|
excluded_nodes = str((env or {}).get("MONITORING_NODE_EXPORTER_EXCLUDE_NODES") or "pi.prole.org")
|
|
excluded_list = [n.strip() for n in excluded_nodes.split(",") if n.strip()]
|
|
|
|
# Node affinity block for components that must land on the monitoring node (local PV binding)
|
|
node_affinity_yaml = (
|
|
" affinity:\n"
|
|
" nodeAffinity:\n"
|
|
" requiredDuringSchedulingIgnoredDuringExecution:\n"
|
|
" nodeSelectorTerms:\n"
|
|
" - matchExpressions:\n"
|
|
" - key: kubernetes.io/hostname\n"
|
|
" operator: In\n"
|
|
" values:\n"
|
|
f" - {monitoring_node}\n"
|
|
)
|
|
|
|
# Node-exporter: exclude broken nodes
|
|
values_yaml = "prometheus-node-exporter:\n"
|
|
if excluded_list:
|
|
values_yaml += (
|
|
" affinity:\n"
|
|
" nodeAffinity:\n"
|
|
" requiredDuringSchedulingIgnoredDuringExecution:\n"
|
|
" nodeSelectorTerms:\n"
|
|
" - matchExpressions:\n"
|
|
" - key: kubernetes.io/hostname\n"
|
|
" operator: NotIn\n"
|
|
" values:\n"
|
|
)
|
|
for node in excluded_list:
|
|
values_yaml += f" - {node}\n"
|
|
|
|
# Prometheus: pin to monitoring node + persistent storage
|
|
values_yaml += (
|
|
"prometheus:\n"
|
|
" prometheusSpec:\n"
|
|
+ node_affinity_yaml
|
|
+ " storageSpec:\n"
|
|
" volumeClaimTemplate:\n"
|
|
" spec:\n"
|
|
f" storageClassName: {sc_prom}\n"
|
|
" accessModes: [ReadWriteOnce]\n"
|
|
" resources:\n"
|
|
" requests:\n"
|
|
" storage: 30Gi\n"
|
|
)
|
|
|
|
# Alertmanager: pin to monitoring node + persistent storage
|
|
values_yaml += (
|
|
"alertmanager:\n"
|
|
" alertmanagerSpec:\n"
|
|
+ node_affinity_yaml
|
|
+ " storage:\n"
|
|
" volumeClaimTemplate:\n"
|
|
" spec:\n"
|
|
f" storageClassName: {sc_alert}\n"
|
|
" accessModes: [ReadWriteOnce]\n"
|
|
" resources:\n"
|
|
" requests:\n"
|
|
" storage: 5Gi\n"
|
|
)
|
|
|
|
# Grafana: persistent storage (Deployment can float; local PV affinity will pull it to merlin)
|
|
values_yaml += (
|
|
"grafana:\n"
|
|
" adminUser: admin\n"
|
|
f" adminPassword: {grafana_password}\n"
|
|
" persistence:\n"
|
|
" type: sts\n"
|
|
" enabled: true\n"
|
|
f" storageClassName: {sc_grafana}\n"
|
|
" accessModes: [ReadWriteOnce]\n"
|
|
" size: 10Gi\n"
|
|
)
|
|
|
|
tmp_values = tempfile.NamedTemporaryFile(
|
|
mode="w", suffix=".yaml", prefix="monitoring-values-", delete=False
|
|
)
|
|
try:
|
|
tmp_values.write(values_yaml)
|
|
tmp_values.flush()
|
|
tmp_values.close()
|
|
|
|
_log(log, f"[MONITORING] Deploying {release} in namespace {ns}")
|
|
_helm(
|
|
[
|
|
"upgrade",
|
|
"--install",
|
|
release,
|
|
chart,
|
|
"--namespace",
|
|
ns,
|
|
"-f",
|
|
tmp_values.name,
|
|
"--wait",
|
|
],
|
|
env=env,
|
|
timeout=600,
|
|
check=True,
|
|
)
|
|
finally:
|
|
os.unlink(tmp_values.name)
|
|
|
|
|
|
def restart(
|
|
*,
|
|
namespace: str | None = None,
|
|
env: dict | None = None,
|
|
log: _LogFn | None = None,
|
|
) -> None:
|
|
ns = _monitoring_namespace(namespace, env)
|
|
_log(log, f"[MONITORING] Restarting grafana deployment in namespace {ns}")
|
|
_kubectl(["-n", ns, "rollout", "restart", "deployment/prometheus-grafana"], env=env, timeout=180)
|
|
|
|
|
|
def stop(
|
|
*,
|
|
namespace: str | None = None,
|
|
env: dict | None = None,
|
|
log: _LogFn | None = None,
|
|
) -> None:
|
|
ns = _monitoring_namespace(namespace, env)
|
|
release = _release_name(env)
|
|
_log(log, f"[MONITORING] Uninstalling release {release} in namespace {ns}")
|
|
_helm(["uninstall", release, "--namespace", ns], env=env, timeout=180)
|
|
|
|
|
|
def status(
|
|
*,
|
|
namespace: str | None = None,
|
|
env: dict | None = None,
|
|
) -> bool:
|
|
ns = _monitoring_namespace(namespace, env)
|
|
dep = _kubectl(
|
|
[
|
|
"-n",
|
|
ns,
|
|
"get",
|
|
"deployment",
|
|
"prometheus-grafana",
|
|
"-o",
|
|
"jsonpath={.status.readyReplicas}",
|
|
],
|
|
env=env,
|
|
timeout=20,
|
|
)
|
|
return _to_bool(dep.returncode == 0 and (dep.stdout or "0").strip() not in {"", "0"})
|