# kube-prometheus-stack (kps) Helm values — GKE deploy mode. # # This file is the canonical source of the values applied to the `kps` # release in namespace `monitoring` on knoe-dev-0. It captures the existing # inline values that were `helm install`'d 2026-04-28 PLUS today's grafana # subpath + Google OAuth additions. # # Apply with: # helm upgrade kps prometheus-community/kube-prometheus-stack \ # --namespace monitoring \ # -f monitoring/kps-values-gke.yaml # # NOTE on prole.org/node-role nodeSelectors below: this is a legacy label # from the prole-era cluster — knoe-dev-0 nodes still carry it for # compatibility. Once the cluster is fully relabelled to knoe.dev/* a # follow-up will rename these. Tracked in docs/TODO.md (Reality TODO list). --- alertmanager: alertmanagerSpec: affinity: nodeAffinity: requiredDuringSchedulingIgnoredDuringExecution: nodeSelectorTerms: - matchExpressions: - key: prole.org/node-role operator: In values: - general nodeSelector: prole.org/node-role: general storage: volumeClaimTemplate: spec: accessModes: - ReadWriteOnce resources: requests: storage: 5Gi storageClassName: standard-hdd grafana: enabled: true # adminPassword is the local break-glass; primary auth is Google OAuth (below). # Rotate this whenever a person who once knew it leaves the team. Stored in # 1Password (admin scope, separate from per-engineer entries). adminPassword: admin affinity: nodeAffinity: requiredDuringSchedulingIgnoredDuringExecution: nodeSelectorTerms: - matchExpressions: - key: prole.org/node-role operator: In values: - general nodeSelector: prole.org/node-role: general initChownData: enabled: false persistence: accessModes: - ReadWriteOnce enabled: true size: 10Gi storageClassName: standard-hdd type: sts service: port: 80 targetPort: 3000 sidecar: dashboards: enabled: true label: grafana_dashboard labelValue: "1" datasources: enabled: true label: grafana_datasource labelValue: "1" # Mount Google OAuth client credentials from the grafana-google-oidc Secret. # Created by etc/init_grafana_oauth.sh from etc/secrets/grafana-google-oidc-*. # Provides GF_AUTH_GOOGLE_CLIENT_ID and GF_AUTH_GOOGLE_CLIENT_SECRET env vars. envFromSecret: grafana-google-oidc # Additional Prometheus datasource pointing at the DB-cluster (knoe-dev-cnpg-0) # Prometheus install (monitoring/kps-cnpg-values.yaml). The cnpg-grafana # dashboards' DS_PROMETHEUS template variable can switch to this datasource # to render CNPG metrics — Kubernetes service discovery is cluster-local, so # the local app-cluster Prometheus can't see CNPG pods on its own. # Reachable across clusters via the internal-LB IP allocated to # `prometheus-cnpg-ilb` in monitoring ns of the DB cluster (same VPC subnet). additionalDataSources: - name: cnpg-prometheus # Stable UID so the cnpg-grafana dashboard's DS_PROMETHEUS current.value # (set by the dashboard transform spec) can refer to this datasource by # name instead of an auto-generated random UID. Also makes the datasource # identity-preserving across kps re-installs (which would otherwise # regenerate the auto UID and silently break the dashboard). uid: cnpg-prometheus type: prometheus url: http://10.180.15.216:9090 access: proxy isDefault: false editable: false jsonData: timeInterval: 15s manageAlerts: false prometheusType: Prometheus # grafana.ini — appended to the chart's defaults. # Subpath: served at https://svc.knoe.dev/grafana via the knoe-svc-kong route. # Auth: Google OAuth restricted to @knoey.com Workspace; chrisfu + ron mapped # to Admin via JMESPath, everyone else in the Workspace gets Editor. grafana.ini: server: domain: svc.knoe.dev root_url: "https://svc.knoe.dev/grafana" serve_from_sub_path: true security: # Behind Kong + GCE LB on HTTPS — issue cookies with the Secure flag set # so browsers send them on every request. Without this, Grafana 13's # session-token rotation logic fights with the proxy chain and every # API call returns 401 with `[session.token.rotate] token needs to be # rotated`, breaking panel data fetches in a loop. cookie_secure: true cookie_samesite: lax # Grafana 10+ enforces a same-origin CSRF check on every state-changing # method (POST/PUT/PATCH/DELETE), comparing Origin/Referer to root_url. # Behind Kong → grafana the upstream Host header is the cluster-internal # service name (kps-grafana.monitoring.svc.cluster.local), not # svc.knoe.dev, so the CSRF middleware rejects POSTs from the public # origin with 403. Symptoms: every /api/ds/query → 403 (panels render # empty); Share → Copy Link → "origin not allowed" toast. # Trust the public hostname explicitly. csrf_trusted_origins takes # space-separated bare hostnames (no scheme); csrf_additional_headers # tells Grafana to also accept X-Forwarded-Host (which Kong sets # correctly) as a valid origin source — this is the documented pairing # for proxied installs. csrf_trusted_origins: svc.knoe.dev csrf_additional_headers: X-Forwarded-Host "live": # WebSocket origin check (Grafana live streaming, dashboard refresh). # The default rejects any Origin not exactly matching root_url, which # surfaces in the UI as a "origin not allowed" toast popup. Allow the # public hostname explicitly. Multiple origins comma-separated if needed. allowed_origins: "https://svc.knoe.dev" auth: # Keep the local login form available as a break-glass for adminPassword. disable_login_form: false # Increase the session-token rotation interval so the rotation race # condition is rare enough not to break panel queries. Default was # 10 minutes; bumped to 24 hours. The "right" fix is figuring out why # rotation fails through the Kong proxy at all (TODO follow-up); this # is the pragmatic mitigation for tonight. token_rotation_interval_minutes: 1440 "auth.google": enabled: true # client_id / client_secret arrive via env (GF_AUTH_GOOGLE_CLIENT_ID etc.) # from the grafana-google-oidc Secret. Don't duplicate here. allowed_domains: knoey.com scopes: "openid email profile" auth_url: https://accounts.google.com/o/oauth2/v2/auth token_url: https://oauth2.googleapis.com/token api_url: https://openidconnect.googleapis.com/v1/userinfo # JMESPath: chrisfu + ron get Admin; every other knoey.com user gets Editor. # auto_assign_org_role below is the fallback if role_attribute_path produces # an empty result. role_attribute_path: "contains(['chrisfu@knoey.com', 'ron@knoey.com'], email) && 'Admin' || 'Editor'" # Re-evaluate role on each login so a dropped engineer immediately loses # the elevated bit; if you want manual elevation in Studio to survive, # set this to true. skip_org_role_sync: false users: auto_assign_org_role: Editor kube-state-metrics: affinity: nodeAffinity: requiredDuringSchedulingIgnoredDuringExecution: nodeSelectorTerms: - matchExpressions: - key: prole.org/node-role operator: In values: - general nodeSelector: prole.org/node-role: general prometheus: prometheusSpec: additionalScrapeConfigs: - job_name: kubernetes-pods kubernetes_sd_configs: - role: pod relabel_configs: - action: keep regex: true source_labels: - __meta_kubernetes_pod_annotation_prometheus_io_scrape - action: replace regex: (.+) source_labels: - __meta_kubernetes_pod_annotation_prometheus_io_path target_label: __metrics_path__ - action: replace regex: (.*?):\d+;(\d+) replacement: $1:$2 source_labels: - __address__ - __meta_kubernetes_pod_annotation_prometheus_io_port target_label: __address__ - job_name: cnpg-metrics kubernetes_sd_configs: - role: pod relabel_configs: - action: keep regex: .+ source_labels: - __meta_kubernetes_pod_label_cnpg_io_cluster - action: keep regex: Running source_labels: - __meta_kubernetes_pod_phase - action: replace replacement: $1:9187 source_labels: - __meta_kubernetes_pod_ip target_label: __address__ - action: replace source_labels: - __meta_kubernetes_namespace target_label: namespace - action: replace source_labels: - __meta_kubernetes_pod_name target_label: pod - action: replace source_labels: - __meta_kubernetes_pod_label_cnpg_io_cluster target_label: cluster affinity: nodeAffinity: requiredDuringSchedulingIgnoredDuringExecution: nodeSelectorTerms: - matchExpressions: - key: prole.org/node-role operator: In values: - general nodeSelector: prole.org/node-role: general storageSpec: volumeClaimTemplate: spec: accessModes: - ReadWriteOnce resources: requests: storage: 30Gi storageClassName: standard-hdd prometheus-node-exporter: affinity: nodeAffinity: requiredDuringSchedulingIgnoredDuringExecution: nodeSelectorTerms: - matchExpressions: - key: kubernetes.io/hostname operator: NotIn values: - pi.prole.org nodeSelector: prole.org/node-role: general prometheusOperator: affinity: nodeAffinity: requiredDuringSchedulingIgnoredDuringExecution: nodeSelectorTerms: - matchExpressions: - key: prole.org/node-role operator: In values: - general nodeSelector: prole.org/node-role: general