prole/deploy/gcp/gke/knoe-db.yaml
chrisfu c3fae73de3 fix(cnpg,kong): wire cnpg-backup-sa, migrate PodMonitor, drop DASHBOARD consumer
Three Junie briefs landed in one commit:

#07 — Wire cnpg-backup-sa into CNPG cluster spec (drift R8)
  deploy/gcp/gke/knoe-db.yaml: add spec.serviceAccountName: cnpg-backup-sa
  (requires CNPG v1.29+, which is the live operator version).
  etc/init_cnpg_gke.sh: operator install URL now uses CNPG_OPERATOR_VERSION
  variable (default 1.29.0); new §11 patches knoe-db and
  knoe-db-barman-cloud RoleBindings to add cnpg-backup-sa as a subject
  if not already present — matching the 2026-04-29 live stabilization.

#13 — Migrate off deprecated enablePodMonitor + podMonitorRelabelings
  Both deprecated fields removed from deploy/gcp/gke/knoe-db.yaml
  spec.monitoring. New deploy/gcp/gke/knoe-db-podmonitor.yaml carries
  the PodMonitor with the cluster relabeling rule (cnpg.io/cluster pod
  label → cluster label; required for all 85 CNPG Grafana panels).
  Apply alongside knoe-db.yaml on next cluster patch.

#15 — Remove dead DASHBOARD consumer + basicauth_credentials
  supabase/helm/knoe-supabase:
  - wrapper.sh: drop DASHBOARD_USERNAME / DASHBOARD_PASSWORD envsubst lines
  - config.yaml: drop DASHBOARD consumer + basicauth_credentials block
  - kong/deployment.yaml: drop both DASHBOARD env-var secret refs
  - values.yaml: rename secret.dashboard → secret.openai (apiKey only;
    username/password dropped — no enforcer since commit 25f1b2e)
  - secrets/dashboard.yaml + _helpers.tpl: renamed to openai /
    supabase.secret.openai
  - studio/deployment.yaml: reads from secret.openai.apiKey
  - ci/example.yaml: updated to secret.openai.apiKey
  helm template confirms knoe-supabase-openai secret referenced; no
  DASHBOARD output.

docs/TODO.md: queue items #7, #13, #15 + drift row R8 archived to Done.

Co-authored-by: Junie <junie@jetbrains.com>
2026-05-02 03:08:34 -07:00

211 lines
10 KiB
YAML
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

apiVersion: postgresql.cnpg.io/v1
kind: Cluster
metadata:
name: knoe-db
namespace: knoe-db-0
spec:
# Run cluster pods as cnpg-backup-sa (annotated for Workload Identity to the
# cnpg-backup@... GCP SA). This is how barman-cloud authenticates to
# gs://knoe-0-backups/ without a static key. Requires CNPG v1.29+.
# The SA is provisioned by etc/init_cnpg_gke.sh § "Apply ServiceAccount + annotate with WI".
serviceAccountName: cnpg-backup-sa
instances: 3
enablePDB: false
# Image pulled from GCP Artifact Registry — set ARTIFACT_REGISTRY in conf/prod/gcp.cfg
# e.g. us-central1-docker.pkg.dev/<project>/knoe-system/knoe-db:<pg-release-tag>
imageName: "${ARTIFACT_REGISTRY}/knoe-db:${KNOE_DB_IMAGE_TAG}"
postgresUID: 100
postgresGID: 101
maxSyncReplicas: 1
affinity:
enablePodAntiAffinity: true
# Keep spread as a preference for small dedicated Standard DB clusters so 3 pods can still
# schedule while nodes reconcile; strict topology can be enforced in later rollout.
podAntiAffinityType: preferred
topologyKey: kubernetes.io/hostname # physical node boundary (not zone)
tolerations:
# Allow scheduling on GKE Spot nodes when explicitly enabled for this DB cluster.
# Without this toleration the cluster-autoscaler predicate simulation fails
# for any MIG whose nodes carry the spot taint, blocking scale-up entirely.
- key: "cloud.google.com/gke-spot"
operator: "Equal"
value: "true"
effect: "NoSchedule"
# nodeSelector removed: knoe-cnpg-0 is a dedicated DB cluster — all nodes are
# available to CNPG. A workload label selector here causes scheduling failures
# when CNPG v1.28 translates it into requiredDuringScheduling nodeAffinity.
postgresql:
parameters:
shared_buffers: 64MB # ~25% of 256Mi request; restore to 128MB when resources increase
pg_stat_statements.max: '10000'
pg_stat_statements.track: all
shared_preload_libraries:
- pg_stat_statements
- pg_tde
pg_hba:
# Local Unix-socket connections (CNPG default + knoe role)
- local all postgres trust
- local all knoe scram-sha-256
# postgres / knoe-db / knoe roles: cluster-internal (RFC1918) only.
# Cluster pod CIDRs: db cluster 10.24.0.0/14, app cluster 10.84.0.0/14;
# node subnet 10.180.0.0/16. 10.0.0.0/8 covers all of those.
- host all postgres 10.0.0.0/8 scram-sha-256
- host knoe knoe-db 10.0.0.0/8 scram-sha-256
- hostssl knoe knoe-db 10.0.0.0/8 scram-sha-256
# PHASE 1 EXTERNAL ACCESS — any member of `knoe_developer`, over TLS+SCRAM.
# `+rolename` in pg_hba matches role membership (not just literal name),
# so `etc/onboard_engineer.sh` adds new engineers via `GRANT knoe_developer
# TO <user>` without ever editing pg_hba — that's the reusable property.
# Phase 2 (queued for Junie) replaces this with libpq OAUTHBEARER:
# hostssl all all 0.0.0.0/0 oauth issuer=https://accounts.google.com validator=knoe_oauth scope="openid email"
- hostssl all +knoe_developer all scram-sha-256
# Internal cluster (RFC1918) — all roles, SCRAM (allows the supabase
# services in app cluster knoe-dev-0 to reach the DB cluster).
- host all all 10.0.0.0/8 scram-sha-256
- hostssl all all 10.0.0.0/8 scram-sha-256
# Block any plaintext from external (TLS required for the public LB)
- hostnossl all all 0.0.0.0/0 reject
# Catch-all reject for anything not matched above
- host all all 0.0.0.0/0 reject
- hostssl all all 0.0.0.0/0 reject
bootstrap:
initdb:
database: knoe-db
owner: knoe
localeCollate: 'en_US.utf8'
localeCType: 'en_US.utf8'
secret:
name: knoe-db-user
postInitTemplateSQL:
# Supabase convention: relocatable extensions live in `extensions`, not
# `public`. Studio's Database Advisor flags `public.pg_stat_statements`
# as a Security warning the moment a user opens the dashboard. Without
# an explicit SCHEMA clause `CREATE EXTENSION` lands the relocatable
# extension in the first writable schema in the connecting role's
# search_path, which for `postgres` is `public`.
- CREATE SCHEMA IF NOT EXISTS extensions;
- CREATE EXTENSION IF NOT EXISTS pg_stat_statements SCHEMA extensions;
postInitSQL:
- DO $do$ BEGIN IF NOT EXISTS (SELECT FROM pg_roles WHERE rolname = 'knoe') THEN CREATE ROLE knoe LOGIN NOSUPERUSER NOCREATEDB NOCREATEROLE INHERIT; END IF; END $do$;
- DO $do$ DECLARE owner_password text; BEGIN SELECT rolpassword INTO owner_password FROM pg_authid WHERE rolname = 'knoe'; IF owner_password IS NOT NULL THEN EXECUTE format('ALTER ROLE knoe PASSWORD %L', owner_password); END IF; END $do$;
- DO $do$ BEGIN IF NOT EXISTS (SELECT FROM pg_roles WHERE rolname = 'knoe_catalog_executor') THEN CREATE ROLE knoe_catalog_executor NOLOGIN; END IF; END $do$;
- CREATE SCHEMA IF NOT EXISTS knoe AUTHORIZATION knoe;
- ALTER SCHEMA knoe OWNER TO knoe;
- REVOKE ALL ON SCHEMA knoe FROM PUBLIC;
- ALTER ROLE knoe SET search_path TO knoe, public;
- CREATE EXTENSION IF NOT EXISTS pg_tde SCHEMA knoe;
- CREATE EXTENSION IF NOT EXISTS pgcrypto SCHEMA knoe;
- CREATE EXTENSION IF NOT EXISTS postgis SCHEMA knoe;
- CREATE EXTENSION IF NOT EXISTS postgis_topology;
- ALTER SCHEMA topology OWNER TO knoe;
- CREATE EXTENSION IF NOT EXISTS vector SCHEMA knoe;
- CREATE EXTENSION IF NOT EXISTS tds_fdw SCHEMA knoe;
- GRANT USAGE ON SCHEMA knoe TO knoe;
- GRANT EXECUTE ON ALL FUNCTIONS IN SCHEMA knoe TO knoe;
- GRANT USAGE ON SCHEMA knoe TO knoe_catalog_executor;
- GRANT EXECUTE ON ALL FUNCTIONS IN SCHEMA knoe TO knoe_catalog_executor;
- ALTER DEFAULT PRIVILEGES FOR ROLE knoe IN SCHEMA knoe GRANT EXECUTE ON FUNCTIONS TO knoe_catalog_executor;
- GRANT EXECUTE ON ALL FUNCTIONS IN SCHEMA topology TO knoe;
- CREATE SCHEMA IF NOT EXISTS storage;
- CREATE SCHEMA IF NOT EXISTS graphql_public;
- DO $do$ BEGIN IF NOT EXISTS (SELECT FROM pg_roles WHERE rolname = 'anon') THEN CREATE ROLE anon NOLOGIN; END IF; END $do$;
- DO $do$ BEGIN IF NOT EXISTS (SELECT FROM pg_roles WHERE rolname = 'authenticator') THEN CREATE ROLE authenticator LOGIN; END IF; END $do$;
- GRANT USAGE ON SCHEMA public TO anon;
- GRANT USAGE ON SCHEMA storage TO anon;
- GRANT USAGE ON SCHEMA graphql_public TO anon;
- GRANT anon TO authenticator;
# demo schema for guest read-only access (evolves over time)
- CREATE SCHEMA IF NOT EXISTS demo;
- DO $do$ BEGIN IF NOT EXISTS (SELECT FROM pg_roles WHERE rolname = 'guest') THEN CREATE ROLE guest NOLOGIN; END IF; END $do$;
- GRANT USAGE ON SCHEMA demo TO guest;
- ALTER DEFAULT PRIVILEGES IN SCHEMA demo GRANT SELECT ON TABLES TO guest;
# knoe.user — identity registry (Knoey Users)
- CREATE TABLE IF NOT EXISTS knoe.user (id SERIAL PRIMARY KEY, username TEXT NOT NULL UNIQUE, realm TEXT NOT NULL DEFAULT 'PROLE.LOCAL', email TEXT, display_name TEXT, tenant_realm TEXT, is_realm_admin BOOLEAN DEFAULT false, created_at TIMESTAMPTZ DEFAULT now(), updated_at TIMESTAMPTZ DEFAULT now());
- CREATE TABLE IF NOT EXISTS knoe.user_role (user_id INT NOT NULL REFERENCES knoe.user(id) ON DELETE CASCADE, role TEXT NOT NULL, granted_at TIMESTAMPTZ DEFAULT now(), PRIMARY KEY (user_id, role));
- GRANT SELECT, INSERT, UPDATE ON knoe.user TO knoe;
- GRANT SELECT, INSERT, UPDATE ON knoe.user_role TO knoe;
- GRANT USAGE, SELECT ON SEQUENCE knoe.user_id_seq TO knoe;
managed:
roles:
- name: admin
ensure: present
login: true
superuser: true
comment: "Admin principal — full cluster database access"
- name: guest
ensure: present
login: true
superuser: false
comment: "Guest principal — read-only access to demo schema"
- name: developer
ensure: present
login: false
superuser: false
comment: "Developer group role — granted to knoe-system user accounts"
resources:
requests:
cpu: "100m"
# 512Mi (was 128Mi) — postgres baseline working set is ~290Mi on the
# primary (shared_buffers + wal_buffers + per-backend memory + a small
# OS page cache visible to cgroups), so 128Mi caused the cnpg-grafana
# "Resource Pressure" tile to flag Memory at working_set / request ≈ 2x,
# which the dashboard maps to a red "Data Loss" label (>0.98 ratio).
# Right-sizing to 512Mi puts the steady-state ratio in the green
# "Healthy" zone (<0.8) and gives the scheduler an accurate signal for
# spreading replicas across nodes. Pods still have plenty of headroom:
# 2Gi limit is unchanged.
memory: "512Mi"
limits:
cpu: "500m"
# 2Gi (was 512Mi) — barman-cloud-backup is single-threaded gzip + GCS
# upload buffering and the throughput tops out at the memory ceiling.
# 2Gi cuts a 9 GB DB backup from 3090 min down to 510 min.
memory: "2Gi"
enableSuperuserAccess: true
# CNPG-issued server cert is auto-rotated by the operator. Listing
# pg.0.knoe.dev as an alt DNS name lets engineers connect with
# `sslmode=verify-full` after fetching the CNPG-issued CA cert from the
# `knoe-db-ca` Secret. Phase 1 of the per-engineer psql access plan; replaced
# by libpq OAUTHBEARER + Let's Encrypt in Phase 2.
certificates:
serverAltDNSNames:
- pg.0.knoe.dev
storage:
size: 50Gi
pvcTemplate:
accessModes:
- ReadWriteOnce
resources:
requests:
storage: 50Gi
storageClassName: premium-rwo # pd-ssd; 3×50Gi PGDATA + 3×50Gi WAL = 300Gi total (fits 300GB SSD quota)
walStorage:
size: 50Gi
pvcTemplate:
accessModes:
- ReadWriteOnce
resources:
requests:
storage: 50Gi
storageClassName: premium-rwo # pd-ssd; restore to premium-rwo after quota increase (matches PGDATA above)
monitoring:
# enablePodMonitor and podMonitorRelabelings removed — both fields are
# deprecated by the CNPG operator and will be removed in a future release.
# The PodMonitor is now managed as a sibling resource:
# deploy/gcp/gke/knoe-db-podmonitor.yaml (queue #13).