apiVersion: postgresql.cnpg.io/v1 kind: Cluster metadata: name: knoe-db namespace: knoe-db-0 spec: # Run cluster pods as cnpg-backup-sa (annotated for Workload Identity to the # cnpg-backup@... GCP SA). This is how barman-cloud authenticates to # gs://knoe-0-backups/ without a static key. Requires CNPG v1.29+. # The SA is provisioned by etc/init_cnpg_gke.sh § "Apply ServiceAccount + annotate with WI". serviceAccountName: cnpg-backup-sa instances: 3 enablePDB: false # Image pulled from GCP Artifact Registry — set ARTIFACT_REGISTRY in conf/prod/gcp.cfg # e.g. us-central1-docker.pkg.dev//knoe-system/knoe-db: imageName: "${ARTIFACT_REGISTRY}/knoe-db:${KNOE_DB_IMAGE_TAG}" postgresUID: 100 postgresGID: 101 maxSyncReplicas: 1 affinity: enablePodAntiAffinity: true # Keep spread as a preference for small dedicated Standard DB clusters so 3 pods can still # schedule while nodes reconcile; strict topology can be enforced in later rollout. podAntiAffinityType: preferred topologyKey: kubernetes.io/hostname # physical node boundary (not zone) tolerations: # Allow scheduling on GKE Spot nodes when explicitly enabled for this DB cluster. # Without this toleration the cluster-autoscaler predicate simulation fails # for any MIG whose nodes carry the spot taint, blocking scale-up entirely. - key: "cloud.google.com/gke-spot" operator: "Equal" value: "true" effect: "NoSchedule" # nodeSelector removed: knoe-cnpg-0 is a dedicated DB cluster — all nodes are # available to CNPG. A workload label selector here causes scheduling failures # when CNPG v1.28 translates it into requiredDuringScheduling nodeAffinity. postgresql: parameters: shared_buffers: 64MB # ~25% of 256Mi request; restore to 128MB when resources increase pg_stat_statements.max: '10000' pg_stat_statements.track: all shared_preload_libraries: - pg_stat_statements - pg_tde pg_hba: # Local Unix-socket connections (CNPG default + knoe role) - local all postgres trust - local all knoe scram-sha-256 # postgres / knoe-db / knoe roles: cluster-internal (RFC1918) only. # Cluster pod CIDRs: db cluster 10.24.0.0/14, app cluster 10.84.0.0/14; # node subnet 10.180.0.0/16. 10.0.0.0/8 covers all of those. - host all postgres 10.0.0.0/8 scram-sha-256 - host knoe knoe-db 10.0.0.0/8 scram-sha-256 - hostssl knoe knoe-db 10.0.0.0/8 scram-sha-256 # PHASE 1 EXTERNAL ACCESS — any member of `knoe_developer`, over TLS+SCRAM. # `+rolename` in pg_hba matches role membership (not just literal name), # so `etc/onboard_engineer.sh` adds new engineers via `GRANT knoe_developer # TO ` without ever editing pg_hba — that's the reusable property. # Phase 2 (queued for Junie) replaces this with libpq OAUTHBEARER: # hostssl all all 0.0.0.0/0 oauth issuer=https://accounts.google.com validator=knoe_oauth scope="openid email" - hostssl all +knoe_developer all scram-sha-256 # Internal cluster (RFC1918) — all roles, SCRAM (allows the supabase # services in app cluster knoe-dev-0 to reach the DB cluster). - host all all 10.0.0.0/8 scram-sha-256 - hostssl all all 10.0.0.0/8 scram-sha-256 # Block any plaintext from external (TLS required for the public LB) - hostnossl all all 0.0.0.0/0 reject # Catch-all reject for anything not matched above - host all all 0.0.0.0/0 reject - hostssl all all 0.0.0.0/0 reject bootstrap: initdb: database: knoe-db owner: knoe localeCollate: 'en_US.utf8' localeCType: 'en_US.utf8' secret: name: knoe-db-user postInitTemplateSQL: # Supabase convention: relocatable extensions live in `extensions`, not # `public`. Studio's Database Advisor flags `public.pg_stat_statements` # as a Security warning the moment a user opens the dashboard. Without # an explicit SCHEMA clause `CREATE EXTENSION` lands the relocatable # extension in the first writable schema in the connecting role's # search_path, which for `postgres` is `public`. - CREATE SCHEMA IF NOT EXISTS extensions; - CREATE EXTENSION IF NOT EXISTS pg_stat_statements SCHEMA extensions; postInitSQL: - DO $do$ BEGIN IF NOT EXISTS (SELECT FROM pg_roles WHERE rolname = 'knoe') THEN CREATE ROLE knoe LOGIN NOSUPERUSER NOCREATEDB NOCREATEROLE INHERIT; END IF; END $do$; - DO $do$ DECLARE owner_password text; BEGIN SELECT rolpassword INTO owner_password FROM pg_authid WHERE rolname = 'knoe'; IF owner_password IS NOT NULL THEN EXECUTE format('ALTER ROLE knoe PASSWORD %L', owner_password); END IF; END $do$; - DO $do$ BEGIN IF NOT EXISTS (SELECT FROM pg_roles WHERE rolname = 'knoe_catalog_executor') THEN CREATE ROLE knoe_catalog_executor NOLOGIN; END IF; END $do$; - CREATE SCHEMA IF NOT EXISTS knoe AUTHORIZATION knoe; - ALTER SCHEMA knoe OWNER TO knoe; - REVOKE ALL ON SCHEMA knoe FROM PUBLIC; - ALTER ROLE knoe SET search_path TO knoe, public; - CREATE EXTENSION IF NOT EXISTS pg_tde SCHEMA knoe; - CREATE EXTENSION IF NOT EXISTS pgcrypto SCHEMA knoe; - CREATE EXTENSION IF NOT EXISTS postgis SCHEMA knoe; - CREATE EXTENSION IF NOT EXISTS postgis_topology; - ALTER SCHEMA topology OWNER TO knoe; - CREATE EXTENSION IF NOT EXISTS vector SCHEMA knoe; - CREATE EXTENSION IF NOT EXISTS tds_fdw SCHEMA knoe; - GRANT USAGE ON SCHEMA knoe TO knoe; - GRANT EXECUTE ON ALL FUNCTIONS IN SCHEMA knoe TO knoe; - GRANT USAGE ON SCHEMA knoe TO knoe_catalog_executor; - GRANT EXECUTE ON ALL FUNCTIONS IN SCHEMA knoe TO knoe_catalog_executor; - ALTER DEFAULT PRIVILEGES FOR ROLE knoe IN SCHEMA knoe GRANT EXECUTE ON FUNCTIONS TO knoe_catalog_executor; - GRANT EXECUTE ON ALL FUNCTIONS IN SCHEMA topology TO knoe; - CREATE SCHEMA IF NOT EXISTS storage; - CREATE SCHEMA IF NOT EXISTS graphql_public; - DO $do$ BEGIN IF NOT EXISTS (SELECT FROM pg_roles WHERE rolname = 'anon') THEN CREATE ROLE anon NOLOGIN; END IF; END $do$; - DO $do$ BEGIN IF NOT EXISTS (SELECT FROM pg_roles WHERE rolname = 'authenticator') THEN CREATE ROLE authenticator LOGIN; END IF; END $do$; - GRANT USAGE ON SCHEMA public TO anon; - GRANT USAGE ON SCHEMA storage TO anon; - GRANT USAGE ON SCHEMA graphql_public TO anon; - GRANT anon TO authenticator; # demo schema for guest read-only access (evolves over time) - CREATE SCHEMA IF NOT EXISTS demo; - DO $do$ BEGIN IF NOT EXISTS (SELECT FROM pg_roles WHERE rolname = 'guest') THEN CREATE ROLE guest NOLOGIN; END IF; END $do$; - GRANT USAGE ON SCHEMA demo TO guest; - ALTER DEFAULT PRIVILEGES IN SCHEMA demo GRANT SELECT ON TABLES TO guest; # knoe.user — identity registry (Knoey Users) - CREATE TABLE IF NOT EXISTS knoe.user (id SERIAL PRIMARY KEY, username TEXT NOT NULL UNIQUE, realm TEXT NOT NULL DEFAULT 'PROLE.LOCAL', email TEXT, display_name TEXT, tenant_realm TEXT, is_realm_admin BOOLEAN DEFAULT false, created_at TIMESTAMPTZ DEFAULT now(), updated_at TIMESTAMPTZ DEFAULT now()); - CREATE TABLE IF NOT EXISTS knoe.user_role (user_id INT NOT NULL REFERENCES knoe.user(id) ON DELETE CASCADE, role TEXT NOT NULL, granted_at TIMESTAMPTZ DEFAULT now(), PRIMARY KEY (user_id, role)); - GRANT SELECT, INSERT, UPDATE ON knoe.user TO knoe; - GRANT SELECT, INSERT, UPDATE ON knoe.user_role TO knoe; - GRANT USAGE, SELECT ON SEQUENCE knoe.user_id_seq TO knoe; # Task 4: align knoe.user with ekosystem user UUIDs - ALTER TABLE knoe.user ADD COLUMN IF NOT EXISTS ekosystem_uuid text UNIQUE; postInitApplicationSQLRefs: configMapRefs: - name: knoe-ekosystem-sql key: ekosystem.sql - name: knoe-ekosystem-sql key: ekosystem_objects.sql managed: roles: - name: admin ensure: present login: true superuser: true comment: "Admin principal — full cluster database access" - name: guest ensure: present login: true superuser: false comment: "Guest principal — read-only access to demo schema" - name: developer ensure: present login: false superuser: false comment: "Developer group role — granted to knoe-system user accounts" resources: requests: cpu: "100m" # 512Mi (was 128Mi) — postgres baseline working set is ~290Mi on the # primary (shared_buffers + wal_buffers + per-backend memory + a small # OS page cache visible to cgroups), so 128Mi caused the cnpg-grafana # "Resource Pressure" tile to flag Memory at working_set / request ≈ 2x, # which the dashboard maps to a red "Data Loss" label (>0.98 ratio). # Right-sizing to 512Mi puts the steady-state ratio in the green # "Healthy" zone (<0.8) and gives the scheduler an accurate signal for # spreading replicas across nodes. Pods still have plenty of headroom: # 2Gi limit is unchanged. memory: "512Mi" limits: cpu: "500m" # 2Gi (was 512Mi) — barman-cloud-backup is single-threaded gzip + GCS # upload buffering and the throughput tops out at the memory ceiling. # 2Gi cuts a 9 GB DB backup from 30–90 min down to 5–10 min. memory: "2Gi" enableSuperuserAccess: true # CNPG-issued server cert is auto-rotated by the operator. Listing # pg.0.knoe.dev as an alt DNS name lets engineers connect with # `sslmode=verify-full` after fetching the CNPG-issued CA cert from the # `knoe-db-ca` Secret. Phase 1 of the per-engineer psql access plan; replaced # by libpq OAUTHBEARER + Let's Encrypt in Phase 2. certificates: serverAltDNSNames: - pg.0.knoe.dev storage: size: 50Gi pvcTemplate: accessModes: - ReadWriteOnce resources: requests: storage: 50Gi storageClassName: premium-rwo # pd-ssd; 3×50Gi PGDATA + 3×50Gi WAL = 300Gi total (fits 300GB SSD quota) walStorage: size: 50Gi pvcTemplate: accessModes: - ReadWriteOnce resources: requests: storage: 50Gi storageClassName: premium-rwo # pd-ssd; restore to premium-rwo after quota increase (matches PGDATA above) monitoring: # enablePodMonitor and podMonitorRelabelings removed — both fields are # deprecated by the CNPG operator and will be removed in a future release. # The PodMonitor is now managed as a sibling resource: # deploy/gcp/gke/knoe-db-podmonitor.yaml (queue #13).