prole/deploy/gcp/gke/knoe-db.yaml
chrisfu d114801758 feat: wire ekosystem UUID system into CNPG manifests (Tasks 1, 3, 4)
Task 1 — ConfigMap + CNPG wiring:
- Add k8s/knoe/knoe-ekosystem-sql.yaml: ConfigMap embedding ekosystem.sql
  and ekosystem_objects.sql for CNPG postInitApplicationSQLRefs
- Add scripts/gen-ekosystem-configmap.py: generation script to keep the
  ConfigMap in sync with knoe-db/schema/ekosystem*.sql source files
- Add Makefile target: make k8s/knoe/knoe-ekosystem-sql.yaml
- Wire postInitApplicationSQLRefs into all three CNPG cluster manifests:
    k8s/knoe/knoe-db.yaml (k3s / prole-service-context production)
    deploy/gcp/gke/knoe-db.yaml (GKE)
    deploy/opentofu/k3s/manifests/knoe/knoe-db.yaml (OpenTofu k3s)
- Add knoe-ekosystem-sql.yaml to k8s/knoe/kustomization.yaml

Task 3 — Python counterpart utility:
- Add knoe/ekosystem.py: thread-safe EkosystemID generator matching the
  PostgreSQL bit layout [49:ts_ms|12:tenant|10:shard|11:seq], with
  decode() and can_access() helpers
- Add tests/test_ekosystem.py: 23 tests covering base36 encoding,
  round-trips, thread safety, can_access, and the spec round-trip assertion

Task 4 — knoe.user ekosystem_uuid column:
- Add ALTER TABLE knoe.user ADD COLUMN IF NOT EXISTS ekosystem_uuid text UNIQUE
  to postInitSQL in all three CNPG manifests

Task 2 (register prole tenant) requires a live DB connection — manual step.
Task 5 (LDAP/AD reconciler) is design-only per spec.

Co-authored-by: Junie <junie@jetbrains.com>
2026-05-30 00:26:48 -07:00

219 lines
11 KiB
YAML
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

apiVersion: postgresql.cnpg.io/v1
kind: Cluster
metadata:
name: knoe-db
namespace: knoe-db-0
spec:
# Run cluster pods as cnpg-backup-sa (annotated for Workload Identity to the
# cnpg-backup@... GCP SA). This is how barman-cloud authenticates to
# gs://knoe-0-backups/ without a static key. Requires CNPG v1.29+.
# The SA is provisioned by etc/init_cnpg_gke.sh § "Apply ServiceAccount + annotate with WI".
serviceAccountName: cnpg-backup-sa
instances: 3
enablePDB: false
# Image pulled from GCP Artifact Registry — set ARTIFACT_REGISTRY in conf/prod/gcp.cfg
# e.g. us-central1-docker.pkg.dev/<project>/knoe-system/knoe-db:<pg-release-tag>
imageName: "${ARTIFACT_REGISTRY}/knoe-db:${KNOE_DB_IMAGE_TAG}"
postgresUID: 100
postgresGID: 101
maxSyncReplicas: 1
affinity:
enablePodAntiAffinity: true
# Keep spread as a preference for small dedicated Standard DB clusters so 3 pods can still
# schedule while nodes reconcile; strict topology can be enforced in later rollout.
podAntiAffinityType: preferred
topologyKey: kubernetes.io/hostname # physical node boundary (not zone)
tolerations:
# Allow scheduling on GKE Spot nodes when explicitly enabled for this DB cluster.
# Without this toleration the cluster-autoscaler predicate simulation fails
# for any MIG whose nodes carry the spot taint, blocking scale-up entirely.
- key: "cloud.google.com/gke-spot"
operator: "Equal"
value: "true"
effect: "NoSchedule"
# nodeSelector removed: knoe-cnpg-0 is a dedicated DB cluster — all nodes are
# available to CNPG. A workload label selector here causes scheduling failures
# when CNPG v1.28 translates it into requiredDuringScheduling nodeAffinity.
postgresql:
parameters:
shared_buffers: 64MB # ~25% of 256Mi request; restore to 128MB when resources increase
pg_stat_statements.max: '10000'
pg_stat_statements.track: all
shared_preload_libraries:
- pg_stat_statements
- pg_tde
pg_hba:
# Local Unix-socket connections (CNPG default + knoe role)
- local all postgres trust
- local all knoe scram-sha-256
# postgres / knoe-db / knoe roles: cluster-internal (RFC1918) only.
# Cluster pod CIDRs: db cluster 10.24.0.0/14, app cluster 10.84.0.0/14;
# node subnet 10.180.0.0/16. 10.0.0.0/8 covers all of those.
- host all postgres 10.0.0.0/8 scram-sha-256
- host knoe knoe-db 10.0.0.0/8 scram-sha-256
- hostssl knoe knoe-db 10.0.0.0/8 scram-sha-256
# PHASE 1 EXTERNAL ACCESS — any member of `knoe_developer`, over TLS+SCRAM.
# `+rolename` in pg_hba matches role membership (not just literal name),
# so `etc/onboard_engineer.sh` adds new engineers via `GRANT knoe_developer
# TO <user>` without ever editing pg_hba — that's the reusable property.
# Phase 2 (queued for Junie) replaces this with libpq OAUTHBEARER:
# hostssl all all 0.0.0.0/0 oauth issuer=https://accounts.google.com validator=knoe_oauth scope="openid email"
- hostssl all +knoe_developer all scram-sha-256
# Internal cluster (RFC1918) — all roles, SCRAM (allows the supabase
# services in app cluster knoe-dev-0 to reach the DB cluster).
- host all all 10.0.0.0/8 scram-sha-256
- hostssl all all 10.0.0.0/8 scram-sha-256
# Block any plaintext from external (TLS required for the public LB)
- hostnossl all all 0.0.0.0/0 reject
# Catch-all reject for anything not matched above
- host all all 0.0.0.0/0 reject
- hostssl all all 0.0.0.0/0 reject
bootstrap:
initdb:
database: knoe-db
owner: knoe
localeCollate: 'en_US.utf8'
localeCType: 'en_US.utf8'
secret:
name: knoe-db-user
postInitTemplateSQL:
# Supabase convention: relocatable extensions live in `extensions`, not
# `public`. Studio's Database Advisor flags `public.pg_stat_statements`
# as a Security warning the moment a user opens the dashboard. Without
# an explicit SCHEMA clause `CREATE EXTENSION` lands the relocatable
# extension in the first writable schema in the connecting role's
# search_path, which for `postgres` is `public`.
- CREATE SCHEMA IF NOT EXISTS extensions;
- CREATE EXTENSION IF NOT EXISTS pg_stat_statements SCHEMA extensions;
postInitSQL:
- DO $do$ BEGIN IF NOT EXISTS (SELECT FROM pg_roles WHERE rolname = 'knoe') THEN CREATE ROLE knoe LOGIN NOSUPERUSER NOCREATEDB NOCREATEROLE INHERIT; END IF; END $do$;
- DO $do$ DECLARE owner_password text; BEGIN SELECT rolpassword INTO owner_password FROM pg_authid WHERE rolname = 'knoe'; IF owner_password IS NOT NULL THEN EXECUTE format('ALTER ROLE knoe PASSWORD %L', owner_password); END IF; END $do$;
- DO $do$ BEGIN IF NOT EXISTS (SELECT FROM pg_roles WHERE rolname = 'knoe_catalog_executor') THEN CREATE ROLE knoe_catalog_executor NOLOGIN; END IF; END $do$;
- CREATE SCHEMA IF NOT EXISTS knoe AUTHORIZATION knoe;
- ALTER SCHEMA knoe OWNER TO knoe;
- REVOKE ALL ON SCHEMA knoe FROM PUBLIC;
- ALTER ROLE knoe SET search_path TO knoe, public;
- CREATE EXTENSION IF NOT EXISTS pg_tde SCHEMA knoe;
- CREATE EXTENSION IF NOT EXISTS pgcrypto SCHEMA knoe;
- CREATE EXTENSION IF NOT EXISTS postgis SCHEMA knoe;
- CREATE EXTENSION IF NOT EXISTS postgis_topology;
- ALTER SCHEMA topology OWNER TO knoe;
- CREATE EXTENSION IF NOT EXISTS vector SCHEMA knoe;
- CREATE EXTENSION IF NOT EXISTS tds_fdw SCHEMA knoe;
- GRANT USAGE ON SCHEMA knoe TO knoe;
- GRANT EXECUTE ON ALL FUNCTIONS IN SCHEMA knoe TO knoe;
- GRANT USAGE ON SCHEMA knoe TO knoe_catalog_executor;
- GRANT EXECUTE ON ALL FUNCTIONS IN SCHEMA knoe TO knoe_catalog_executor;
- ALTER DEFAULT PRIVILEGES FOR ROLE knoe IN SCHEMA knoe GRANT EXECUTE ON FUNCTIONS TO knoe_catalog_executor;
- GRANT EXECUTE ON ALL FUNCTIONS IN SCHEMA topology TO knoe;
- CREATE SCHEMA IF NOT EXISTS storage;
- CREATE SCHEMA IF NOT EXISTS graphql_public;
- DO $do$ BEGIN IF NOT EXISTS (SELECT FROM pg_roles WHERE rolname = 'anon') THEN CREATE ROLE anon NOLOGIN; END IF; END $do$;
- DO $do$ BEGIN IF NOT EXISTS (SELECT FROM pg_roles WHERE rolname = 'authenticator') THEN CREATE ROLE authenticator LOGIN; END IF; END $do$;
- GRANT USAGE ON SCHEMA public TO anon;
- GRANT USAGE ON SCHEMA storage TO anon;
- GRANT USAGE ON SCHEMA graphql_public TO anon;
- GRANT anon TO authenticator;
# demo schema for guest read-only access (evolves over time)
- CREATE SCHEMA IF NOT EXISTS demo;
- DO $do$ BEGIN IF NOT EXISTS (SELECT FROM pg_roles WHERE rolname = 'guest') THEN CREATE ROLE guest NOLOGIN; END IF; END $do$;
- GRANT USAGE ON SCHEMA demo TO guest;
- ALTER DEFAULT PRIVILEGES IN SCHEMA demo GRANT SELECT ON TABLES TO guest;
# knoe.user — identity registry (Knoey Users)
- CREATE TABLE IF NOT EXISTS knoe.user (id SERIAL PRIMARY KEY, username TEXT NOT NULL UNIQUE, realm TEXT NOT NULL DEFAULT 'PROLE.LOCAL', email TEXT, display_name TEXT, tenant_realm TEXT, is_realm_admin BOOLEAN DEFAULT false, created_at TIMESTAMPTZ DEFAULT now(), updated_at TIMESTAMPTZ DEFAULT now());
- CREATE TABLE IF NOT EXISTS knoe.user_role (user_id INT NOT NULL REFERENCES knoe.user(id) ON DELETE CASCADE, role TEXT NOT NULL, granted_at TIMESTAMPTZ DEFAULT now(), PRIMARY KEY (user_id, role));
- GRANT SELECT, INSERT, UPDATE ON knoe.user TO knoe;
- GRANT SELECT, INSERT, UPDATE ON knoe.user_role TO knoe;
- GRANT USAGE, SELECT ON SEQUENCE knoe.user_id_seq TO knoe;
# Task 4: align knoe.user with ekosystem user UUIDs
- ALTER TABLE knoe.user ADD COLUMN IF NOT EXISTS ekosystem_uuid text UNIQUE;
postInitApplicationSQLRefs:
configMapRefs:
- name: knoe-ekosystem-sql
key: ekosystem.sql
- name: knoe-ekosystem-sql
key: ekosystem_objects.sql
managed:
roles:
- name: admin
ensure: present
login: true
superuser: true
comment: "Admin principal — full cluster database access"
- name: guest
ensure: present
login: true
superuser: false
comment: "Guest principal — read-only access to demo schema"
- name: developer
ensure: present
login: false
superuser: false
comment: "Developer group role — granted to knoe-system user accounts"
resources:
requests:
cpu: "100m"
# 512Mi (was 128Mi) — postgres baseline working set is ~290Mi on the
# primary (shared_buffers + wal_buffers + per-backend memory + a small
# OS page cache visible to cgroups), so 128Mi caused the cnpg-grafana
# "Resource Pressure" tile to flag Memory at working_set / request ≈ 2x,
# which the dashboard maps to a red "Data Loss" label (>0.98 ratio).
# Right-sizing to 512Mi puts the steady-state ratio in the green
# "Healthy" zone (<0.8) and gives the scheduler an accurate signal for
# spreading replicas across nodes. Pods still have plenty of headroom:
# 2Gi limit is unchanged.
memory: "512Mi"
limits:
cpu: "500m"
# 2Gi (was 512Mi) — barman-cloud-backup is single-threaded gzip + GCS
# upload buffering and the throughput tops out at the memory ceiling.
# 2Gi cuts a 9 GB DB backup from 3090 min down to 510 min.
memory: "2Gi"
enableSuperuserAccess: true
# CNPG-issued server cert is auto-rotated by the operator. Listing
# pg.0.knoe.dev as an alt DNS name lets engineers connect with
# `sslmode=verify-full` after fetching the CNPG-issued CA cert from the
# `knoe-db-ca` Secret. Phase 1 of the per-engineer psql access plan; replaced
# by libpq OAUTHBEARER + Let's Encrypt in Phase 2.
certificates:
serverAltDNSNames:
- pg.0.knoe.dev
storage:
size: 50Gi
pvcTemplate:
accessModes:
- ReadWriteOnce
resources:
requests:
storage: 50Gi
storageClassName: premium-rwo # pd-ssd; 3×50Gi PGDATA + 3×50Gi WAL = 300Gi total (fits 300GB SSD quota)
walStorage:
size: 50Gi
pvcTemplate:
accessModes:
- ReadWriteOnce
resources:
requests:
storage: 50Gi
storageClassName: premium-rwo # pd-ssd; restore to premium-rwo after quota increase (matches PGDATA above)
monitoring:
# enablePodMonitor and podMonitorRelabelings removed — both fields are
# deprecated by the CNPG operator and will be removed in a future release.
# The PodMonitor is now managed as a sibling resource:
# deploy/gcp/gke/knoe-db-podmonitor.yaml (queue #13).