diff --git a/CLAUDE.md b/CLAUDE.md index 30377d0..b679984 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -42,9 +42,32 @@ Setup script: `etc/init_cnpg_gke.sh` (creates buckets, GCP SA, WI binding, appli --- -## install.sh pre-flight checklist +## Service mesh (Cloud Service Mesh / Istio) -Before running `./install.sh` (especially "Initialization Scripts"), confirm these are correct in `conf/prod/prole.cfg`: +Both clusters are registered in the **knoe-0** GCP fleet with automatic Cloud Service Mesh management. This is automated in `scripts/reset_clusters.sh` (Phase 7) — no longer requires GCP web console. + +```bash +# Check mesh provisioning status (~10 min after cluster creation): +gcloud container fleet mesh describe --project=plenary-truck-485623-p7 + +# Manual re-registration if needed: +gcloud container fleet memberships register knoe-dev-0 \ + --gke-cluster=us-west3/knoe-dev-0 \ + --enable-workload-identity \ + --project=plenary-truck-485623-p7 +gcloud container fleet mesh update \ + --management=automatic \ + --memberships=knoe-dev-0 \ + --project=plenary-truck-485623-p7 +``` + +--- + +## install.sh / deploy.sh pre-flight checklist + +### `conf/prod/prole.cfg` (interactive installer — `./install.sh`) + +Before running `./install.sh` (especially "Initialization Scripts"), confirm these are correct: ```ini [Inputs] @@ -54,12 +77,35 @@ env_setup.APP_CLUSTER_KUBECONTEXT = gke_plenary-truck-485623-p7_us-west3_knoe env_setup.DB_CLUSTER_KUBECONTEXT = gke_plenary-truck-485623-p7_us-west3_knoe-cnpg-0 [Global] -KUBECONTEXT = gke_plenary-truck-485623-p7_us-west3_knoe-cnpg-0 +KUBECONTEXT = gke_plenary-truck-485623-p7_us-west3_knoe-dev-0 APP_CLUSTER_KUBECONTEXT = gke_plenary-truck-485623-p7_us-west3_knoe-dev-0 +DB_CLUSTER_KUBECONTEXT = gke_plenary-truck-485623-p7_us-west3_knoe-cnpg-0 CNPG_ELIGIBLE_NODES = ``` -Missing `init_cluster.app_cluster_kubecontext` → `_cluster_kubecontext("app")` returns `""` → installer falls back to `Global.KUBECONTEXT` (DB cluster) for **both** app and db environments → **Garage deploys to knoe-cnpg-0** (wrong). +### `conf/service/prod.cfg` (unattended deploy — `./deploy.sh`) + +Same cluster context entries are required here too: + +```ini +[Inputs] +init_cluster.app_cluster_kubecontext = gke_plenary-truck-485623-p7_us-west3_knoe-dev-0 +init_cluster.db_cluster_kubecontext = gke_plenary-truck-485623-p7_us-west3_knoe-cnpg-0 +env_setup.APP_CLUSTER_KUBECONTEXT = gke_plenary-truck-485623-p7_us-west3_knoe-dev-0 +env_setup.DB_CLUSTER_KUBECONTEXT = gke_plenary-truck-485623-p7_us-west3_knoe-cnpg-0 + +[Global] +KUBECONTEXT = gke_plenary-truck-485623-p7_us-west3_knoe-dev-0 +APP_CLUSTER_KUBECONTEXT = gke_plenary-truck-485623-p7_us-west3_knoe-dev-0 +DB_CLUSTER_KUBECONTEXT = gke_plenary-truck-485623-p7_us-west3_knoe-cnpg-0 +ARTIFACT_REGISTRY = us-west3-docker.pkg.dev/plenary-truck-485623-p7/knoe-system +SERVICE_NAMESPACE = knoe-system +REGISTRY_NAMESPACE = knoe-system +``` + +**Why these matter:** `Milestone._get_script_env()` (in `knoe/milestone.py`) reads these to set `KUBECONTEXT=app_ctx` for common services and `DB_CLUSTER_KUBECONTEXT=db_ctx` for CNPG ops. Without them, all kubectl calls use the ambient context, which may be the DB cluster. + +Missing `init_cluster.app_cluster_kubecontext` → `_cluster_kubecontext("app")` returns `""` → installer falls back to `Global.KUBECONTEXT` for **both** app and db environments → **Garage deploys to knoe-cnpg-0** (wrong). ### Get current CNPG node names diff --git a/conf/cnpg-placement/ecosystem-0-knoe-db.json b/conf/cnpg-placement/ecosystem-0-knoe-db.json index 9248ab1..44820ad 100644 --- a/conf/cnpg-placement/ecosystem-0-knoe-db.json +++ b/conf/cnpg-placement/ecosystem-0-knoe-db.json @@ -1,15 +1,15 @@ { "assignments": { - "0": "gke-knoe-dev-0-default-pool-503666dd-d39h", - "1": "gke-knoe-dev-0-default-pool-98ee334c-qpf9", - "2": "gke-knoe-dev-0-default-pool-b26cfdbc-r144" + "0": "gke-knoe-dev-0-default-pool-02b40136-m9h9", + "1": "gke-knoe-dev-0-default-pool-6fb9e725-8ldx", + "2": "gke-knoe-dev-0-default-pool-d65ac672-9ncf" }, "cluster_name": "knoe-db", "desired_instances": 3, "eligible_nodes": [ - "gke-knoe-dev-0-default-pool-503666dd-d39h", - "gke-knoe-dev-0-default-pool-98ee334c-qpf9", - "gke-knoe-dev-0-default-pool-b26cfdbc-r144" + "gke-knoe-dev-0-default-pool-02b40136-m9h9", + "gke-knoe-dev-0-default-pool-6fb9e725-8ldx", + "gke-knoe-dev-0-default-pool-d65ac672-9ncf" ], "metadata": { "prior_plan_present": true, @@ -17,7 +17,7 @@ "regenerated": true, "reused": false }, - "plan_hash": "c7f53cb912958603", - "plan_id": "cnpg-placement-c7f53cb912958603", + "plan_hash": "a44825aeb5ed4e13", + "plan_id": "cnpg-placement-a44825aeb5ed4e13", "schema_version": "v1" } diff --git a/conf/cnpg-placement/knoe-system-knoe-db.json b/conf/cnpg-placement/knoe-system-knoe-db.json index 2f9aafe..0d1f250 100644 --- a/conf/cnpg-placement/knoe-system-knoe-db.json +++ b/conf/cnpg-placement/knoe-system-knoe-db.json @@ -1,15 +1,15 @@ { "assignments": { - "0": "gke-knoe-dev-0-default-pool-503666dd-d39h", - "1": "gke-knoe-dev-0-default-pool-98ee334c-qpf9", - "2": "gke-knoe-dev-0-default-pool-b26cfdbc-r144" + "0": "gke-knoe-dev-0-default-pool-02b40136-m9h9", + "1": "gke-knoe-dev-0-default-pool-6fb9e725-8ldx", + "2": "gke-knoe-dev-0-default-pool-d65ac672-9ncf" }, "cluster_name": "knoe-db", "desired_instances": 3, "eligible_nodes": [ - "gke-knoe-dev-0-default-pool-503666dd-d39h", - "gke-knoe-dev-0-default-pool-98ee334c-qpf9", - "gke-knoe-dev-0-default-pool-b26cfdbc-r144" + "gke-knoe-dev-0-default-pool-02b40136-m9h9", + "gke-knoe-dev-0-default-pool-6fb9e725-8ldx", + "gke-knoe-dev-0-default-pool-d65ac672-9ncf" ], "metadata": { "prior_plan_present": true, @@ -17,7 +17,7 @@ "regenerated": false, "reused": true }, - "plan_hash": "c7f53cb912958603", - "plan_id": "cnpg-placement-c7f53cb912958603", + "plan_hash": "a44825aeb5ed4e13", + "plan_id": "cnpg-placement-a44825aeb5ed4e13", "schema_version": "v1" } diff --git a/conf/dev/prole.cfg b/conf/dev/prole.cfg index 745e21f..d64e6df 100644 --- a/conf/dev/prole.cfg +++ b/conf/dev/prole.cfg @@ -1,5 +1,5 @@ ; Prole Master Configuration File -; Generated by install.py on 2026-04-08 23:04:37 +; Generated by install.py on 2026-04-09 16:20:34 ; This file is used as input for Ansible deployment and k8s cluster creation. [User] @@ -56,7 +56,7 @@ init_cluster.cluster_env = dev init_cluster.deployment_target = prole-dev-cluster init_cluster.gitops_enabled = false init_cluster.k3s_server_url = https://myrddin.prole.org:6443 -init_cluster.k3s_token = ${PROLE_SECRET:v1:VehUafohCaBAyMN6:5gV4RGzYRcxlD4_bQg4Z30m9dq6SGuVGFBFF7osArk8RVc1vgNdIW7qtFewIwkskNe9xbSiC2-gvjwKKG5Nv1JJlzhaAR8NNH-_g4PaoGychXb18YL-8oWDFByGMNgokUObDEc3uJ1CNXCkcLtawaruRNwQ_Myz48TT1TQ==} +init_cluster.k3s_token = ${PROLE_SECRET:v1:Am7ac6FTkH1zgfZ8:sWyQ3RTstjMoggXbGEiSTmL1cN5-ECCjyHQG_vbRl66n3pHrm5AC5k3dZlRCudXjXul1Mx5YPl6QmBE5xCHl4Kncxai7G555M6vn5mOikpAkZQyyPcIql2_2ITq6asFhv03xyFc0tNvzbjFVdivqa7rqNSYdSkJrImVHWg==} init_cluster.kerberos_enabled = true init_cluster.mode = k3d init_cluster.start_cluster = true @@ -92,7 +92,7 @@ CLUSTER_NAME = knoe-db CNPG_PLACEMENT_PLAN_FILE = /Users/chrisfu/dev/prole/conf/cnpg-placement/knoe-system-knoe-db.json CNPG_PLACEMENT_PLAN_HASH = 965d2fde4035059f CNPG_PLACEMENT_PLAN_ID = cnpg-placement-965d2fde4035059f -DATABASE_NAMESPACE = knoe-db-18-008-18-009-18-013 +DATABASE_NAMESPACE = knoe-db-18-008-18-009-18-013-18-014 DB_HOST_PORT = 5432 DEPLOYMENT_MODE = k3d DEPLOYMENT_TARGET = prole-dev-cluster @@ -100,7 +100,7 @@ DOCKER_PRELOAD = false KNOE_DB_USER = root OPTIONAL_WORKLOADS_MIN_READY_SCHEDULABLE_NODES = 2 PROLE_K3S_SERVER = https://myrddin.prole.org:6443 -PROLE_K3S_TOKEN = ${PROLE_SECRET:v1:aOx6XJ6T1bwAWJqO:U_Q9bFmBqns_qvVdmIiwlHkz7Y_vRxnlbFij7pHoTfJZP8iPblmtfZjdi-Z1xOAuxnkYy6-0j1SMBLsIJWcJ0G1g5laIfqBB9d1iv9VLKeNmnUrGV-xgiv8-aPn2TM3UtN8hFrfaRKfeiTG01iyKLG6IoNngF2xA8rb7KQ==} +PROLE_K3S_TOKEN = ${PROLE_SECRET:v1:0bTiOIXG8-_kE0cV:Xu7WFcxvmIcsEUumAANi5wspfeuTgJmI5vwXCvmnc4Sygz3S2DBcnMJkQozlCmP4XBHMJhFZ8LsE-nDYXCWX5Ndy9_8WmPGFPBx79Df1AEBSK9qgIEZnPbn3prMu5x7iD2hutWELYDn9gog6LbH4qn_nWF5ZL0jbjyJHzA==} PROLE_OPENTOFU_URL = http://127.0.0.1:8080 SERVICE_NAMESPACE = knoe-system SUPABASE_PV_BASE = /synology/d005 @@ -162,7 +162,7 @@ SUPABASE_ENABLED = False [Initialize Cluster] ENVIRONMENT = dev K3S_SERVER_URL = https://myrddin.prole.org:6443 -K3S_TOKEN = ${PROLE_SECRET:v1:Puz4fzkm528-XoYM:YlPecZrTA2orKSD6NeuquMD3J39L4RGXqOzGzgNzHqmrhLPPlpLaczU3CwoM3omEuF6rjd0NVlf7Ief5HSKPm49S-X70x-vH_9QaiY8bPx3xtfypGC--p-05NQIw_rDWz5mdonzzj41U-zxA9N3r5mVILiBAkI1trkPRLg==} +K3S_TOKEN = ${PROLE_SECRET:v1:ozzcomisjsQYIkSH:Ytp91WR_iP4tJyTAmdH_SRhcKycgzea0zLAgTBNxDsQaBPM-pR_VK3u9wc5QkFzszdAHZGBhVN2HKyqnz-cqDR0WAus88DFbF4zWlgvl6gKEAynaXbdMwAa6vYLUGi8ZE0u1pRiO4KJyiulhBIpfoMReM1Wu6Mj1-20hXw==} [Dev Cluster (k3d)] CLUSTER_ENV = dev @@ -173,7 +173,7 @@ MODE = k3d CLUSTER_ENV = prole-service-cluster DISPLAY_NAME = prole-service-cluster K3S_SERVER_URL = https://myrddin.prole.org:6443 -K3S_TOKEN = ${PROLE_SECRET:v1:7OALwMwbkRA9fIQG:RKg4R1_K6niZRAf9rCngO_Yvss4jMKsvQ1iVyZoNJa9yMgAOlKBZBKRY74qsnpRgfcjyxAbj9V42crkaEUaExR4Rr64cvAk5C3aUrC8kmBNLnnbSzydGxsF8Vgh4SCIg0xo9RyJF5h9M1r-F2Ua2P6aYZTjRYcW-Ro0xiA==} +K3S_TOKEN = ${PROLE_SECRET:v1:-cslrAhG8WhxxJLY:vAx5MGOBcU1NBMjf4U-3y1djMDmwwOjiWkrutxPa4Li5P8RBjCBhEOCEdGP2CiPJJ6UQK-ietX-mu_5nO3yNNjpwDququT4U6lWHPEInSvUNH6ImU-HzPvm_diL2FwJtZm3sY5HgecIy2dwXB_vLnYaA_7VXt9zf0T70rw==} MODE = k3s PIPELINE_URL = http://127.0.0.1:8080 diff --git a/conf/prod/prole.cfg b/conf/prod/prole.cfg index 676805e..2889ed9 100644 --- a/conf/prod/prole.cfg +++ b/conf/prod/prole.cfg @@ -1,5 +1,5 @@ ; Prole Master Configuration File -; Generated by install.py on 2026-04-08 23:05:25 +; Generated by install.py on 2026-04-09 17:30:49 ; This file is used as input for Ansible deployment and k8s cluster creation. [User] @@ -56,7 +56,7 @@ init_cluster.cluster_env = prod init_cluster.deployment_target = prole-prod-cluster init_cluster.gitops_enabled = false init_cluster.k3s_server_url = https://myrddin.prole.org:6443 -init_cluster.k3s_token = ${PROLE_SECRET:v1:EnJkApzkJeOQYGnQ:wDfFl2LAFzu6KmMpucPyIiKVcffzx7J2dPg8HmdN2o-4LSEhZMBjQupAJZuBL0Uy6AiEB1aO4UKjEAy-ahUKU2GbkJBodTdZrjFkUc-KHdn-MINCQ9V4S3NspZLEVNizKYtEuXr9aUpmKdMv-NllrFx-9K8YAfclOPikEw==} +init_cluster.k3s_token = ${PROLE_SECRET:v1:vxIxsLYZSClD_vzj:aSQOAaWbKOIT26PQPCqqIwenTzN4YKcaCa8OihlkDA9cdhbHrPeHtaeT3Pu140a67xKPV45ExSEiJ7jnWSW3lOH0KSO56qgzKxMXzrynKFfQVyJp5OzzQeCl-jrz15AdknLhPeh3eTmOEfLuhNY2Yy9fsBQ-l0gYikw37Q==} init_cluster.kerberos_enabled = true init_cluster.mode = k8s init_cluster.start_cluster = true @@ -67,8 +67,8 @@ init_db_build.run_build = true init_password.cluster_name = ${CLUSTER_NAME} init_password.db_host_port = 5432 init_password.db_namespace = ${DATABASE_NAMESPACE} -init_password.db_password = -init_password.db_password_confirm = +init_password.db_password = ${PROLE_SECRET:v1:2CBUs--mwViqouh9:Q0s7wpwoIfKi22i66X1h3o9eAZetKkVDfO6o2dbU1jBFv8ti} +init_password.db_password_confirm = ${PROLE_SECRET:v1:2CBUs--mwViqouh9:Q0s7wpwoIfKi22i66X1h3o9eAZetKkVDfO6o2dbU1jBFv8ti} init_password.db_username = root init_password.generate_ssh_key = true init_scripts.run_scripts = true @@ -92,23 +92,23 @@ ARTIFACT_REGISTRY = us-west3-docker.pkg.dev/plenary-truck-485623-p7/knoe-system ARTIFACT_REGISTRY_AVAILABLE = true CLUSTER_ENV = prod CLUSTER_NAME = knoe-db -CNPG_ELIGIBLE_NODES = gke-knoe-dev-0-default-pool-503666dd-d39h,gke-knoe-dev-0-default-pool-98ee334c-qpf9,gke-knoe-dev-0-default-pool-b26cfdbc-r144 +CNPG_ELIGIBLE_NODES = gke-knoe-cnpg-0-default-pool-56407930-wlg9,gke-knoe-cnpg-0-default-pool-b142a290-bmr3,gke-knoe-cnpg-0-default-pool-d8e3a5ef-lmpl CNPG_PLACEMENT_PLAN_FILE = /Users/chrisfu/dev/prole/conf/cnpg-placement/knoe-system-knoe-db.json -CNPG_PLACEMENT_PLAN_HASH = c7f53cb912958603 -CNPG_PLACEMENT_PLAN_ID = cnpg-placement-c7f53cb912958603 -CNPG_STAGE1_NODE = gke-knoe-dev-0-default-pool-503666dd-d39h -DATABASE_NAMESPACE = knoe-db-18-008-18-009-18-013 +CNPG_PLACEMENT_PLAN_HASH = a44825aeb5ed4e13 +CNPG_PLACEMENT_PLAN_ID = cnpg-placement-a44825aeb5ed4e13 +CNPG_STAGE1_NODE = gke-knoe-cnpg-0-default-pool-56407930-wlg9 +DATABASE_NAMESPACE = knoe-db-0 DB_CLUSTER_KUBECONTEXT = gke_plenary-truck-485623-p7_us-west3_knoe-cnpg-0 DB_HOST_PORT = 5432 +DB_PASSWORD = ${PROLE_SECRET:v1:vnEWu4ZkBlVtoi-J:Dq_Gluj0Yv_3HdbN_9V53QMf9mPRXotsX04HWsMeY92GBj3O} DEPLOYMENT_MODE = k8s DEPLOYMENT_TARGET = prole-prod-cluster DOCKER_PRELOAD = false KNOE_DB_USER = root -KUBECONFIG = /Users/chrisfu/.kube/config KUBECONTEXT = gke_plenary-truck-485623-p7_us-west3_knoe-dev-0 OPTIONAL_WORKLOADS_MIN_READY_SCHEDULABLE_NODES = 2 PROLE_K3S_SERVER = https://myrddin.prole.org:6443 -PROLE_K3S_TOKEN = ${PROLE_SECRET:v1:on0oK-bUBCGGiibf:RQSubonBNj8V60-zU10D2FYToViJUNt-WyC23-bb6yQrf72WcvkfFMWsiEQA5OxRWvKM4TaaezLF4WFW484p3S9g0Y3oB_9qv7CSc15ZaOOHTUk2DpNOKahV5fItY8bPjmUKWTgKrShSWl6j2C5SFV7fYwmq6TTOuygFPQ==} +PROLE_K3S_TOKEN = ${PROLE_SECRET:v1:wEwOvfE5WeOdbVjW:yjYu4AJqgYo_EnwvYx4fgeSA1coQb8uv8FxE4_NNSIkQadVroCjaEpOyHXcStehLxGl7Bk-elU43yKzYQT_85BkLXp4-9k-LVh_kAfJp2cXlw7q8PczBr3-_MJXeWv1Cw7mCiMuWEfVpKjpCzZ5-URsBsIH3iYz20_6qwg==} SERVICE_NAMESPACE = knoe-system SUPABASE_PV_BASE = /synology/d005 SUPABASE_PV_BASE_DIR = /synology/d005 @@ -170,19 +170,19 @@ DB_BOOT_DISK_SIZE_GB = 50 DB_BOOT_DISK_TYPE = pd-standard ENVIRONMENT = prod K3S_SERVER_URL = https://myrddin.prole.org:6443 -K3S_TOKEN = ${PROLE_SECRET:v1:y0saV3fVIh0T5E6G:ek4jKHO09N2Z5uRO3mwsSgimKpVodGN6zE6d6TtYHMqoFlcJ_AZRUf5dMCqAZ8eE9gydBu_k-ey44CQOAmK8m9KryFU9HDL1ioL_AWGfVJIcG16vgVsaiaVO3h7Eppv__qdCzSqaEgwJqNHC8fTYQtfAXBouk3arsM8_Ew==} +K3S_TOKEN = ${PROLE_SECRET:v1:4gEuXvZk46ZZEE84:n0hMI_eOPmwOWax62B7G1n5M-aEA39gMkEoTl4m8d_zvn-fBhoSsyjebENabZc8cDbGroo1Cunge3tLH9oZ_KHIz_tLQBGvGgG9pLg9X_raTNoH-ezMgbllWWfUH3BHznawYDxun3hDDgqTuOdhBPctVg3nrW8kSa7gJqQ==} [Dev Cluster (k3d)] CLUSTER_ENV = dev DISPLAY_NAME = knoe-dev-cluster -KUBECTL_CONTEXT = gke_plenary-truck-485623-p7_us-west3_knoe-dev-0 +KUBECTL_CONTEXT = gke_plenary-truck-485623-p7_us-west3_knoe-cnpg-0 MODE = k3d [Service Cluster (k3s)] CLUSTER_ENV = prole-service-cluster DISPLAY_NAME = prole-service-cluster K3S_SERVER_URL = https://myrddin.prole.org:6443 -K3S_TOKEN = ${PROLE_SECRET:v1:ts8tzrsAcvLHInVF:xQVZWcdVVVI7tAYqG_048NKLPw25ENxCqtcdiyPb4tGSuyjeqSn9kXoMMR-8TvcBGPpug4dXa3IQZEwIHksQvsWkD-MtbPG5uaehhz7wsLJJ1ILkhd3PGjeyCkFn9zFzLEyPBbJoTRY2I91V_Qck7Vlkkn8zOb3Y9NXuFg==} +K3S_TOKEN = ${PROLE_SECRET:v1:6-xBvCegq8BIAUbO:li8GVpawfjJUs4YhZE1ho5o2S3p2BAE6IH1cMX0SftuGkHOHWrwv_rYmufzVzCVt7RV0PtJevGSGOS_hoaPKA5EPtSE520HcQX0f2UJfAMfpvFzsKziVa0UqIkKYGics2c3JXsydP6eVerWagtJ9ciaESbCoy0YnvVgLKg==} MODE = k3s [GCP] @@ -190,6 +190,10 @@ BILLING_ACCOUNT = 01193C-25783B-3211AD BILLING_PROJECT = plenary-truck-485623-p7 ORG_ID = 584001916389 PROJECT_ID = plenary-truck-485623-p7 +billing_account = 01193C-25783B-3211AD +billing_project = plenary-truck-485623-p7 +org_id = 584001916389 +project_id = plenary-truck-485623-p7 [Prod Cluster (k8s)] ARTIFACTS_DIR = /Users/chrisfu/dev/prole/data/staging @@ -205,7 +209,7 @@ Configure routing, TLS, and frontdoor domain mappings Prepare migration artifacts from source environment Generate OpenTofu plan and apply production pipeline MODE = k8s -OPENTOFU_VARS = {"admin_user":"knoe_admin","app_database":"knoey","app_user":"knoey_app","backup_bucket":"knoe-0-backups","backup_provider":"gcs","backup_retention_days":14,"bootstrap_admin_email":"admin@knoey.com","cluster_name":"knoe-cnpg-0","cnpg_cluster_name":"knoe-db","db_instances":3,"db_pgdata_size_gi":100,"db_pgdata_storage_class":"premium-rwo","db_storage_class":"premium-rwo","db_storage_size_gi":100,"db_wal_size_gi":25,"db_wal_storage_class":"premium-rwo","ecosystem_id":0,"environment":"production","frontdoor_host":"knoey.com","gcp_project_id":"plenary-truck-485623-p7","gcp_region":"\u2713 us-west3","google_hosted_domain":"","meta_database":"knoe_meta","migration_continuous_until_cutover":false,"migration_mode":"snapshot-restore","migration_source_database":"knoey","migration_source_environment":"prole.org","migration_source_host":"knoe-local-db.prole.org","migration_source_password_ref":"secretref://local-source-db-password","migration_source_port":5432,"migration_source_user":"replication_user","namespace":"ecosystem-0","oidc_client_id_ref":"secretref://google-oidc-client-id","oidc_client_secret_ref":"secretref://google-oidc-client-secret","oidc_issuer":"https://accounts.google.com","platform_domain":"knoe.dev","postgres_version":"16","project_id":"plenary-truck-485623-p7","region":"\u2713 us-west3","tls_mode":"managed","wal_bucket":"knoe-0-wal","workload_identity_sa":"cnpg-backup"} +OPENTOFU_VARS = {"admin_user":"knoe_admin","app_database":"knoey","app_user":"knoey_app","backup_bucket":"knoe-0-backups","backup_provider":"gcs","backup_retention_days":14,"bootstrap_admin_email":"admin@knoey.com","cluster_name":"knoe-prod","cnpg_cluster_name":"knoe-db","db_instances":3,"db_pgdata_size_gi":100,"db_pgdata_storage_class":"premium-rwo","db_storage_class":"premium-rwo","db_storage_size_gi":100,"db_wal_size_gi":25,"db_wal_storage_class":"premium-rwo","ecosystem_id":0,"environment":"production","frontdoor_host":"knoey.com","gcp_project_id":"plenary-truck-485623-p7","gcp_region":"us-west3","google_hosted_domain":"","meta_database":"knoe_meta","migration_continuous_until_cutover":false,"migration_mode":"snapshot-restore","migration_source_database":"knoey","migration_source_environment":"prole.org","migration_source_host":"knoe-local-db.prole.org","migration_source_password_ref":"secretref://local-source-db-password","migration_source_port":5432,"migration_source_user":"replication_user","namespace":"ecosystem-0","oidc_client_id_ref":"secretref://google-oidc-client-id","oidc_client_secret_ref":"secretref://google-oidc-client-secret","oidc_issuer":"https://accounts.google.com","platform_domain":"knoe.dev","postgres_version":"16","project_id":"plenary-truck-485623-p7","region":"us-west3","tls_mode":"managed","wal_bucket":"knoe-0-wal","workload_identity_sa":"cnpg-backup"} PRODUCTION_CONFIG_YAML = kind: KnoeProductionConfig metadata: ecosystemId: 0 @@ -215,8 +219,8 @@ spec: cloud: provider: gcp projectId: plenary-truck-485623-p7 - region: "\u2713 us-west3" - clusterName: knoe-cnpg-0 + region: us-west3 + clusterName: knoe-prod vpcMode: managed vpcName: null subnetName: null diff --git a/conf/service/prod.cfg b/conf/service/prod.cfg index a56b151..42f3945 100644 --- a/conf/service/prod.cfg +++ b/conf/service/prod.cfg @@ -1,8 +1,211 @@ -# Prole GCP deployment entrypoint -# Used by deploy.sh for unattended production pipeline runs +; Prole Master Configuration File +; Generated by install.py on 2026-04-09 23:19:13 +; This file is used as input for Ansible deployment and k8s cluster creation. -[Global] -CLUSTER_ENV = prod +[User] +; User-editable values; derived values below reference these by default. +; No user values captured yet for this section. [Inputs] -# Values will be populated/updated by deploy_pipeline.py +; Screen-scoped inputs used for unattended replays (-S) +build.deploy_env = Dev +build.run_build = false +dependencies.ansible.install = true +dependencies.auto_install_missing = true +dependencies.brew.install = true +dependencies.docker.install = true +dependencies.k3d.install = true +dependencies.kubectl.install = true +dependencies.kubectx.install = true +dependencies.opentofu.install = true +dependencies.python.install = true +dependencies.verify_all = false +disk_selection.disk_type = local +disk_selection.local_path = /Users/chrisfu +disk_selection.removable_mount = +env_setup.APP_CLUSTER_KUBECONTEXT = gke_plenary-truck-485623-p7_us-west3_knoe-dev-0 +env_setup.APP_CLUSTER_MODE = standard +env_setup.APP_CLUSTER_NAME = knoe-dev-0 +env_setup.CLUSTER_NAME = ${CLUSTER_NAME} +env_setup.DATABASE_NAMESPACE = ${DATABASE_NAMESPACE} +env_setup.DB_CLUSTER_KUBECONTEXT = gke_plenary-truck-485623-p7_us-west3_knoe-cnpg-0 +env_setup.DB_CLUSTER_MODE = standard +env_setup.DB_CLUSTER_NAME = knoe-cnpg-0 +env_setup.PROLE_CONF = /Users/chrisfu/dev/prole/conf +env_setup.PROLE_DATA = /Users/chrisfu/dev/prole/data +env_setup.PROLE_HOME = /Users/chrisfu/dev/prole +env_setup.PROLE_LOGS = /Users/chrisfu/dev/prole/logs +env_setup.PROLE_SERVICE = /Users/chrisfu/dev/prole/etc +init_cluster.app_cluster_kubecontext = gke_plenary-truck-485623-p7_us-west3_knoe-dev-0 +init_cluster.app_cluster_machine_type = e2-small +init_cluster.app_cluster_mode = standard +init_cluster.app_cluster_name = knoe-dev-0 +init_cluster.app_cluster_node_count = 3 +init_cluster.app_cluster_region = us-west3 +init_cluster.at_rest_encryption_enabled = true +init_cluster.cluster_env = prod +init_cluster.db_boot_disk_size_gb = 50 +init_cluster.db_boot_disk_type = pd-standard +init_cluster.db_cluster_kubecontext = gke_plenary-truck-485623-p7_us-west3_knoe-cnpg-0 +init_cluster.db_cluster_machine_type = e2-standard-2 +init_cluster.db_cluster_mode = standard +init_cluster.db_cluster_name = knoe-cnpg-0 +init_cluster.db_cluster_node_count = 3 +init_cluster.db_cluster_region = us-west3 +init_cluster.db_cluster_zones = +init_cluster.gitops_enabled = false +init_cluster.k3s_server_url = +init_cluster.k3s_token = +init_cluster.kerberos_enabled = false +init_cluster.project_id = plenary-truck-485623-p7 +init_cluster.start_cluster = true +init_cluster.supabase_analytics_enabled = true +init_cluster.supabase_auth_enabled = true +init_cluster.supabase_enabled = false +init_cluster.supabase_meta_enabled = true +init_cluster.supabase_realtime_enabled = true +init_cluster.supabase_studio_enabled = false +init_cluster.supabase_studio_url = db.knoe.org +init_cnpg_deploy.force_rollout = false +init_cnpg_deploy.run_deploy = true +init_db_build.run_build = true +init_password.app_cluster_name = knoe-dev-0 +init_password.cluster_name = ${CLUSTER_NAME} +init_password.db_cluster_name = knoe-cnpg-0 +init_password.db_host_port = 5432 +init_password.db_namespace = ${DATABASE_NAMESPACE} +init_password.db_password = ${PROLE_SECRET:v1:aqtfPnE_i0CuDU42:KvAQK1cUXKkFPta7NGyaWJVX1Ufw6xH432ZLzg==} +init_password.db_password_confirm = ${PROLE_SECRET:v1:aqtfPnE_i0CuDU42:KvAQK1cUXKkFPta7NGyaWJVX1Ufw6xH432ZLzg==} +init_password.db_username = chrisfu +init_password.generate_ssh_key = true +init_scripts.run_scripts = true +kerberos_config.enabled = True +kerberos_config.init_authority = false +kerberos_config.kdc = 10.0.0.196 +kerberos_config.password = +kerberos_config.realm = +kerberos_config.test_connection = false +kerberos_config.user = administrator +network_scan.run = true +ollama_config.model = +ollama_config.server_host = +ollama_config.server_port = 11434 + +[Global] +; Variables used by name in more than one place or assumed global scope +APP_CLUSTER_KUBECONTEXT = gke_plenary-truck-485623-p7_us-west3_knoe-dev-0 +ARGOCD_NAMESPACE = argocd +ARTIFACT_REGISTRY = us-west3-docker.pkg.dev/plenary-truck-485623-p7/knoe-system +CLUSTER_ENV = prod +CLUSTER_NAME = knoe-db +CLUSTER_SIZE_PROFILE = small +CNPG_ELIGIBLE_NODES = gke-knoe-cnpg-0-default-pool-56407930-wlg9,gke-knoe-cnpg-0-default-pool-b142a290-bmr3,gke-knoe-cnpg-0-default-pool-d8e3a5ef-lmpl +CNPG_INSTANCES = 3 +CNPG_SIZE_PROFILE = small +CNPG_STAGE1_NODE = gke-knoe-cnpg-0-default-pool-56407930-wlg9 +DATABASE_NAMESPACE = knoe-db-0 +DB_CLUSTER_KUBECONTEXT = gke_plenary-truck-485623-p7_us-west3_knoe-cnpg-0 +DB_HOST_PORT = 5432 +DB_PASSWORD = ${PROLE_SECRET:v1:BQs_xFgVxqbFwmtj:bXF155fo4KH1ZSwz9NvhgqTwQ7_oRkHewFsm5w==} +DEPLOYMENT_MODE = k8s +DEPLOYMENT_TARGET = prole-prod-cluster +DOCKER_IMPORT_DIR = +DOCKER_PRELOAD = false +K3S_SERVER = +K3S_TOKEN = +KNOE_DB_USER = chrisfu +KUBECONTEXT = gke_plenary-truck-485623-p7_us-west3_knoe-dev-0 +OPENTOFU_URL = http://127.0.0.1:8080 +OPTIONAL_WORKLOADS_MIN_READY_SCHEDULABLE_NODES = 2 +PROLE_HOME = /Users/chrisfu/dev/prole +REGISTRY_NAMESPACE = knoe-system +SERVICE_NAMESPACE = knoe-system + +[Welcome] +; No configuration values captured yet for this section. + +[Dependencies] +STATUS = All installed + +[Network] +KDC_AUTO_DETECTED = 10.0.0.196 +KERBEROS_AUTO_ENABLED = True + +[Port Forwards] +PORT_FORWARD_K3D_MAPPING_1 = id=argocd;namespace=argocd;target=svc/argocd-server;address=0.0.0.0;hostPort=8081;servicePort=80;protocol=TCP;description=ArgoCD +PORT_FORWARD_K3D_MAPPING_2 = id=garage;namespace=knoe-system;target=svc/garage;address=0.0.0.0;hostPort=3900;servicePort=3900;protocol=TCP;description=Garage S3 +PORT_FORWARD_K3D_MAPPING_3 = id=openbao;namespace=knoe-system;target=svc/openbao;address=0.0.0.0;hostPort=8200;servicePort=8200;protocol=TCP;description=OpenBao +PORT_FORWARD_K3D_MAPPING_4 = id=opentofu;namespace=knoe-system;target=svc/opentofu;address=0.0.0.0;hostPort=8080;servicePort=8080;protocol=TCP;description=OpenTofu +PORT_FORWARD_K3D_MAPPING_5 = id=dashboard;namespace=kubernetes-dashboard;target=svc/kubernetes-dashboard-kong-proxy;address=127.0.0.1;hostPort=8443;servicePort=443;protocol=TCP;description=Kubernetes Dashboard +PORT_FORWARD_K3D_MAPPING_6 = id=postgres;namespace=${DATABASE_NAMESPACE};target=svc/knoe-db-rw;address=0.0.0.0;hostPort=5432;servicePort=5432;protocol=TCP;description=PostgreSQL (primary) +PORT_FORWARD_K3D_MAPPING_7 = id=prometheus;namespace=monitoring;target=svc/kps-kube-prometheus-stack-prometheus;address=127.0.0.1;hostPort=9090;servicePort=9090;protocol=TCP;description=Prometheus UI +PORT_FORWARD_K3D_MAPPING_8 = id=grafana;namespace=monitoring;target=svc/kps-grafana;address=0.0.0.0;hostPort=3000;servicePort=80;protocol=TCP;description=Grafana UI + +[System Environment] +PROLE_CONF = /Users/chrisfu/dev/prole/conf +PROLE_DATA = /Users/chrisfu/dev/prole/data +PROLE_HOME = /Users/chrisfu/dev/prole +PROLE_LOGS = /Users/chrisfu/dev/prole/logs +PROLE_SERVICE = /Users/chrisfu/dev/prole/etc + +[Monitoring] +; No configuration values captured yet for this section. + +[Kerberos Authentication] +; No configuration values captured yet for this section. + +[Ollama] +; No configuration values captured yet for this section. + +[Optional Features] +AT_REST_ENCRYPTION_ENABLED = true +GITOPS_ENABLED = false +KERBEROS_ENABLED = false +SUPABASE_ENABLED = false + +[GitOps] + +[Database Creation] +APP_CLUSTER_NAME = knoe-dev-0 +DB_CLUSTER_NAME = knoe-cnpg-0 +DB_USER = chrisfu + +[Initialize Cluster] +; No configuration values captured yet for this section. + +[Dev Cluster (k3d)] +CLUSTER_ENV = k3d-knoe-dev-cluster +DISPLAY_NAME = knoe-dev-cluster +KUBECTL_CONTEXT = prod +MODE = k3d + +[Service Cluster (k3s)] +CLUSTER_ENV = prole-service-cluster +DISPLAY_NAME = prole-service-cluster +K3S_SERVER_URL = +K3S_TOKEN = +MODE = k3s +PIPELINE_URL = http://127.0.0.1:8080 + +[GCP] +; No configuration values captured yet for this section. + +[Prod Cluster (k8s)] +ARTIFACTS_DIR = +CLUSTER_ENV = prole-prod-cluster +DISPLAY_NAME = prole-prod-cluster +MODE = k8s +PIPELINE_URL = http://127.0.0.1:8080 + +[Docker Build] +; No configuration values captured yet for this section. + +[Initialization Scripts] +; No configuration values captured yet for this section. + +[Deployment] +MODE = k8s +TARGET = prole-prod-cluster + +[Install] +STATUS = Failed diff --git a/deploy/gcp/gke/knoe-db.yaml b/deploy/gcp/gke/knoe-db.yaml index 27453c0..6080aec 100644 --- a/deploy/gcp/gke/knoe-db.yaml +++ b/deploy/gcp/gke/knoe-db.yaml @@ -26,6 +26,9 @@ spec: operator: "Equal" value: "true" effect: "NoSchedule" + # nodeSelector removed: knoe-cnpg-0 is a dedicated DB cluster — all nodes are + # available to CNPG. A workload label selector here causes scheduling failures + # when CNPG v1.28 translates it into requiredDuringScheduling nodeAffinity. postgresql: parameters: @@ -108,11 +111,11 @@ spec: resources: requests: - cpu: "250m" # halved for e2-small quota-increase window; restore to 500m after increase - memory: "256Mi" # halved for e2-small quota-increase window; restore to 512Mi after increase + cpu: "100m" + memory: "128Mi" limits: - cpu: "1" - memory: "1Gi" + cpu: "500m" + memory: "512Mi" enableSuperuserAccess: true diff --git a/etc/init_common_services.sh b/etc/init_common_services.sh index a3072b5..5592857 100755 --- a/etc/init_common_services.sh +++ b/etc/init_common_services.sh @@ -25,7 +25,8 @@ usage() { cat < -Deploys common infrastructure services (Registry, OpenTofu, Garage, OpenBao, Kong, Cert-Manager) +Deploys common infrastructure services (Registry, Garage, OpenBao, Kong, Cert-Manager; +OpenTofu on non-k8s modes) into the given Kubernetes namespace. Use -k to include the Kerberos/KDC service. EOF } @@ -540,17 +541,19 @@ collect_other_namespaces() { migrate_common_services() { local old_ns - for old_ns in $(collect_other_namespaces "$NS" "$OPENTOFU_NAME" deployment service); do - echo "Found OpenTofu in namespace '$old_ns'; removing before deploy to '$NS' ..." - if [ -x "$SCRIPT_DIR/init_opentofu.sh" ]; then - "$SCRIPT_DIR/init_opentofu.sh" -n "$old_ns" stop || true - else - kubectl delete -n "$old_ns" deploy "$OPENTOFU_NAME" --ignore-not-found >/dev/null 2>&1 || true - kubectl delete -n "$old_ns" svc "$OPENTOFU_NAME" --ignore-not-found >/dev/null 2>&1 || true - fi - kubectl delete -n "$old_ns" configmap "$OPENTOFU_CONFIGMAP" --ignore-not-found >/dev/null 2>&1 || true - kubectl delete -n "$old_ns" secret "$OPENTOFU_SECRET" --ignore-not-found >/dev/null 2>&1 || true - done + if [ "${PROLE_MODE:-}" != "k8s" ]; then + for old_ns in $(collect_other_namespaces "$NS" "$OPENTOFU_NAME" deployment service); do + echo "Found OpenTofu in namespace '$old_ns'; removing before deploy to '$NS' ..." + if [ -x "$SCRIPT_DIR/init_opentofu.sh" ]; then + "$SCRIPT_DIR/init_opentofu.sh" -n "$old_ns" stop || true + else + kubectl delete -n "$old_ns" deploy "$OPENTOFU_NAME" --ignore-not-found >/dev/null 2>&1 || true + kubectl delete -n "$old_ns" svc "$OPENTOFU_NAME" --ignore-not-found >/dev/null 2>&1 || true + fi + kubectl delete -n "$old_ns" configmap "$OPENTOFU_CONFIGMAP" --ignore-not-found >/dev/null 2>&1 || true + kubectl delete -n "$old_ns" secret "$OPENTOFU_SECRET" --ignore-not-found >/dev/null 2>&1 || true + done + fi for old_ns in $(collect_other_namespaces "$REGISTRY_NS" "$REGISTRY_NAME" deployment service); do echo "Found Registry ($REGISTRY_NAME) in namespace '$old_ns'; removing before deploy to '$REGISTRY_NS' ..." @@ -666,7 +669,7 @@ esac # 1. Registry – no dependencies; other services pull images from it # 2. OpenBao – secrets vault; needed by downstream services # 3. Garage – object storage -# 4. OpenTofu – IaC engine; depends on registry + secrets +# 4. OpenTofu – IaC engine; depends on registry + secrets (non-k8s) # --------------------------------------------------------------------------- if [ "${PROLE_MODE:-}" = "k8s" ]; then @@ -706,7 +709,9 @@ else echo "WARN: init_garage_store.sh not found; garage deploy skipped." fi -if [ -x "$SCRIPT_DIR/init_opentofu.sh" ]; then +if [ "${PROLE_MODE:-}" = "k8s" ]; then + echo "[INFO] k8s mode: skipping OpenTofu deploy." +elif [ -x "$SCRIPT_DIR/init_opentofu.sh" ]; then ROLLOUT_TIMEOUT="${ROLLOUT_TIMEOUT:-300s}" "$SCRIPT_DIR/init_opentofu.sh" -n "$NS" "$ACTION" || rc=$? else echo "WARN: init_opentofu.sh not found; skipping OpenTofu." diff --git a/etc/init_service_layer.sh b/etc/init_service_layer.sh index 1adc36b..a355bda 100755 --- a/etc/init_service_layer.sh +++ b/etc/init_service_layer.sh @@ -4,7 +4,7 @@ set -euo pipefail # init_service_layer.sh # Purpose: -# - Deploy the Prole service layer (OpenTofu, Garage, OpenBao, Kong; Kerberos optional) +# - Deploy the Prole service layer (Garage, OpenBao, Kong; OpenTofu in non-k8s modes) # - Keep service-layer resources grouped in SERVICE_NAMESPACE # - Migrate service layer to a new namespace @@ -169,6 +169,11 @@ deploy_service_layer() { local action="$1" local ns="$2" local rc=0 + local manage_opentofu=1 + + if [[ "${PROLE_MODE:-}" == "k8s" ]]; then + manage_opentofu=0 + fi ensure_namespace "$ns" label_namespace "$ns" @@ -182,13 +187,15 @@ deploy_service_layer() { *) argocd_action="update" ;; esac - case "$action" in - start|initialize|update|reload) opentofu_action="update" ;; - restart) opentofu_action="restart" ;; - stop) opentofu_action="stop" ;; - status) opentofu_action="status" ;; - *) opentofu_action="update" ;; - esac + if [[ "$manage_opentofu" == "1" ]]; then + case "$action" in + start|initialize|update|reload) opentofu_action="update" ;; + restart) opentofu_action="restart" ;; + stop) opentofu_action="stop" ;; + status) opentofu_action="status" ;; + *) opentofu_action="update" ;; + esac + fi case "$action" in start|initialize|update|reload) openbao_action="update" ;; @@ -247,7 +254,7 @@ deploy_service_layer() { # 2. OpenBao – secrets vault; needed by downstream services # 3. Garage – object storage # 4. Kong – API gateway - # 5. OpenTofu – IaC engine; depends on registry + secrets (last) + # 5. OpenTofu – IaC engine; depends on registry + secrets (non-k8s, last) # ------------------------------------------------------------------------- if [[ -x "$SCRIPT_DIR/init_registry.sh" ]]; then @@ -275,8 +282,12 @@ deploy_service_layer() { KONG_NAMESPACE="$ns" SERVICE_NAMESPACE="$ns" \ "$SCRIPT_DIR/init_kong.sh" -n "$ns" "$kong_action" || rc=$? - OPENTOFU_NAMESPACE="$ns" OPENTOFU_SECRET_NAMESPACE="${NAMESPACE:-$ns}" OPENTOFU_OPENBAO_NAMESPACE="$ns" SERVICE_NAMESPACE="$ns" \ - "$SCRIPT_DIR/init_opentofu.sh" -n "$ns" "$opentofu_action" || rc=$? + if [[ "$manage_opentofu" == "1" ]]; then + OPENTOFU_NAMESPACE="$ns" OPENTOFU_SECRET_NAMESPACE="${NAMESPACE:-$ns}" OPENTOFU_OPENBAO_NAMESPACE="$ns" SERVICE_NAMESPACE="$ns" \ + "$SCRIPT_DIR/init_opentofu.sh" -n "$ns" "$opentofu_action" || rc=$? + else + log "[INFO] k8s mode: skipping OpenTofu deploy." + fi if [[ "$ENABLE_KERBEROS" == "1" ]]; then # KDC is embedded in `knoe-auth` by default. Only deploy standalone KDC when requested. diff --git a/etc/prole_cfg.sh b/etc/prole_cfg.sh index c5a9c57..6711e23 100644 --- a/etc/prole_cfg.sh +++ b/etc/prole_cfg.sh @@ -533,8 +533,13 @@ if [[ -n "$_prole_cfg_file" ]]; then _cfg_ctx=$(_prole_cfg_extract_key_in_files "KUBECONTEXT" "${_prole_cfg_files[@]}") fi if [[ -n "$_cfg_ctx" ]]; then - export KUBE_CONTEXT_NAME="$_cfg_ctx" - export KUBECONTEXT="$_cfg_ctx" + # Respect explicitly provided runtime context (e.g., installer per-role env) + # and only fall back to prole.cfg when no runtime context is set. + if [[ -z "${KUBE_CONTEXT_NAME:-}" && -z "${KUBECTL_CONTEXT:-}" && -z "${KUBECONTEXT:-}" ]]; then + export KUBE_CONTEXT_NAME="$_cfg_ctx" + export KUBECTL_CONTEXT="$_cfg_ctx" + export KUBECONTEXT="$_cfg_ctx" + fi fi unset _cfg_ns _cfg_sns _cfg_dm _cfg_sh _cfg_sbh _cfg_ctx _prole_cfg_files _prole_cfg_f fi diff --git a/etc/status_common_services.sh b/etc/status_common_services.sh index a194d27..9a94a01 100755 --- a/etc/status_common_services.sh +++ b/etc/status_common_services.sh @@ -112,6 +112,10 @@ REGISTRY_CHECK=0 if [[ "$MODE" == "k3s" || "$MODE" == "k3d" ]]; then REGISTRY_CHECK=1 fi +OPENTOFU_CHECK=1 +if [[ "$MODE" == "k8s" ]]; then + OPENTOFU_CHECK=0 +fi timestamp=$(date "+%Y-%m-%d %H:%M:%S") ctx=$(kubectl config current-context 2>/dev/null || true) @@ -143,7 +147,13 @@ if [[ "$REGISTRY_CHECK" -eq 1 && "$MODE" != "k3d" ]]; then elif [[ "$REGISTRY_CHECK" -eq 1 && "$MODE" == "k3d" ]]; then run_cmd k3d registry list fi -run_cmd kubectl -n "$NS" get svc opentofu garage openbao +if [[ "$OPENTOFU_CHECK" -eq 1 ]]; then + run_cmd kubectl -n "$NS" get svc opentofu garage openbao +else + run_cmd kubectl -n "$NS" get svc garage openbao + echo "[INFO] k8s mode: skipping OpenTofu service check." + echo "" +fi if [[ "$ENABLE_KERBEROS" == "1" ]]; then run_cmd kubectl -n "$NS" get svc auth fi @@ -154,7 +164,12 @@ echo "== Workloads ==" if [[ "$REGISTRY_CHECK" -eq 1 && "$MODE" != "k3d" ]]; then run_cmd kubectl -n "$REGISTRY_NS" get deploy registry fi -run_cmd kubectl -n "$NS" get deploy opentofu +if [[ "$OPENTOFU_CHECK" -eq 1 ]]; then + run_cmd kubectl -n "$NS" get deploy opentofu +else + echo "[INFO] k8s mode: skipping OpenTofu workload check." + echo "" +fi if kubectl -n "$NS" get statefulset openbao >/dev/null 2>&1; then run_cmd kubectl -n "$NS" get statefulset openbao else @@ -171,7 +186,11 @@ echo "== Pods ==" if [[ "$REGISTRY_CHECK" -eq 1 && "$MODE" != "k3d" ]]; then run_cmd kubectl -n "$REGISTRY_NS" get pods | grep -Ei "registry" || true fi -run_cmd kubectl -n "$NS" get pods | grep -Ei "opentofu|garage|openbao" || true +if [[ "$OPENTOFU_CHECK" -eq 1 ]]; then + run_cmd kubectl -n "$NS" get pods | grep -Ei "opentofu|garage|openbao" || true +else + run_cmd kubectl -n "$NS" get pods | grep -Ei "garage|openbao" || true +fi if [[ "$ENABLE_KERBEROS" == "1" ]]; then run_cmd kubectl -n "$NS" get pods | grep -Ei "auth" || true fi @@ -296,7 +315,9 @@ if [[ "$REGISTRY_CHECK" -eq 1 ]]; then fi fi -check_resource svc opentofu "$NS" +if [[ "$OPENTOFU_CHECK" -eq 1 ]]; then + check_resource svc opentofu "$NS" +fi check_resource svc garage "$NS" check_resource svc openbao "$NS" if [[ "$ENABLE_KERBEROS" == "1" ]]; then @@ -306,7 +327,9 @@ check_resource svc "$KONG_NAME" "$KONG_NS" check_resource svc cert-manager "$CERTMGR_NS" check_resource svc cert-manager-webhook "$CERTMGR_NS" -check_resource deploy opentofu "$NS" +if [[ "$OPENTOFU_CHECK" -eq 1 ]]; then + check_resource deploy opentofu "$NS" +fi if kubectl -n "$NS" get statefulset openbao >/dev/null 2>&1; then check_resource statefulset openbao "$NS" else @@ -321,9 +344,12 @@ check_resource deploy cert-manager "$CERTMGR_NS" check_resource deploy cert-manager-cainjector "$CERTMGR_NS" check_resource deploy cert-manager-webhook "$CERTMGR_NS" -pod_filter="opentofu|garage|openbao|kong|cert-manager" +pod_filter="garage|openbao|kong|cert-manager" if [[ "$ENABLE_KERBEROS" == "1" ]]; then - pod_filter="opentofu|garage|openbao|auth|kong|cert-manager" + pod_filter="garage|openbao|auth|kong|cert-manager" +fi +if [[ "$OPENTOFU_CHECK" -eq 1 ]]; then + pod_filter="opentofu|${pod_filter}" fi analyze_pods "$NS" "$pod_filter" analyze_pods "$KONG_NS" "kong" @@ -391,7 +417,11 @@ _workloads_for_comp() { echo "deployment openbao $_ns" fi ;; - opentofu) echo "deployment opentofu $_ns" ;; + opentofu) + if [[ "$OPENTOFU_CHECK" -eq 1 ]]; then + echo "deployment opentofu $_ns" + fi + ;; garage) echo "statefulset garage $_ns" ;; auth) echo "deployment auth $_ns" ;; kong) echo "deployment $KONG_NAME $KONG_NS" ;; @@ -427,6 +457,7 @@ _blocked_comps() { for _bc in registry openbao garage opentofu auth kong certmgr; do [[ "$_bc" == "auth" && "$ENABLE_KERBEROS" != "1" ]] && continue [[ "$_bc" == "registry" && ( "$REGISTRY_CHECK" -ne 1 || "$MODE" == "k3d" ) ]] && continue + [[ "$_bc" == "opentofu" && "$OPENTOFU_CHECK" -ne 1 ]] && continue [[ ${BLOCKED_COUNT["$_bc"]:-0} -gt 0 ]] && echo "$_bc" done } diff --git a/infrastructure/inventory/group_vars/all/vault_db_master.yml b/infrastructure/inventory/group_vars/all/vault_db_master.yml new file mode 100644 index 0000000..1afed76 --- /dev/null +++ b/infrastructure/inventory/group_vars/all/vault_db_master.yml @@ -0,0 +1,7 @@ +vault_prole_db_master_password: !vault | + $ANSIBLE_VAULT;1.1;AES256 + 34636665663663616661623235656234636236363764383766353934306466363238326165616632 + 3937613765353935636661333032306333626234306564370a373630393535643531623764383661 + 63616233643631666539346534666332343463633435653736336666643066383062303566326235 + 3138626139363635390a376335666665366261623761383764393833396533396662643439666339 + 34333464623839666532313463363062366533393033626130613132346639633030 diff --git a/knoe/core/actions.py b/knoe/core/actions.py index 76ae3f7..5904845 100644 --- a/knoe/core/actions.py +++ b/knoe/core/actions.py @@ -26,6 +26,7 @@ from knoe.config import ( _write_k3s_kubeconfig, _encrypt_cfg_secret, _merge_kubeconfig, + _try_read_ansible_vault_value, ) from knoe.core.build_context import copy_build_context_dir from knoe.core.cnpg_placement import ( @@ -122,7 +123,7 @@ DEFAULT_APP_CLUSTER_NODE_COUNT = 3 DEFAULT_DB_CLUSTER_NAME = "knoe-cnpg-0" DEFAULT_DB_CLUSTER_MODE = "standard" DEFAULT_DB_CLUSTER_NODE_COUNT = 3 -DEFAULT_DB_CLUSTER_MACHINE_TYPE = "e2-small" +DEFAULT_DB_CLUSTER_MACHINE_TYPE = "e2-standard-2" DEFAULT_DB_BOOT_DISK_TYPE = "pd-standard" DEFAULT_DB_BOOT_DISK_SIZE_GB = 50 @@ -290,7 +291,14 @@ class KnoeInstaller: ).strip() if ns.startswith("${") and ns.endswith("}"): ns = "" - return ns or "default" + if ns: + return ns + try: + if self._deployment_mode() == "k8s": + return "knoe-db-0" + except Exception: + pass + return "default" def _service_namespace(self) -> str: ns = ( @@ -313,7 +321,7 @@ class KnoeInstaller: except Exception: pass - return "default" + return "knoe-system" # ------------------------------------------ k3s connection helpers def _read_k3s_cfg_values( @@ -1708,6 +1716,37 @@ class KnoeInstaller: if reg_internal: env["LOCAL_REGISTRY_INTERNAL"] = reg_internal + # Artifact Registry is required by CNPG manifests in k8s mode. + artifact_registry = str(env.get("ARTIFACT_REGISTRY") or "").strip().strip('"') + if not artifact_registry: + artifact_registry = ( + (self.prole_cfg_data.get("Global", {}) or {}) + .get("ARTIFACT_REGISTRY", "") + .strip() + .strip('"') + ) + if not artifact_registry: + gcp_cfg = self.prole_cfg_data.get("GCP", {}) or {} + project_id = ( + (self._get_input("init_cluster.project_id", "") or "").strip() + or str(gcp_cfg.get("project_id") or "").strip().strip('"') + or str(gcp_cfg.get("PROJECT_ID") or "").strip().strip('"') + or (os.environ.get("GCP_PROJECT") or "").strip() + ) + region = ( + (self._get_input("init_cluster.db_cluster_region", "") or "").strip() + or (self._get_input("init_cluster.app_cluster_region", "") or "").strip() + or str(gcp_cfg.get("region") or "").strip().strip('"') + or str(gcp_cfg.get("REGION") or "").strip().strip('"') + or str(gcp_cfg.get("location") or "").strip().strip('"') + or (os.environ.get("GCP_REGION") or "").strip() + ) + repo = (service_ns or "knoe-system").strip() or "knoe-system" + if project_id and region: + artifact_registry = f"{region}-docker.pkg.dev/{project_id}/{repo}" + if artifact_registry: + env["ARTIFACT_REGISTRY"] = artifact_registry.rstrip("/") + try: placement_plan, placement_path = self._resolve_cnpg_placement_plan(namespace) eligible_nodes = placement_plan.get("eligible_nodes") or [] @@ -1805,12 +1844,70 @@ class KnoeInstaller: if role == "app" else "env_setup.DB_CLUSTER_KUBECONTEXT" ) - context = (self._get_input(key, "") or self._get_input(fallback_key, "")).strip() + context = _safe_str(self._get_input(key, "") or self._get_input(fallback_key, "")) if context: return context + + glob = self.prole_cfg_data.get("Global", {}) or {} + global_key = "APP_CLUSTER_KUBECONTEXT" if role == "app" else "DB_CLUSTER_KUBECONTEXT" + context = _safe_str(glob.get(global_key, "")) + if context: + return context + + selected_context = _safe_str( + self._get_input("init_cluster.selected_kubectx", "") or glob.get("KUBECONTEXT", "") + ) + if selected_context and not selected_context.startswith("gke_"): + return selected_context + if role == "app": - return (self._get_input("init_cluster.selected_kubectx", "") or "").strip() - return "" + cluster_name = self._app_cluster_name() + region = _safe_str( + self._get_input("init_cluster.app_cluster_region", "") + or self._get_input("env_setup.APP_CLUSTER_REGION", "") + or (self.gcp_cfg if isinstance(getattr(self, "gcp_cfg", None), dict) else {}).get("region") + or (self.gcp_cfg if isinstance(getattr(self, "gcp_cfg", None), dict) else {}).get("REGION") + or (self.gcp_cfg if isinstance(getattr(self, "gcp_cfg", None), dict) else {}).get("location") + or (self.gcp_cfg if isinstance(getattr(self, "gcp_cfg", None), dict) else {}).get("LOCATION") + ) + else: + cluster_name = _safe_str( + self._get_input("init_password.db_cluster_name", "") + or self._get_input("env_setup.DB_CLUSTER_NAME", "") + or glob.get("DB_CLUSTER_NAME", "") + or glob.get("CNPG_CLUSTER_NAME", "") + or DEFAULT_DB_CLUSTER_NAME + ) or DEFAULT_DB_CLUSTER_NAME + region = _safe_str( + self._get_input("init_cluster.db_cluster_region", "") + or self._get_input("env_setup.DB_CLUSTER_REGION", "") + or (self.gcp_cfg if isinstance(getattr(self, "gcp_cfg", None), dict) else {}).get("region") + or (self.gcp_cfg if isinstance(getattr(self, "gcp_cfg", None), dict) else {}).get("REGION") + or (self.gcp_cfg if isinstance(getattr(self, "gcp_cfg", None), dict) else {}).get("location") + or (self.gcp_cfg if isinstance(getattr(self, "gcp_cfg", None), dict) else {}).get("LOCATION") + ) + + gcp_cfg = self.gcp_cfg if isinstance(getattr(self, "gcp_cfg", None), dict) else {} + project_id = _safe_str( + self._get_input("init_cluster.project_id", "") + or gcp_cfg.get("project_id") + or gcp_cfg.get("PROJECT_ID") + ) + if not region: + region = _safe_str( + gcp_cfg.get("region") + or gcp_cfg.get("REGION") + or gcp_cfg.get("location") + or gcp_cfg.get("LOCATION") + ) + if project_id and region: + return f"gke_{project_id}_{region}_{cluster_name}" + + parts = selected_context.split("_", 3) + if len(parts) == 4: + return f"{parts[0]}_{parts[1]}_{parts[2]}_{cluster_name}" + + return selected_context def _cnpg_desired_instances(self) -> int: glob = self.prole_cfg_data.get("Global", {}) or {} @@ -5571,7 +5668,10 @@ class KnoeConsoleInstaller(KnoeInstaller): service_ns = self._service_namespace() env = self._script_env_for_namespace(service_ns) mode = env.get("PROLE_MODE") or self._deployment_mode() + manage_opentofu = str(mode).lower() != "k8s" registry_ns = self._registry_namespace() + if not manage_opentofu: + self.log("[INFO] k8s mode: skipping OpenTofu checks/deploy in common services.") try: # Avoid deploying into a namespace that is currently being deleted. @@ -5595,36 +5695,38 @@ class KnoeConsoleInstaller(KnoeInstaller): except Exception: return False - is_healthy = all( - [ - _status_safe( - lambda: registry_ops.status( - namespace=registry_ns, - env=env, - mode=mode, - ) - ), - _status_safe( - lambda: openbao_ops.status( - namespace=service_ns, - env=env, - mode=mode, - ) - ), - _status_safe( - lambda: garage_store_ops.status( - namespace=service_ns, - env=env, - ) - ), + status_checks = [ + _status_safe( + lambda: registry_ops.status( + namespace=registry_ns, + env=env, + mode=mode, + ) + ), + _status_safe( + lambda: openbao_ops.status( + namespace=service_ns, + env=env, + mode=mode, + ) + ), + _status_safe( + lambda: garage_store_ops.status( + namespace=service_ns, + env=env, + ) + ), + ] + if manage_opentofu: + status_checks.append( _status_safe( lambda: opentofu_ops.status( namespace=service_ns, env=env, ) - ), - ] - ) + ) + ) + is_healthy = all(status_checks) if is_healthy: self.log( f"[OK] Common services in {service_ns} are healthy. Skipping update." @@ -5656,47 +5758,50 @@ class KnoeConsoleInstaller(KnoeInstaller): project_root=self.project_root, log=self.log, ) - opentofu_ops.update( - namespace=service_ns, - env=env, - mode=mode, - project_root=self.project_root, - log=self.log, - ) + if manage_opentofu: + opentofu_ops.update( + namespace=service_ns, + env=env, + mode=mode, + project_root=self.project_root, + log=self.log, + ) except Exception as e: self.err(f"[ERROR] Common services deploy failed: {e}") else: # Re-validate after fix. - repaired = all( - [ - _status_safe( - lambda: registry_ops.status( - namespace=registry_ns, - env=env, - mode=mode, - ) - ), - _status_safe( - lambda: openbao_ops.status( - namespace=service_ns, - env=env, - mode=mode, - ) - ), - _status_safe( - lambda: garage_store_ops.status( - namespace=service_ns, - env=env, - ) - ), + repair_checks = [ + _status_safe( + lambda: registry_ops.status( + namespace=registry_ns, + env=env, + mode=mode, + ) + ), + _status_safe( + lambda: openbao_ops.status( + namespace=service_ns, + env=env, + mode=mode, + ) + ), + _status_safe( + lambda: garage_store_ops.status( + namespace=service_ns, + env=env, + ) + ), + ] + if manage_opentofu: + repair_checks.append( _status_safe( lambda: opentofu_ops.status( namespace=service_ns, env=env, ) - ), - ] - ) + ) + ) + repaired = all(repair_checks) if not repaired: self.err( f"[ERROR] Common services still unhealthy after repair attempt." @@ -5862,11 +5967,13 @@ class KnoeConsoleInstaller(KnoeInstaller): return self.log("==> Initialization scripts") - ns = ( - self._get_input("init_password.db_namespace", "") or "" - ).strip() or "default" - db_env = self._script_env_for_namespace(ns, cluster_role="db") - app_env = self._script_env_for_namespace(ns, cluster_role="app") + mode = self._deployment_mode() + db_ns = self._secret_namespace() + if mode == "k8s" and db_ns == "default": + db_ns = "knoe-db-0" + service_ns = self._service_namespace() + db_env = self._script_env_for_namespace(db_ns, cluster_role="db") + app_env = self._script_env_for_namespace(service_ns, cluster_role="app") opt_allowed, _opt_count, opt_reason = self._optional_workloads_policy(app_env) @@ -5884,20 +5991,22 @@ class KnoeConsoleInstaller(KnoeInstaller): if not opt_allowed: self.log(f"[SKIP] Monitoring disabled by policy: {opt_reason}") self.prole_cfg_data.setdefault("Monitoring", {})["STATUS"] = "Skipped" - mode = self._deployment_mode() if mode != "k3d": post_cnpg_steps.append(("init_nginx_ingress.sh", ["initialize"], False, "app")) overall_success = True # Common services (Python owners) — replaces init_common_services.sh - self.log(f"[app-cluster {self._app_cluster_name()}] common services (python owners)") + self.log( + f"[app-cluster {self._app_cluster_name()} context={app_env.get('KUBECONTEXT','')}] " + f"common services namespace={service_ns}" + ) try: - registry_ns = str(app_env.get("REGISTRY_NAMESPACE") or ns).strip() or ns + registry_ns = str(app_env.get("REGISTRY_NAMESPACE") or service_ns).strip() or service_ns registry_ops.update(namespace=registry_ns, env=app_env, mode=mode, log=self.log) - openbao_ops.update(namespace=ns, env=app_env, mode=mode, log=self.log) - garage_store_ops.update(namespace=ns, env=app_env, mode=mode, log=self.log) - opentofu_ops.update(namespace=ns, env=app_env, mode=mode, log=self.log) + openbao_ops.update(namespace=service_ns, env=app_env, mode=mode, log=self.log) + garage_store_ops.update(namespace=service_ns, env=app_env, mode=mode, log=self.log) + opentofu_ops.update(namespace=service_ns, env=app_env, mode=mode, log=self.log) except Exception as e: self.err(f"[ERROR] common services (python owners) failed: {e}") overall_success = False @@ -5919,16 +6028,19 @@ class KnoeConsoleInstaller(KnoeInstaller): # CNPG initialization (Python) — replaces init_cloudnative_pg.sh initialize if overall_success: - self.log(f"[db-cluster {self._cnpg_cluster_name()}] cnpg_initialize (Python)") + self.log( + f"[db-cluster {self._cnpg_cluster_name()} context={db_env.get('KUBECONTEXT','')}] " + f"cnpg_initialize namespace={db_ns}" + ) try: - self.ensure_db_k8s_secrets(ns, password, log_fn=self.log) + self.ensure_db_k8s_secrets(db_ns, password, log_fn=self.log) except Exception as e: self.err(f"[ERROR] Failed to ensure DB secrets before CNPG init: {e}") overall_success = False if overall_success: if self._cluster_storage_milestone_enabled(): try: - self._ensure_cnpg_storage_provisioned(ns, db_env) + self._ensure_cnpg_storage_provisioned(db_ns, db_env) except Exception as e: self.err(f"[WARN] Failed to provision CNPG storage before init: {e}") else: @@ -5938,7 +6050,7 @@ class KnoeConsoleInstaller(KnoeInstaller): ) try: cnpg_initialize( - namespace=ns, + namespace=db_ns, cluster_name=str(db_env.get("CLUSTER_NAME") or db_env.get("CNPG_CLUSTER_NAME") or "knoe-db").strip(), env=db_env, project_root=self.project_root, @@ -5983,17 +6095,20 @@ class KnoeConsoleInstaller(KnoeInstaller): self._scripts_success = overall_success # Verify critical secrets - self.log(f"==> Verifying critical secrets in namespace {ns}") + self.log(f"==> Verifying critical secrets in namespace {db_ns}") critical_secrets = ["knoe-db-user", "knoe-db-superuser", "cnpg-admin-key"] missing_secrets = [] for secret in critical_secrets: - rc_s = self._run_cmd(["kubectl", "get", "secret", secret, "-n", ns]) + rc_s = self._run_cmd( + ["kubectl", "get", "secret", secret, "-n", db_ns], + env=db_env, + ) if rc_s != 0: missing_secrets.append(secret) if missing_secrets: self.err( - f"[CRITICAL] Missing secrets in namespace '{ns}': {', '.join(missing_secrets)}" + f"[CRITICAL] Missing secrets in namespace '{db_ns}': {', '.join(missing_secrets)}" ) self.err("Database initialization will fail without these secrets.") self._scripts_success = False @@ -6012,11 +6127,9 @@ class KnoeConsoleInstaller(KnoeInstaller): return db_cluster = self._cnpg_cluster_name() self.log(f"==> [db-cluster {db_cluster}] Deploy CloudNativePG") - ns = (self._get_input("init_password.db_namespace", "") or "").strip() - if not ns: - ns = (self._get_input("env_setup.DATABASE_NAMESPACE", "") or "").strip() - if not ns: - ns = "default" + ns = self._secret_namespace() + if self._deployment_mode() == "k8s" and ns == "default": + ns = "knoe-db-0" env = self._script_env_for_namespace(ns, cluster_role="db") cluster_name = str( env.get("CLUSTER_NAME") @@ -6537,6 +6650,133 @@ class KnoeConsoleInstaller(KnoeInstaller): ) out.flush() + def _db_master_vault_file(self) -> Path: + raw = (os.environ.get("PROLE_DB_MASTER_VAULT_FILE") or "").strip() + if raw: + expanded = _expand_cfg_value(raw, _collect_cfg_vars()) + return Path(expanded).expanduser() + return ( + self.project_root + / "infrastructure" + / "inventory" + / "group_vars" + / "all" + / "vault_db_master.yml" + ) + + def _db_master_vault_key(self) -> str: + key = (os.environ.get("PROLE_DB_MASTER_VAULT_KEY") or "").strip() + return key or "vault_prole_db_master_password" + + def _resolve_ansible_vault_password_file(self) -> str: + vault_file = (os.environ.get("ANSIBLE_VAULT_PASSWORD_FILE") or "").strip() + if vault_file and Path(vault_file).expanduser().is_file(): + return str(Path(vault_file).expanduser()) + for base in (self.project_root, Path.cwd()): + candidate = base / ".vault_pass" + if candidate.is_file(): + os.environ["ANSIBLE_VAULT_PASSWORD_FILE"] = str(candidate) + return str(candidate) + return "" + + def _load_db_password_from_ansible_vault(self, *, log_found: bool = True) -> str: + vault_file = self._db_master_vault_file() + vault_key = self._db_master_vault_key() + password = _try_read_ansible_vault_value(vault_file, vault_key).strip() + if password and log_found: + self.log( + f"[CONFIG] Loaded DB master password from Ansible Vault ({vault_file})." + ) + return password + + def _persist_db_password_to_ansible_vault(self, password: str) -> None: + if not password: + raise RuntimeError("Cannot persist an empty database master password.") + if shutil.which("ansible-vault") is None: + raise RuntimeError( + "ansible-vault command is required to store the database master password securely." + ) + + vault_file = self._db_master_vault_file() + vault_key = self._db_master_vault_key() + vault_file.parent.mkdir(parents=True, exist_ok=True) + + vault_password_file = self._resolve_ansible_vault_password_file() + vault_password = (os.environ.get("ANSIBLE_VAULT_PASSWORD") or "").strip() + if not vault_password_file and not vault_password: + raise RuntimeError( + "Ansible Vault password is not configured. Set ANSIBLE_VAULT_PASSWORD_FILE (or create .vault_pass) " + "or set ANSIBLE_VAULT_PASSWORD before bootstrap." + ) + + tmp_password_file = None + if not vault_password_file and vault_password: + tmp_pw = tempfile.NamedTemporaryFile(mode="w", delete=False) + tmp_pw.write(vault_password) + tmp_pw.flush() + tmp_pw.close() + tmp_password_file = tmp_pw.name + vault_password_file = tmp_password_file + + cmd = [ + "ansible-vault", + "encrypt_string", + "--name", + vault_key, + password, + ] + if vault_password_file: + cmd += ["--vault-password-file", vault_password_file] + + try: + res = subprocess.run( + cmd, + capture_output=True, + text=True, + timeout=20, + env=os.environ.copy(), + stdin=subprocess.DEVNULL, + ) + if res.returncode != 0: + raise RuntimeError( + "Failed to write database master password to Ansible Vault: " + f"{(res.stderr or res.stdout or '').strip()}" + ) + + rendered_block = (res.stdout or "").rstrip() + "\n" + if not rendered_block.strip(): + raise RuntimeError( + "Failed to write database master password to Ansible Vault: empty encrypted output." + ) + + existing = "" + if vault_file.exists(): + try: + existing = vault_file.read_text() + except Exception: + existing = "" + + key_pattern = rf"(?ms)^\s*{re.escape(vault_key)}\s*:.*?(?=^\S|\Z)" + if existing and re.search(key_pattern, existing): + updated = re.sub(key_pattern, rendered_block.rstrip(), existing, count=1) + output_text = updated.rstrip() + "\n" + elif existing.strip(): + output_text = existing.rstrip() + "\n\n" + rendered_block + else: + output_text = rendered_block + + vault_file.write_text(output_text) + os.chmod(vault_file, 0o600) + self.log( + f"[CONFIG] Persisted DB master password into Ansible Vault ({vault_file}:{vault_key})." + ) + finally: + if tmp_password_file: + try: + os.unlink(tmp_password_file) + except Exception: + pass + def run(self) -> int: _configure_unbuffered_io() self.log(f"[CONFIG] Using {self.cfg_path}") @@ -6547,23 +6787,65 @@ class KnoeConsoleInstaller(KnoeInstaller): # If the value is an OpenBao reference or an encrypted `${PROLE_SECRET:...}` # placeholder, allow downstream secret-management steps to resolve it. # Only prompt/generate when the value is truly missing. - db_pw = self._get_input("init_password.db_password", "") + db_pw = (self._get_input("init_password.db_password", "") or "").strip() + + env_db_pw = ( + os.environ.get("KNOE_DB_PASSWORD") + or os.environ.get("DB_PASSWORD") + or "" + ).strip() + + # If runtime env provides a password, reconcile with vault for + # consistency and bootstrap vault when missing. + if db_pw and env_db_pw and db_pw == env_db_pw: + vault_pw = self._load_db_password_from_ansible_vault(log_found=False) + if vault_pw and vault_pw != db_pw: + self.log( + "[WARN] Runtime DB password differs from Ansible Vault value; " + "using vault password for consistency." + ) + self.inputs["init_password.db_password"] = vault_pw + self.inputs["init_password.db_password_confirm"] = vault_pw + db_pw = vault_pw + elif not vault_pw: + self._persist_db_password_to_ansible_vault(db_pw) + self.log( + "[CONFIG] Bootstrapped Ansible Vault from KNOE_DB_PASSWORD/DB_PASSWORD." + ) + + if not db_pw: + vault_pw = self._load_db_password_from_ansible_vault() + if vault_pw: + self.inputs["init_password.db_password"] = vault_pw + self.inputs["init_password.db_password_confirm"] = vault_pw + db_pw = vault_pw needs_prompt = not bool(db_pw) if needs_prompt: # In silent installs, never block on an interactive prompt in environments # that cannot provide input (CI/pytest/non-interactive runners). stdin_is_tty = getattr(sys.__stdin__, "isatty", lambda: False)() + stdout_is_tty = getattr(sys.__stdout__, "isatty", lambda: False)() running_under_pytest = "PYTEST_CURRENT_TEST" in os.environ running_in_ci = bool(os.environ.get("CI")) - allow_prompt = stdin_is_tty and not running_under_pytest and not running_in_ci + allow_prompt = ( + stdin_is_tty + and stdout_is_tty + and not running_under_pytest + and not running_in_ci + ) if allow_prompt: new_pw = self._prompt_for_master_password() - else: - self.err( - "[WARN] Silent install requires a database master password but prompting is not available; generating one automatically." + self._persist_db_password_to_ansible_vault(new_pw) + self.log( + "[CONFIG] Master password captured interactively and saved for subsequent runs." + ) + else: + raise RuntimeError( + "Database master password is missing and prompting is unavailable. " + "Run once in an interactive terminal to bootstrap Ansible Vault, " + "or provide KNOE_DB_PASSWORD/DB_PASSWORD." ) - new_pw = secrets.token_urlsafe(24) self.inputs["init_password.db_password"] = new_pw self.inputs["init_password.db_password_confirm"] = new_pw # Save immediately to prole.cfg so subsequent steps/scripts see it. diff --git a/knoe/core/milestones.py b/knoe/core/milestones.py index 75244ac..fd2e5aa 100644 --- a/knoe/core/milestones.py +++ b/knoe/core/milestones.py @@ -623,22 +623,25 @@ class ClusterLifecycleMilestone(Milestone): # completed automatically — mirroring the UI shortcut behaviour. if env_key in ("service", "k3s"): service_ns = env.get("SERVICE_NAMESPACE", env.get("NAMESPACE", "default")) - status_ok = all( - [ - registry_ops.status( - namespace=env.get("REGISTRY_NAMESPACE", service_ns), - env=env, - mode=env.get("PROLE_MODE"), - ), - openbao_ops.status( - namespace=service_ns, - env=env, - mode=env.get("PROLE_MODE"), - ), - garage_store_ops.status(namespace=service_ns, env=env), - opentofu_ops.status(namespace=service_ns, env=env), - ] - ) + mode = str(env.get("PROLE_MODE", "")).strip().lower() + status_checks = [ + registry_ops.status( + namespace=env.get("REGISTRY_NAMESPACE", service_ns), + env=env, + mode=env.get("PROLE_MODE"), + ), + openbao_ops.status( + namespace=service_ns, + env=env, + mode=env.get("PROLE_MODE"), + ), + garage_store_ops.status(namespace=service_ns, env=env), + ] + if mode != "k8s": + status_checks.append(opentofu_ops.status(namespace=service_ns, env=env)) + else: + self.logger.info("k8s mode: skipping OpenTofu health check.") + status_ok = all(status_checks) if status_ok: self.logger.info( "All common services healthy – running service layer migration inline." @@ -728,6 +731,7 @@ class InitializationScriptsMilestone(Milestone): ) -> None: env = self._get_script_env(state) mode = env.get("PROLE_MODE", "") + manage_opentofu = str(mode).strip().lower() != "k8s" mode_args = ["--mode", mode] if mode else [] raw_min = (state.config_data.get("Global", {}) or {}).get(POLICY_CFG_KEY, "") @@ -771,13 +775,16 @@ class InitializationScriptsMilestone(Milestone): mode=mode, log=self.logger.info, ) - opentofu_ops.update( - namespace=service_ns, - env=env, - project_root=project_root, - mode=mode, - log=self.logger.info, - ) + if manage_opentofu: + opentofu_ops.update( + namespace=service_ns, + env=env, + project_root=project_root, + mode=mode, + log=self.logger.info, + ) + else: + self.logger.info("k8s mode: skipping OpenTofu deploy.") except Exception as e: msg = f"Common services (python owners) failed: {e}" self.logger.error(msg) @@ -788,6 +795,15 @@ class InitializationScriptsMilestone(Milestone): progress("Initializing CloudNative-PG...", 1 / 6) ns = env.get("DATABASE_NAMESPACE") or env.get("NAMESPACE", "default") cluster_name = env.get("CLUSTER_NAME") or env.get("CNPG_CLUSTER_NAME") or "knoe-db" + + # Build a DB-cluster-scoped env for CNPG operations. + # In k8s mode the default KUBECONTEXT is the app cluster; CNPG lives on the + # dedicated DB cluster, so override KUBECONTEXT to DB_CLUSTER_KUBECONTEXT. + cnpg_env = dict(env) + _db_ctx = env.get("DB_CLUSTER_KUBECONTEXT", "").strip() + if _db_ctx: + cnpg_env["KUBECONTEXT"] = _db_ctx + # Ensure DB secrets exist before CNPG init (mirrors _step_init_scripts). # env["DB_PASSWORD"] is already resolved by _script_env_for_namespace. db_pw = (env.get("DB_PASSWORD") or state.inputs.get("init_password.db_password") or "").strip() @@ -799,7 +815,7 @@ class InitializationScriptsMilestone(Milestone): cnpg_initialize( namespace=ns, cluster_name=cluster_name, - env=env, + env=cnpg_env, project_root=project_root, log=self.logger.info, mode=mode, @@ -819,13 +835,17 @@ class InitializationScriptsMilestone(Milestone): if script == "init_cnpg_backup.sh": args: list[str] = mode_args + ["start"] + # CNPG backup runs on the DB cluster, not the app cluster + script_env = cnpg_env elif script == "init_kong.sh": args = mode_args + ["start"] + script_env = env else: args = mode_args + ["initialize"] + script_env = env rc = state.controller.run_script( - script, args=args, env=env, on_line=_stream_line + script, args=args, env=script_env, on_line=_stream_line ) if rc != 0: msg = f"Script {script} failed (code {rc})" @@ -848,13 +868,19 @@ class InitializationScriptsMilestone(Milestone): self.logger.error(msg) raise Exception(msg) - # Verify critical secrets - ns = env.get("NAMESPACE", "default") + # Verify critical secrets (in the DB cluster namespace where CNPG lives) + ns = env.get("DATABASE_NAMESPACE") or env.get("NAMESPACE", "default") self.logger.info(f"Verifying critical secrets in namespace {ns}...") critical_secrets = ["knoe-db-user", "knoe-db-superuser", "cnpg-admin-key"] missing_secrets = [] for secret in critical_secrets: - if self._run_cmd(f"kubectl get secret {secret} -n {ns}") != 0: + if ( + self._run_cmd( + ["kubectl", "get", "secret", secret, "-n", ns], + env=cnpg_env, + ) + != 0 + ): missing_secrets.append(secret) if missing_secrets: diff --git a/knoe/core/ops/cloudnative_pg.py b/knoe/core/ops/cloudnative_pg.py index 7f06808..30d8697 100644 --- a/knoe/core/ops/cloudnative_pg.py +++ b/knoe/core/ops/cloudnative_pg.py @@ -62,26 +62,42 @@ def _log(log: _LogFn | None, msg: str) -> None: _LOG.info(msg) +def _kubectl_cmd(args: list[str], env: dict | None = None) -> list[str]: + """Build a kubectl command that honors env['KUBECONTEXT'] as --context.""" + ctx = str((env or {}).get("KUBECONTEXT") or "").strip() + cmd = ["kubectl"] + if ctx: + cmd.extend(["--context", ctx]) + cmd.extend(args) + return cmd + + +def _kubectl_run( + args: list[str], + env: dict | None = None, + timeout: int = 30, + check: bool = False, + input_text: str | None = None, +) -> subprocess.CompletedProcess: + """Run kubectl with optional stdin payload and context-aware command building.""" + return subprocess.run( + _kubectl_cmd(args, env=env), + env=env, + capture_output=True, + text=True, + timeout=timeout, + check=check, + input=input_text, + ) + + def _kubectl( args: list[str], env: dict | None = None, timeout: int = 30, check: bool = False, ) -> subprocess.CompletedProcess: - # KUBECONTEXT is not a real kubectl env var — extract it and pass as --context flag. - ctx = str((env or {}).get("KUBECONTEXT") or "").strip() - cmd = ["kubectl"] - if ctx: - cmd.extend(["--context", ctx]) - cmd.extend(args) - return subprocess.run( - cmd, - env=env, - capture_output=True, - text=True, - timeout=timeout, - check=check, - ) + return _kubectl_run(args, env=env, timeout=timeout, check=check) def _kubectl_ok(args: list[str], env: dict | None = None, timeout: int = 15) -> bool: @@ -98,10 +114,7 @@ def _release_stale_pv_claims( Only touches PVs whose status.phase is 'Released' for the given storageClass. Idempotent — Available/Bound PVs are left untouched. """ - r = subprocess.run( - ["kubectl", "get", "pv", "-o", "json"], - text=True, capture_output=True, env=env, timeout=20, - ) + r = _kubectl(["get", "pv", "-o", "json"], env=env, timeout=20) if r.returncode != 0: _log(log, f"WARN: could not list PVs: {r.stderr.strip()}") return @@ -119,9 +132,10 @@ def _release_stale_pv_claims( return _log(log, f"Releasing claimRef on {len(released)} Released '{storage_class}' PV(s): {', '.join(released)}") for pv_name in released: - pr = subprocess.run( - ["kubectl", "patch", "pv", pv_name, "-p", '{"spec":{"claimRef":null}}'], - text=True, capture_output=True, env=env, timeout=15, + pr = _kubectl( + ["patch", "pv", pv_name, "-p", '{"spec":{"claimRef":null}}'], + env=env, + timeout=15, ) if pr.returncode != 0: _log(log, f"WARN: could not release claimRef on PV '{pv_name}': {pr.stderr.strip()}") @@ -196,13 +210,7 @@ def _ensure_namespace(namespace: str, env: dict | None = None) -> None: "metadata:\n" f" name: {namespace}\n" ) - res = subprocess.run( - ["kubectl", "apply", "-f", "-"], - input=manifest, - text=True, - capture_output=True, - env=env, - ) + res = _kubectl_run(["apply", "-f", "-"], env=env, input_text=manifest) if res.returncode != 0: raise RuntimeError(f"Failed to ensure namespace '{namespace}':\n{res.stderr}") @@ -416,6 +424,113 @@ def _wait_pod_ready( _log(log, f"WARN: pod {pod} did not become Ready within {timeout}s") +def _cnpg_pod_readiness_lines(namespace: str, cluster_name: str, env: dict | None) -> list[str]: + r = _kubectl( + [ + "-n", namespace, "get", "pods", + "-l", f"cnpg.io/cluster={cluster_name}", + "-o", + "jsonpath={range .items[*]}{.metadata.name}={.status.conditions[?(@.type==\"Ready\")].status}\\n{end}", + ], + env=env, + timeout=15, + ) + if r.returncode != 0: + return [] + # kubectl jsonpath emits literal \n (backslash-n) not real newlines + raw = (r.stdout or "").replace("\\n", "\n") + return [line for line in raw.splitlines() if "=" in line] + + +def _cnpg_timeout_diagnostics(namespace: str, cluster_name: str, env: dict | None) -> str: + parts: list[str] = [] + + pod_status = _kubectl( + [ + "-n", namespace, "get", "pods", + "-l", f"cnpg.io/cluster={cluster_name}", + "-o", + "jsonpath={range .items[*]}{.metadata.name}|phase={.status.phase}|node={.spec.nodeName}|ready={.status.conditions[?(@.type==\"Ready\")].status}|reason={.status.containerStatuses[0].state.waiting.reason}\\n{end}", + ], + env=env, + timeout=20, + ) + if pod_status.returncode == 0: + pod_lines = [line for line in (pod_status.stdout or "").replace("\\n", "\n").splitlines() if line.strip()] + if pod_lines: + parts.append("Pods: " + "; ".join(pod_lines)) + + failed_sched = _kubectl( + [ + "-n", namespace, "get", "events", + "--field-selector", "reason=FailedScheduling,involvedObject.kind=Pod", + "-o", "jsonpath={range .items[*]}{.involvedObject.name}|{.message}\\n{end}", + ], + env=env, + timeout=20, + ) + if failed_sched.returncode == 0: + event_lines = [line for line in (failed_sched.stdout or "").replace("\\n", "\n").splitlines() if line.strip()] + if event_lines: + parts.append("FailedScheduling: " + " | ".join(event_lines[-3:])) + + if not parts: + return "No additional scheduling diagnostics available." + return "\n".join(parts) + + +def _relax_cnpg_workload_node_selector_if_unmatched( + namespace: str, + cluster_name: str, + env: dict | None, + log: _LogFn | None, +) -> bool: + cfg = env or {} + relax_enabled = str(cfg.get("CNPG_AUTO_RELAX_WORKLOAD_SELECTOR", "true") or "true").strip().lower() + if relax_enabled in {"0", "false", "no", "off"}: + return False + + selector_key = str(cfg.get("CNPG_WORKLOAD_SELECTOR_KEY", "workload") or "workload").strip() or "workload" + selector_value_res = _kubectl( + [ + "-n", namespace, "get", "cluster", cluster_name, + "-o", f"jsonpath={{.spec.affinity.nodeSelector.{selector_key}}}", + ], + env=env, + timeout=15, + ) + selector_value = (selector_value_res.stdout or "").strip() + if selector_value_res.returncode != 0 or not selector_value: + return False + + selector_expr = f"{selector_key}={selector_value}" + matching_nodes = _kubectl(["get", "nodes", "-l", selector_expr, "--no-headers"], env=env, timeout=15) + if matching_nodes.returncode != 0: + return False + matched = len([line for line in (matching_nodes.stdout or "").splitlines() if line.strip()]) + if matched > 0: + return False + + _log( + log, + "WARN: CNPG pods appear unschedulable because no nodes match " + f"'{selector_expr}'. Removing spec.affinity.nodeSelector as tiny-cluster fallback.", + ) + patch = json.dumps({"spec": {"affinity": {"nodeSelector": None}}}) + patched = _kubectl( + ["-n", namespace, "patch", "cluster", cluster_name, "--type", "merge", "-p", patch], + env=env, + timeout=30, + ) + if patched.returncode != 0: + reason = (patched.stderr or patched.stdout or "").strip() + _log(log, f"WARN: failed to relax CNPG nodeSelector: {reason}") + return False + + _log(log, "CNPG cluster nodeSelector relaxed successfully; retrying pod readiness wait.") + return True + + def _wait_cnpg_pods( namespace: str, cluster_name: str, @@ -440,29 +555,29 @@ def _wait_cnpg_pods( _log(log, f"Waiting for {target} CNPG pod(s) for cluster '{cluster_name}' in '{namespace}' (timeout={timeout}s)...") deadline = time.monotonic() + timeout + selector_relaxed = False last_feedback = 0.0 while True: now = time.monotonic() if now >= deadline: + if not selector_relaxed and _relax_cnpg_workload_node_selector_if_unmatched(namespace, cluster_name, env, log): + selector_relaxed = True + grace = int((env or {}).get("CNPG_SELECTOR_RELAX_GRACE_TIMEOUT", "300") or "300") + timeout = max(30, grace) + deadline = now + timeout + last_feedback = 0.0 + _log(log, f"Waiting up to {timeout}s for CNPG pods after nodeSelector relaxation...") + continue + + diagnostics = _cnpg_timeout_diagnostics(namespace, cluster_name, env) raise RuntimeError( f"Timed out ({timeout}s) waiting for {target} CNPG pods to be Ready " - f"in namespace '{namespace}'" + f"in namespace '{namespace}'.\n" + f"{diagnostics}" ) - r2 = _kubectl( - [ - "-n", namespace, "get", "pods", - "-l", f"cnpg.io/cluster={cluster_name}", - "-o", - "jsonpath={range .items[*]}{.metadata.name}={.status.conditions[?(@.type==\"Ready\")].status}\\n{end}", - ], - env=env, - timeout=15, - ) - if r2.returncode == 0: - # kubectl jsonpath emits literal \n (backslash-n) not real newlines - raw = (r2.stdout or "").replace("\\n", "\n") - lines = [l for l in raw.splitlines() if "=" in l] + lines = _cnpg_pod_readiness_lines(namespace, cluster_name, env) + if lines: ready = sum(1 for l in lines if l.split("=", 1)[1].strip() == "True") if ready >= target: _log(log, f"All {ready}/{target} CNPG pods are Ready.") @@ -873,10 +988,7 @@ def install_barman_plugin( _last_error: str = "" _apply_ok = False for _attempt in range(1, 7): # up to 6 attempts; first triggers cert-manager recovery - r = subprocess.run( - ["kubectl", "apply", "-f", url], - env=env, capture_output=True, text=True, timeout=120, - ) + r = _kubectl(["apply", "-f", url], env=env, timeout=120) if r.returncode == 0: if r.stdout.strip(): _log(log, r.stdout.strip()) @@ -926,7 +1038,7 @@ def install_barman_plugin( def _ensure_cert_manager_for_barman( env: dict | None = None, log: _LogFn | None = None, -) -> None: +) -> bool: def _exists(kind: str, name: str) -> bool: return _kubectl(["-n", "cert-manager", "get", kind, name], env=env, timeout=20).returncode == 0 @@ -980,13 +1092,7 @@ def _ensure_cert_manager_for_barman( "https://github.com/cert-manager/cert-manager/releases/download/" f"{certmgr_version}/cert-manager.yaml" ) - r = subprocess.run( - ["kubectl", "apply", "-f", certmgr_url], - env=env, - capture_output=True, - text=True, - timeout=180, - ) + r = _kubectl(["apply", "-f", certmgr_url], env=env, timeout=180) if r.returncode != 0: raise RuntimeError( f"Failed to bootstrap cert-manager from {certmgr_url}: " @@ -1028,12 +1134,10 @@ def _try_inject_ca_bundle_manually( Returns True if the patch succeeded and the caBundle is now non-empty. """ for secret_name in ("cert-manager-webhook-ca", "cert-manager-cainjector-leader-election"): - r = subprocess.run( - [ - "kubectl", "-n", "cert-manager", "get", "secret", secret_name, - "-o", "jsonpath={.data.tls\\.crt}", - ], - env=env, capture_output=True, text=True, timeout=15, + r = _kubectl( + ["-n", "cert-manager", "get", "secret", secret_name, "-o", "jsonpath={.data.tls\\.crt}"], + env=env, + timeout=15, ) ca_bundle = (r.stdout or "").strip() if r.returncode == 0 and ca_bundle: @@ -1046,12 +1150,16 @@ def _try_inject_ca_bundle_manually( patch_json = json.dumps( [{"op": "replace", "path": "/webhooks/0/clientConfig/caBundle", "value": ca_bundle}] ) - r2 = subprocess.run( + r2 = _kubectl( [ - "kubectl", "patch", "validatingwebhookconfiguration", "cert-manager-webhook", - "--type=json", f"-p={patch_json}", + "patch", + "validatingwebhookconfiguration", + "cert-manager-webhook", + "--type=json", + f"-p={patch_json}", ], - env=env, capture_output=True, text=True, timeout=20, + env=env, + timeout=20, ) if r2.returncode != 0: _log(log, f"WARN: Failed to patch caBundle: {(r2.stderr or r2.stdout or '').strip()}") @@ -1074,13 +1182,16 @@ def _wait_for_webhook_ca_bundle( deadline = time.time() + timeout interval = 5 while time.time() < deadline: - r = subprocess.run( + r = _kubectl( [ - "kubectl", "get", "validatingwebhookconfiguration", + "get", + "validatingwebhookconfiguration", "cert-manager-webhook", - "-o", "jsonpath={.webhooks[0].clientConfig.caBundle}", + "-o", + "jsonpath={.webhooks[0].clientConfig.caBundle}", ], - env=env, capture_output=True, text=True, timeout=15, + env=env, + timeout=15, ) if r.returncode == 0 and r.stdout.strip(): return True @@ -1109,12 +1220,18 @@ def _wait_for_barman_tls_secrets( all_ready = True for secret in ("barman-cloud-client-tls", "barman-cloud-server-tls"): for field in ("tls.crt", "tls.key"): - r = subprocess.run( + r = _kubectl( [ - "kubectl", "-n", "cnpg-system", "get", "secret", secret, - "-o", f"jsonpath={{.data.{field.replace('.', '\\.')}}}", + "-n", + "cnpg-system", + "get", + "secret", + secret, + "-o", + f"jsonpath={{.data.{field.replace('.', '\\.')}}}", ], - env=env, capture_output=True, text=True, timeout=15, + env=env, + timeout=15, ) if r.returncode != 0 or not (r.stdout or "").strip(): all_ready = False @@ -1151,12 +1268,10 @@ def _bootstrap_barman_tls_secrets( def _secret_has_tls(name: str) -> bool: for field in ("tls.crt", "tls.key"): - r = subprocess.run( - [ - "kubectl", "-n", namespace, "get", "secret", name, - "-o", f"jsonpath={{.data.{field.replace('.', '\\.')}}}", - ], - env=env, capture_output=True, text=True, timeout=15, + r = _kubectl( + ["-n", namespace, "get", "secret", name, "-o", f"jsonpath={{.data.{field.replace('.', '\\.')}}}"], + env=env, + timeout=15, ) if r.returncode != 0 or not (r.stdout or "").strip(): return False @@ -1230,9 +1345,10 @@ def _bootstrap_barman_tls_secrets( f" tls.crt: {base64.b64encode(cert_pem.encode()).decode()}\n" f" tls.key: {base64.b64encode(key_pem.encode()).decode()}\n" ) - res = subprocess.run( - ["kubectl", "apply", "-n", namespace, "-f", "-"], - input=manifest, text=True, capture_output=True, env=env, + res = _kubectl_run( + ["apply", "-n", namespace, "-f", "-"], + env=env, + input_text=manifest, ) if res.returncode != 0: raise RuntimeError(f"Failed to apply Barman TLS secret '{name}': {res.stderr}") @@ -1298,21 +1414,28 @@ def _bootstrap_db_user_secrets( if _kubectl_ok(["-n", namespace, "get", "secret", secret_name], env=env): return _log(log, f"Creating missing secret '{secret_name}' in '{namespace}'...") - manifest_res = subprocess.run( + manifest_res = _kubectl( [ - "kubectl", "create", "secret", "generic", secret_name, - "-n", namespace, + "create", + "secret", + "generic", + secret_name, + "-n", + namespace, f"--from-literal=username={username}", f"--from-literal=password={password}", - "--dry-run=client", "-o", "yaml", + "--dry-run=client", + "-o", + "yaml", ], - text=True, capture_output=True, env=env, + env=env, ) if manifest_res.returncode != 0: raise RuntimeError(f"Failed to render secret '{secret_name}':\n{manifest_res.stderr}") - apply_res = subprocess.run( - ["kubectl", "apply", "-n", namespace, "-f", "-"], - input=manifest_res.stdout, text=True, capture_output=True, env=env, + apply_res = _kubectl_run( + ["apply", "-n", namespace, "-f", "-"], + env=env, + input_text=manifest_res.stdout, ) if apply_res.returncode != 0: raise RuntimeError(f"Failed to apply secret '{secret_name}':\n{apply_res.stderr}") @@ -1363,21 +1486,28 @@ def _bootstrap_cnpg_admin_key( serialization.PublicFormat.SubjectPublicKeyInfo, ).decode() - manifest_res = subprocess.run( + manifest_res = _kubectl( [ - "kubectl", "create", "secret", "generic", secret_name, - "-n", namespace, + "create", + "secret", + "generic", + secret_name, + "-n", + namespace, f"--from-literal=admin.key={priv_pem}", f"--from-literal=admin.pub={pub_pem}", - "--dry-run=client", "-o", "yaml", + "--dry-run=client", + "-o", + "yaml", ], - text=True, capture_output=True, env=env, + env=env, ) if manifest_res.returncode != 0: raise RuntimeError(f"Failed to render secret '{secret_name}':\n{manifest_res.stderr}") - apply_res = subprocess.run( - ["kubectl", "apply", "-n", namespace, "-f", "-"], - input=manifest_res.stdout, text=True, capture_output=True, env=env, + apply_res = _kubectl_run( + ["apply", "-n", namespace, "-f", "-"], + env=env, + input_text=manifest_res.stdout, ) if apply_res.returncode != 0: raise RuntimeError(f"Failed to apply secret '{secret_name}':\n{apply_res.stderr}") @@ -1485,12 +1615,10 @@ def bootstrap_cnpg_tls_secrets( ).decode() def _apply_secret(manifest: str) -> None: - res = subprocess.run( - ["kubectl", "apply", "-n", namespace, "-f", "-"], - input=manifest, - text=True, - capture_output=True, + res = _kubectl_run( + ["apply", "-n", namespace, "-f", "-"], env=env, + input_text=manifest, ) if res.returncode != 0: raise RuntimeError(f"kubectl apply failed:\n{res.stderr}") @@ -1552,6 +1680,7 @@ def initialize( """ _log(log, f"Initializing CNPG cluster '{cluster_name}' in namespace '{namespace}'...") + _log(log, "[CNPG] Ensuring operator is installed and healthy...") ensure_operator( env=env, log=log, @@ -1559,16 +1688,21 @@ def initialize( allow_reapply_if_unhealthy=True, aggressive_cleanup=True, ) + _log(log, "[CNPG] Pinning controller placement...") pin_controller(env=env, log=log, mode=mode) + _log(log, "[CNPG] Installing Barman Cloud plugin...") install_barman_plugin(env=env, log=log) # Bootstrap TLS secrets (create-if-missing; no rotation of existing certs) + _log(log, "[CNPG] Bootstrapping TLS secrets...") bootstrap_cnpg_tls_secrets(namespace=namespace, cluster_name=cluster_name, env=env, log=log) # Bootstrap DB user secrets from env if missing (create-if-missing only) + _log(log, "[CNPG] Ensuring DB user secrets are present...") _bootstrap_db_user_secrets(namespace=namespace, env=env, log=log) # Bootstrap admin keypair secret (create-if-missing; no rotation) + _log(log, "[CNPG] Ensuring admin keypair secret is present...") _bootstrap_cnpg_admin_key(namespace=namespace, env=env, log=log) # Preflight: required secrets must exist before applying the cluster @@ -1592,10 +1726,12 @@ def initialize( env=env, log=log, ) + _log(log, "[CNPG] Applying cluster manifest...") manifest = _resolve_cnpg_manifest(project_root, env) _apply_manifest(namespace, manifest, env, log) wait_timeout = int((env or {}).get("CNPG_WAIT_TIMEOUT", "900") or "900") + _log(log, f"[CNPG] Waiting for cluster pods to become Ready (timeout={wait_timeout}s)...") _wait_cnpg_pods(namespace, cluster_name, env, timeout=wait_timeout, log=log) _log(log, f"Initialization complete for CNPG cluster '{cluster_name}'.") diff --git a/knoe/core/ops/gke_clusters.py b/knoe/core/ops/gke_clusters.py index ccb757c..a2d479d 100644 --- a/knoe/core/ops/gke_clusters.py +++ b/knoe/core/ops/gke_clusters.py @@ -19,6 +19,7 @@ class GkeClusterSpec: machine_type: str = "e2-small" node_count: int = 3 node_pool_name: str = "default-pool" + node_labels: str = "workload=db" boot_disk_type: str = "pd-standard" # pd-standard keeps boot disks off pd-ssd quota boot_disk_size_gb: int = 50 @@ -29,6 +30,22 @@ def _run(cmd: list[str], log: _LogFn | None = None) -> subprocess.CompletedProce return subprocess.run(cmd, capture_output=True, text=True) +def _nodes_per_location(node_count: int, location: str) -> int: + """Return GKE --num-nodes value, treating regional locations as per-zone counts. + + For regional clusters, gcloud interprets --num-nodes as per-zone, so convert + requested total nodes to a per-zone value (ceil(total/3)) for the common + 3-zone regional layout. + """ + requested = max(1, int(node_count)) + loc = (location or "").strip().lower() + suffix = loc.rsplit("-", 1)[-1] if "-" in loc else "" + is_zone = len(suffix) == 1 and suffix.isalpha() + if is_zone: + return requested + return max(1, (requested + 2) // 3) + + def build_kubectl_env_for_cluster( base_env: dict | None, kubecontext: str, @@ -37,6 +54,8 @@ def build_kubectl_env_for_cluster( ) -> dict: env = dict(base_env or os.environ) if kubecontext: + env["KUBECONTEXT"] = kubecontext + env["KUBE_CONTEXT_NAME"] = kubecontext env["KUBECTL_CONTEXT"] = kubecontext env["CLUSTER_NAME"] = cluster_name env["KNOE_CLUSTER_ROLE"] = cluster_role @@ -159,6 +178,11 @@ def ensure_db_cluster( spec: GkeClusterSpec, log: _LogFn | None = None, ) -> None: + if spec.mode.lower() != "standard": + raise RuntimeError( + f"Database cluster '{spec.name}' must run in Standard mode, got '{spec.mode}'." + ) + describe_cmd = [ "gcloud", "container", @@ -173,38 +197,112 @@ def ensure_db_cluster( "--quiet", ] describe = _run(describe_cmd, log=log) - if describe.returncode == 0 and (describe.stdout or "").strip() == spec.name: + cluster_exists = describe.returncode == 0 and (describe.stdout or "").strip() == spec.name + + if not cluster_exists: + create_cmd = [ + "gcloud", + "container", + "clusters", + "create", + spec.name, + "--project", + project_id, + "--region", + spec.location, + "--num-nodes", + str(_nodes_per_location(spec.node_count, spec.location)), + "--machine-type", + spec.machine_type, + "--disk-type", + spec.boot_disk_type, + "--disk-size", + str(spec.boot_disk_size_gb), + "--node-labels", + spec.node_labels, + "--enable-ip-alias", + "--workload-pool", + f"{project_id}.svc.id.goog", + "--quiet", + ] + created = _run(create_cmd, log=log) + if created.returncode != 0: + stderr = (created.stderr or "").strip() + raise RuntimeError(f"Failed to create DB cluster '{spec.name}': {stderr}") return - if spec.mode.lower() != "standard": - raise RuntimeError( - f"Database cluster '{spec.name}' must run in Standard mode, got '{spec.mode}'." - ) - - create_cmd = [ + pool_describe_cmd = [ "gcloud", "container", - "clusters", - "create", + "node-pools", + "describe", + spec.node_pool_name, + "--cluster", + spec.name, + "--project", + project_id, + "--region", + spec.location, + "--format=value(name)", + "--quiet", + ] + pool_describe = _run(pool_describe_cmd, log=log) + pool_exists = ( + pool_describe.returncode == 0 + and (pool_describe.stdout or "").strip() == spec.node_pool_name + ) + if pool_exists: + update_labels_cmd = [ + "gcloud", + "container", + "node-pools", + "update", + spec.node_pool_name, + "--cluster", + spec.name, + "--project", + project_id, + "--region", + spec.location, + "--node-labels", + spec.node_labels, + "--quiet", + ] + updated = _run(update_labels_cmd, log=log) + if updated.returncode != 0: + stderr = (updated.stderr or "").strip() + raise RuntimeError( + f"Failed to label DB node pool '{spec.node_pool_name}' for cluster '{spec.name}': {stderr}" + ) + return + + create_pool_cmd = [ + "gcloud", + "container", + "node-pools", + "create", + spec.node_pool_name, + "--cluster", spec.name, "--project", project_id, "--region", spec.location, - "--num-nodes", - str(max(3, int(spec.node_count))), "--machine-type", spec.machine_type, "--disk-type", spec.boot_disk_type, "--disk-size", str(spec.boot_disk_size_gb), - "--enable-ip-alias", - "--workload-pool", - f"{project_id}.svc.id.goog", + "--num-nodes", + str(_nodes_per_location(spec.node_count, spec.location)), + "--node-labels", + spec.node_labels, "--quiet", ] - created = _run(create_cmd, log=log) - if created.returncode != 0: - stderr = (created.stderr or "").strip() - raise RuntimeError(f"Failed to create DB cluster '{spec.name}': {stderr}") + pool_created = _run(create_pool_cmd, log=log) + if pool_created.returncode != 0: + stderr = (pool_created.stderr or "").strip() + raise RuntimeError( + f"Failed to create DB node pool '{spec.node_pool_name}' in cluster '{spec.name}': {stderr}" + ) diff --git a/knoe/core/ops/openbao.py b/knoe/core/ops/openbao.py index b24c0d2..4eb7719 100644 --- a/knoe/core/ops/openbao.py +++ b/knoe/core/ops/openbao.py @@ -51,13 +51,30 @@ def _apply( for manifest in _manifest_files(mode, project_root, include_kerberos_configmap): if not manifest.exists(): continue + if manifest.name == "kerberos-configmap.yaml": + _ensure_namespace("knoe-system", env) _log(log, f"[OPENBAO] Applying {manifest}") - _kubectl( - ["-n", namespace, "apply", "-f", str(manifest)], - env=env, - timeout=300, - check=True, - ) + try: + _kubectl( + ["-n", namespace, "apply", "-f", str(manifest)], + env=env, + timeout=300, + check=True, + ) + except RuntimeError as exc: + msg = str(exc) + if "does not match the namespace" not in msg: + raise + _log( + log, + "[OPENBAO] Namespace mismatch in manifest metadata; retrying apply without forced namespace.", + ) + _kubectl( + ["apply", "-f", str(manifest)], + env=env, + timeout=300, + check=True, + ) if mode == "k3s" or _exists("statefulset", "openbao", namespace, env=env): _wait_rollout("statefulset", "openbao", namespace, env=env) diff --git a/knoe/milestone.py b/knoe/milestone.py index 23f2dd6..b70af40 100644 --- a/knoe/milestone.py +++ b/knoe/milestone.py @@ -48,6 +48,7 @@ class Milestone(ABC): self, cmd: str | list[str], cwd: str | None = None, + env: dict | None = None, on_stdout: callable | None = None, ) -> int: if isinstance(cmd, str): @@ -57,7 +58,10 @@ class Milestone(ABC): import subprocess try: - env = inst_config._augment_env_for_brew(os.environ.copy()) + proc_env = os.environ.copy() + if env: + proc_env.update(env) + proc_env = inst_config._augment_env_for_brew(proc_env) proc = subprocess.Popen( cmd, stdout=subprocess.PIPE, @@ -65,7 +69,7 @@ class Milestone(ABC): text=True, bufsize=1, cwd=cwd, - env=env, + env=proc_env, ) for line in iter(proc.stdout.readline, ""): if on_stdout: @@ -172,6 +176,33 @@ class Milestone(ABC): default_kube = str(Path.home() / ".kube" / "config") if Path(default_kube).exists(): env["KUBECONFIG"] = default_kube + elif mode == "k8s": + # GKE / standard k8s — use ~/.kube/config and set explicit cluster contexts. + # APP cluster = common services (garage, openbao, registry, monitoring, kong). + # DB cluster = CNPG/PostgreSQL (knoe-cnpg-0). + default_kube = str(Path.home() / ".kube" / "config") + if not (env.get("KUBECONFIG") or "").strip(): + if Path(default_kube).exists(): + env["KUBECONFIG"] = default_kube + + app_ctx = ( + state.inputs.get("init_cluster.app_cluster_kubecontext", "") + or state.inputs.get("env_setup.APP_CLUSTER_KUBECONTEXT", "") + or (state.config_data.get("Global", {}) or {}).get("APP_CLUSTER_KUBECONTEXT", "") + ).strip() + db_ctx = ( + state.inputs.get("init_cluster.db_cluster_kubecontext", "") + or state.inputs.get("env_setup.DB_CLUSTER_KUBECONTEXT", "") + or (state.config_data.get("Global", {}) or {}).get("DB_CLUSTER_KUBECONTEXT", "") + ).strip() + + if app_ctx: + env["APP_CLUSTER_KUBECONTEXT"] = app_ctx + # Default KUBECONTEXT = app cluster; callers needing the DB cluster + # must override with DB_CLUSTER_KUBECONTEXT before invoking kubectl. + env["KUBECONTEXT"] = app_ctx + if db_ctx: + env["DB_CLUSTER_KUBECONTEXT"] = db_ctx db_pw = state.inputs.get("init_password.db_password", "").strip() if db_pw: diff --git a/modes/k3d/knoe-db/.version b/modes/k3d/knoe-db/.version index ca7bf83..da2d398 100644 --- a/modes/k3d/knoe-db/.version +++ b/modes/k3d/knoe-db/.version @@ -1 +1 @@ -13 \ No newline at end of file +14 \ No newline at end of file diff --git a/modes/k8s/knoe-db/.version b/modes/k8s/knoe-db/.version index 31ff414..7c6ba0f 100644 --- a/modes/k8s/knoe-db/.version +++ b/modes/k8s/knoe-db/.version @@ -1 +1 @@ -48 \ No newline at end of file +55 \ No newline at end of file diff --git a/prole/deploy_pipeline.py b/prole/deploy_pipeline.py index 4d7216f..3e65624 100644 --- a/prole/deploy_pipeline.py +++ b/prole/deploy_pipeline.py @@ -66,11 +66,29 @@ def _prepare_installer(controller: KnoeController, cfg_path: Path) -> KnoeConsol installer.inputs.setdefault("init_cluster.db_boot_disk_type", "pd-standard") installer.inputs.setdefault("init_cluster.db_boot_disk_size_gb", "50") + app_cluster_name = installer.inputs.get("init_cluster.app_cluster_name", "knoe-dev-0") + db_cluster_name = installer.inputs.get("init_cluster.db_cluster_name", "knoe-cnpg-0") + app_cluster_region = region + db_cluster_region = region + + # Production CNPG namespace is fixed by shipped manifests. + installer.inputs["init_password.db_namespace"] = "knoe-db-0" + installer.inputs["env_setup.DATABASE_NAMESPACE"] = "knoe-db-0" + if project_id: installer.inputs["init_cluster.project_id"] = project_id - installer.inputs["init_cluster.app_cluster_region"] = region - installer.inputs["init_cluster.db_cluster_region"] = region + if project_id and app_cluster_region: + installer.inputs["init_cluster.app_cluster_kubecontext"] = ( + f"gke_{project_id}_{app_cluster_region}_{app_cluster_name}" + ) + if project_id and db_cluster_region: + installer.inputs["init_cluster.db_cluster_kubecontext"] = ( + f"gke_{project_id}_{db_cluster_region}_{db_cluster_name}" + ) + + installer.inputs["init_cluster.app_cluster_region"] = app_cluster_region + installer.inputs["init_cluster.db_cluster_region"] = db_cluster_region # Ensure install phases run installer.inputs["init_cluster.start_cluster"] = "true" @@ -82,6 +100,34 @@ def _prepare_installer(controller: KnoeController, cfg_path: Path) -> KnoeConsol glob["CLUSTER_SIZE_PROFILE"] = "small" glob["CNPG_SIZE_PROFILE"] = "small" glob["CNPG_INSTANCES"] = "3" + glob["DATABASE_NAMESPACE"] = "knoe-db-0" + + db_creation = installer.prole_cfg_data.setdefault("Database Creation", {}) + db_creation["DATABASE_NAMESPACE"] = "knoe-db-0" + + gcp_section = installer.prole_cfg_data.setdefault("GCP", {}) + if project_id: + gcp_section["project_id"] = project_id + if region: + gcp_section["region"] = region + + service_ns = ( + installer.inputs.get("init_cluster.service_namespace") + or glob.get("SERVICE_NAMESPACE") + or "knoe-system" + ) + artifact_registry = ( + installer.inputs.get("init_cluster.artifact_registry") + or glob.get("ARTIFACT_REGISTRY") + or os.environ.get("ARTIFACT_REGISTRY") + or "" + ) + if not artifact_registry and project_id and region: + artifact_registry = f"{region}-docker.pkg.dev/{project_id}/{service_ns}" + if artifact_registry: + artifact_registry = artifact_registry.rstrip("/") + os.environ["ARTIFACT_REGISTRY"] = artifact_registry + glob["ARTIFACT_REGISTRY"] = artifact_registry return installer diff --git a/scan/network_description.txt b/scan/network_description.txt index b025759..01963f7 100644 --- a/scan/network_description.txt +++ b/scan/network_description.txt @@ -1,55 +1,53 @@ Network Discovery Summary: Primary Router: 10.0.0.1 (eero_5d:50:f2) -DNS Servers: 100.100.100.100, 10.0.0.5, 10.0.0.4 +DNS Servers: 100.100.100.100, 10.0.0.4, 10.0.0.5 Detected Devices: - 10.0.0.203 [00:11:32:3b:2f:08] (Synology_3b:2f:08): Ports [22, 80, 443, 2049, 445], Services: ['SSH', 'NFS', 'Web Server', 'SMB/CIFS (Possible Windows/AD)'] -- 10.0.0.31 [2e:c1:c1:e9:87:b4] (2e:c1:c1:e9:87:b4): Ports [22, 445, 5900, 11434, 88], Services: ['SSH', 'VNC', 'SMB/CIFS (Possible Windows/AD)', 'Ollama', 'Active Directory Related'] -- 10.0.0.2 [dc:a4:ca:ea:1a:2f] (Apple_ea:1a:2f): Ports [], Services: [] -- 10.0.0.1 [9c:57:bc:5d:50:f2] (eero_5d:50:f2): Ports [53], Services: ['DNS'] +- 10.0.0.39 [7e:19:9a:fe:9f:ed] (7e:19:9a:fe:9f:ed): Ports [22, 445, 5900, 11434, 88], Services: ['SSH', 'VNC', 'SMB/CIFS (Possible Windows/AD)', 'Ollama', 'Active Directory Related'] - 10.0.0.7 [2c:cf:67:ca:93:ce] (RaspberryPi_ca:93:ce): Ports [22], Services: ['SSH'] +- 10.0.0.4 [b8:27:eb:b3:6f:2b] (RaspberryPiF_b3:6f:2b): Ports [22, 53, 443, 2049], Services: ['DNS', 'SSH', 'NFS', 'Web Server'] +- 10.0.0.6 [2c:cf:67:45:8d:97] (RaspberryPi_45:8d:97): Ports [22, 80, 443], Services: ['SSH', 'Web Server'] - 10.0.0.26 [c8:db:26:08:52:f8] (Logitech_08:52:f8): Ports [], Services: [] - 10.0.0.32 [48:a6:b8:a4:8e:cc] (Sonos_a4:8e:cc): Ports [], Services: [] -- 10.0.0.33 [ec:b5:fa:b0:76:e4] (PhilipsLight_b0:76:e4): Ports [80, 443], Services: ['Web Server'] -- 10.0.0.31\ [2e:c1:c1:e9:87:b4] (2e:c1:c1:e9:87:b4): Ports [], Services: [] -- 10.0.0.6 [2c:cf:67:45:8d:97] (RaspberryPi_45:8d:97): Ports [22, 80, 443], Services: ['SSH', 'Web Server'] - 10.0.0.3 [2c:cf:67:8c:2b:47] (RaspberryPi_8c:2b:47): Ports [22, 53, 80, 443, 2049, 445, 5900, 88, 389, 636], Services: ['DNS', 'SSH', 'NFS', 'VNC', 'Web Server', 'SMB/CIFS (Possible Windows/AD)', 'Active Directory Related'] -- 10.0.0.4 [b8:27:eb:b3:6f:2b] (RaspberryPiF_b3:6f:2b): Ports [22, 53, 443, 2049], Services: ['DNS', 'SSH', 'NFS', 'Web Server'] -- 10.0.0.22 [2c:cf:67:45:8d:97] (RaspberryPi_45:8d:97): Ports [22, 80, 443], Services: ['SSH', 'Web Server'] +- 10.0.0.1 [9c:57:bc:5d:50:f2] (eero_5d:50:f2): Ports [53], Services: ['DNS'] +- 10.0.0.2 [dc:a4:ca:ea:1a:2f] (Apple_ea:1a:2f): Ports [], Services: [] - 10.0.0.5 [b8:27:eb:88:3a:41] (RaspberryPiF_88:3a:41): Ports [22, 2049, 5900], Services: ['SSH', 'NFS', 'VNC'] - 10.0.0.205 [a8:20:66:28:12:e7] (Apple_28:12:e7): Ports [22, 445, 5900, 88], Services: ['SSH', 'VNC', 'SMB/CIFS (Possible Windows/AD)', 'Active Directory Related'] +- 10.0.0.33 [ec:b5:fa:b0:76:e4] (PhilipsLight_b0:76:e4): Ports [80, 443], Services: ['Web Server'] - 10.0.0.37 [54:07:7d:22:c0:b2] (Netgear_22:c0:b2): Ports [80], Services: ['Web Server'] -- 10.0.0.38 [24:fc:e5:51:cf:74] (SamsungElect_51:cf:74): Ports [], Services: [] +- 10.0.0.44 [24:fc:e5:51:cf:74] (SamsungElect_51:cf:74): Ports [], Services: [] - 10.0.0.45 [f8:b4:6a:30:a4:1f] (HewlettPacka_30:a4:1f): Ports [80, 443], Services: ['Web Server'] - 10.0.0.48 [0c:fe:45:53:f4:3c] (SonyInteract_53:f4:3c): Ports [], Services: [] -- 10.0.0.52 [2c:cf:67:ca:93:ce] (RaspberryPi_ca:93:ce): Ports [22], Services: ['SSH'] - 10.0.0.36 [7e:46:74:2a:b0:b2] (7e:46:74:2a:b0:b2): Ports [], Services: [] +- 10.0.0.52 [2c:cf:67:ca:93:ce] (RaspberryPi_ca:93:ce): Ports [22], Services: ['SSH'] - 10.0.0.46 [d4:f7:d5:40:ab:17] (SonyInteract_40:ab:17): Ports [], Services: [] -- 10.0.0.107 [b8:06:0d:b7:7c:56] (TuyaSmart_b7:7c:56): Ports [], Services: [] -- 10.0.0.73 [4c:a9:19:b3:12:f8] (TuyaSmart_b3:12:f8): Ports [], Services: [] - 10.0.0.111 [ca:0b:6c:5d:0a:e5] (ca:0b:6c:5d:0a:e5): Ports [], Services: [] - 10.0.0.128 [20:c9:d0:94:30:dd] (Apple_94:30:dd): Ports [22, 445, 5900, 88], Services: ['SSH', 'VNC', 'SMB/CIFS (Possible Windows/AD)', 'Active Directory Related'] -- 10.0.0.95 [b8:27:eb:88:3a:41] (RaspberryPiF_88:3a:41): Ports [22, 2049, 5900], Services: ['SSH', 'NFS', 'VNC'] -- 10.0.0.99 [20:28:bc:f1:dc:ba] (Visionscape_f1:dc:ba): Ports [], Services: [] +- 10.0.0.73 [4c:a9:19:b3:12:f8] (TuyaSmart_b3:12:f8): Ports [], Services: [] +- 10.0.0.94 [f0:20:ff:bf:9a:19] (Intel_bf:9a:19): Ports [], Services: [] + Note: Potential host for NVIDIA cards (PC Hardware vendor) +- 10.0.0.95 [b8:27:eb:88:3a:41] (RaspberryPiF_88:3a:41): Ports [22], Services: ['SSH'] - 10.0.0.106 [48:a6:b8:a7:50:60] (Sonos_a7:50:60): Ports [], Services: [] -- 10.0.0.204 [50:eb:f6:56:f3:26] (ASUSTekCOMPU_56:f3:26): Ports [22, 3389, 445, 5900, 11434], Services: ['SSH', 'VNC', 'SMB/CIFS (Possible Windows/AD)', 'RDP (Windows)', 'Ollama'] - 10.0.0.123 [3c:ef:8c:96:e3:3c] (ZhejiangDahu_96:e3:3c): Ports [80], Services: ['Web Server'] -- 10.0.0.117 [40:f5:20:1e:5e:91] (Espressif_1e:5e:91): Ports [80], Services: ['Web Server'] - 10.0.0.124 [9c:8e:cd:02:73:c4] (AmcrestTechn_02:73:c4): Ports [80], Services: ['Web Server'] - 10.0.0.113 [60:81:10:92:02:4e] (Apple_92:02:4e): Ports [], Services: [] +- 10.0.0.107 [b8:06:0d:b7:7c:56] (TuyaSmart_b7:7c:56): Ports [], Services: [] - 10.0.0.125 [a0:60:32:04:73:ba] (AmcrestTechn_04:73:ba): Ports [80], Services: ['Web Server'] - 10.0.0.127 [58:55:ca:4a:60:6e] (Apple_4a:60:6e): Ports [22], Services: ['SSH'] -- 10.0.0.130 [f8:bb:bf:a3:84:eb] (eero_a3:84:eb): Ports [53], Services: ['DNS'] +- 10.0.0.130 [f8:bb:bf:a3:84:eb] (eero_a3:84:eb): Ports [], Services: [] +- 10.0.0.145 [7c:a6:b0:04:db:1c] (7c:a6:b0:04:db:1c): Ports [], Services: [] - 10.0.0.143 [60:5f:8d:7a:aa:32] (eero_7a:aa:32): Ports [53], Services: ['DNS'] - 10.0.0.155 [62:17:97:b3:06:a1] (62:17:97:b3:06:a1): Ports [], Services: [] -- 10.0.0.206 [00:a0:de:a2:0b:ef] (Yamaha_a2:0b:ef): Ports [80], Services: ['Web Server'] - 10.0.0.170 [b0:ee:7b:ca:73:99] (Roku_ca:73:99): Ports [], Services: [] +- 10.0.0.204 [50:eb:f6:56:f3:26] (ASUSTekCOMPU_56:f3:26): Ports [22, 3389, 445, 5900, 11434], Services: ['SSH', 'VNC', 'SMB/CIFS (Possible Windows/AD)', 'RDP (Windows)', 'Ollama'] - 10.0.0.175 [28:80:88:e4:10:1a] (Netgear_e4:10:1a): Ports [80], Services: ['Web Server'] - 10.0.0.189 [00:17:88:a3:2f:cc] (PhilipsLight_a3:2f:cc): Ports [80, 443], Services: ['Web Server'] - 10.0.0.188 [00:16:6c:c5:1f:54] (SamsungElect_c5:1f:54): Ports [80, 443], Services: ['Web Server'] - 10.0.0.196 [60:5f:8d:88:08:f2] (eero_88:08:f2): Ports [53], Services: ['DNS'] -- 10.0.0.199 [7c:a6:b0:01:a7:33] (7c:a6:b0:01:a7:33): Ports [], Services: [] -- 10.0.0.208 [a0:ad:9f:30:84:5f] (ASUSTekCOMPU_30:84:5f): Ports [], Services: [] -Ollama Instances found at: 10.0.0.31, 10.0.0.204 +- 10.0.0.206 [00:a0:de:a2:0b:ef] (Yamaha_a2:0b:ef): Ports [80], Services: ['Web Server'] +- 10.0.0.22 [Unknown] (Unknown): Ports [22], Services: ['SSH'] +Ollama Instances found at: 10.0.0.39, 10.0.0.204 Ansible Inventory Summary: Inventory directory: /Users/chrisfu/dev/prole/infrastructure diff --git a/scripts/reset_clusters.sh b/scripts/reset_clusters.sh index 4ca7012..2d34f4c 100755 --- a/scripts/reset_clusters.sh +++ b/scripts/reset_clusters.sh @@ -7,7 +7,7 @@ # # Solution: Delete both, recreate with pd-standard boot disks: # knoe-dev-0 → Standard (app workloads: GitLab, platform), e2-small × 3, VPA, pd-standard boot -# knoe-cnpg-0 → Standard (CNPG only, 3× e2-small), pd-standard boot +# knoe-cnpg-0 → Standard (CNPG only, 3× e2-standard-2), pd-standard boot # # SSD quota budget after reset: # Boot disks: pd-standard — does NOT count against SSD_TOTAL_GB @@ -23,7 +23,7 @@ # GCP_REGION (default: us-west3) # APP_CLUSTER (default: knoe-dev-0) # DB_CLUSTER (default: knoe-cnpg-0) -# DB_MACHINE_TYPE (default: e2-small) +# DB_MACHINE_TYPE (default: e2-standard-2) # DB_DISK_TYPE (default: pd-standard) # DB_DISK_SIZE_GB (default: 50) # DB_NODES_PER_ZONE (default: 1 → 3 nodes across 3 zones) @@ -44,7 +44,7 @@ GCP_PROJECT="${GCP_PROJECT:-plenary-truck-485623-p7}" GCP_REGION="${GCP_REGION:-us-west3}" APP_CLUSTER="${APP_CLUSTER:-knoe-dev-0}" DB_CLUSTER="${DB_CLUSTER:-knoe-cnpg-0}" -DB_MACHINE_TYPE="${DB_MACHINE_TYPE:-e2-small}" +DB_MACHINE_TYPE="${DB_MACHINE_TYPE:-e2-standard-2}" DB_DISK_TYPE="${DB_DISK_TYPE:-pd-standard}" DB_DISK_SIZE_GB="${DB_DISK_SIZE_GB:-50}" DB_NODES_PER_ZONE="${DB_NODES_PER_ZONE:-1}" # regional cluster = 3 zones = 3 nodes total @@ -84,6 +84,73 @@ cluster_exists() { --quiet 2>/dev/null | grep -q "${name}" } +wait_for_cluster_absent() { + local name="$1" + local timeout_s="${2:-900}" + local poll_s="${3:-15}" + local elapsed=0 + + while cluster_exists "${name}"; do + if [[ ${elapsed} -ge ${timeout_s} ]]; then + return 1 + fi + log " ${name} still exists; waiting ${poll_s}s (${elapsed}s/${timeout_s}s) ..." + sleep "${poll_s}" + elapsed=$(( elapsed + poll_s )) + done + + return 0 +} + +delete_cluster_until_absent() { + local name="$1" + local max_attempts="${2:-6}" + local retry_sleep_s="${3:-20}" + local attempt=1 + local output="" + local status=0 + + if ! cluster_exists "${name}"; then + log " ${name} not found — skipping." + return 0 + fi + + while [[ ${attempt} -le ${max_attempts} ]]; do + log " Deleting ${name} (attempt ${attempt}/${max_attempts}) ..." + + set +e + output=$(gcloud container clusters delete "${name}" \ + --project="${GCP_PROJECT}" \ + --region="${GCP_REGION}" \ + --quiet 2>&1) + status=$? + set -e + + if [[ ${status} -ne 0 ]]; then + if grep -qi "incompatible operation" <<<"${output}"; then + log " ${name} has an incompatible operation in progress; retrying after ${retry_sleep_s}s." + elif grep -qi "not found" <<<"${output}"; then + log " ${name} already absent." + return 0 + else + printf "%s\n" "${output}" >&2 + die "cluster delete failed for ${name}" + fi + fi + + if wait_for_cluster_absent "${name}" 900 15; then + log " ${name} deletion confirmed." + return 0 + fi + + log " ${name} still present after delete attempt ${attempt}; retrying." + sleep "${retry_sleep_s}" + attempt=$(( attempt + 1 )) + done + + die "timed out deleting ${name} after ${max_attempts} attempts" +} + _ssd_quota_yaml() { # Emit the 3-line YAML block for SSD_TOTAL_GB quota entry, e.g.: # - limit: 300.0 @@ -125,7 +192,7 @@ if [[ "${CONFIRM}" != "true" ]]; then exit 1 fi -log "==> Cluster reset: ${APP_CLUSTER} (Autopilot) + ${DB_CLUSTER} (Standard)" +log "==> Cluster reset: ${APP_CLUSTER} (Standard) + ${DB_CLUSTER} (Standard)" log " Project : ${GCP_PROJECT}" log " Region : ${GCP_REGION}" log " DRY_RUN : ${DRY_RUN}" @@ -157,23 +224,17 @@ for cluster in "${DB_CLUSTER}" "${APP_CLUSTER}"; do if [[ "${DRY_RUN}" == "true" ]]; then echo "[DRY-RUN] gcloud container clusters delete ${cluster} --project=${GCP_PROJECT} --region=${GCP_REGION} --quiet" else - if cluster_exists "${cluster}"; then - log " Deleting ${cluster} (this takes ~5-10 min) ..." - gcloud container clusters delete "${cluster}" \ - --project="${GCP_PROJECT}" \ - --region="${GCP_REGION}" \ - --quiet & - log " ${cluster} deletion running in background (PID $!)." - else - log " ${cluster} not found — skipping." - fi + delete_cluster_until_absent "${cluster}" fi done -# Wait for all background deletions to finish +# Confirm all target clusters are absent before proceeding if [[ "${DRY_RUN}" != "true" ]]; then log " Waiting for cluster deletions to complete ..." - wait + for cluster in "${DB_CLUSTER}" "${APP_CLUSTER}"; do + wait_for_cluster_absent "${cluster}" 900 15 \ + || die "cluster ${cluster} still exists after deletion phase" + done log " All deletions complete." fi @@ -265,6 +326,7 @@ run_cmd gcloud container clusters create "${DB_CLUSTER}" \ --region="${GCP_REGION}" \ --cluster-version=latest \ --machine-type="${DB_MACHINE_TYPE}" \ + --node-labels="workload=db" \ --disk-type="${DB_DISK_TYPE}" \ --disk-size="${DB_DISK_SIZE_GB}" \ --num-nodes="${DB_NODES_PER_ZONE}" \ @@ -333,10 +395,48 @@ else fi # --------------------------------------------------------------------------- -# Phase 7: Show final SSD quota +# Phase 7: Register clusters in knoe-0 fleet + enable service mesh +# --------------------------------------------------------------------------- +# Cloud Service Mesh (CSM) is enabled per-cluster via fleet membership. +# This automates what was previously done via the GCP web console. + +log "[phase 7] Registering clusters in fleet and enabling service mesh ..." + +_register_and_mesh() { + local cluster_name="$1" + log " Registering ${cluster_name} in fleet ..." + if ! run_cmd gcloud container fleet memberships register "${cluster_name}" \ + --gke-cluster="${GCP_REGION}/${cluster_name}" \ + --enable-workload-identity \ + --project="${GCP_PROJECT}" \ + --quiet 2>&1; then + log " WARNING: Fleet registration for ${cluster_name} failed (may already be registered — continuing)." + fi + + log " Enabling automatic service mesh management for ${cluster_name} ..." + if ! run_cmd gcloud container fleet mesh update \ + --management=automatic \ + --memberships="${cluster_name}" \ + --project="${GCP_PROJECT}" \ + --quiet 2>&1; then + log " WARNING: Service mesh update for ${cluster_name} failed — enable manually via GCP console." + fi +} + +if [[ "${DRY_RUN}" != "true" ]]; then + _register_and_mesh "${DB_CLUSTER}" + _register_and_mesh "${APP_CLUSTER}" + log " Fleet + mesh registration submitted. Mesh provisioning is async (~10 min)." + log " Check status: gcloud container fleet mesh describe --project=${GCP_PROJECT}" +else + log " [DRY-RUN] Would register ${DB_CLUSTER} and ${APP_CLUSTER} in fleet + enable service mesh." +fi + +# --------------------------------------------------------------------------- +# Phase 8: Show final SSD quota # --------------------------------------------------------------------------- -log "[phase 7] Final SSD quota ..." +log "[phase 8] Final SSD quota ..." if [[ "${DRY_RUN}" != "true" ]]; then usage=$(ssd_usage_gb) limit=$(ssd_limit_gb) @@ -351,7 +451,7 @@ echo "" log "==> Reset complete." log "" log "Contexts:" -log " App (Autopilot): ${app_ctx}" +log " App (Standard) : ${app_ctx}" log " DB (Standard) : ${db_ctx}" log "" log "Next steps:" @@ -365,3 +465,6 @@ log "" log " 3. Check your quota increase request:" log " gcloud compute regions describe ${GCP_REGION} --project=${GCP_PROJECT} \\" log " --format='table(quotas.metric,quotas.usage,quotas.limit)' | grep SSD" +log "" +log " 4. Check service mesh provisioning status (~10 min after cluster creation):" +log " gcloud container fleet mesh describe --project=${GCP_PROJECT}" diff --git a/tests/etc/test_status_common_services_k8s_skips_opentofu.sh b/tests/etc/test_status_common_services_k8s_skips_opentofu.sh new file mode 100644 index 0000000..5d8cc99 --- /dev/null +++ b/tests/etc/test_status_common_services_k8s_skips_opentofu.sh @@ -0,0 +1,167 @@ +#!/usr/bin/env bash +set -euo pipefail + +SCRIPT_DIR=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd) +PROLE_HOME_REPO=$(cd "$SCRIPT_DIR/../.." && pwd) +SCRIPT_UNDER_TEST="$PROLE_HOME_REPO/etc/status_common_services.sh" + +TMP_DIR=$(mktemp -d) +trap 'rm -rf "$TMP_DIR"' EXIT +export TMP_DIR + +BIN_DIR="$TMP_DIR/bin" +mkdir -p "$BIN_DIR" + +cat <<'K_EOF' >"$BIN_DIR/kubectl" +#!/usr/bin/env bash +set -euo pipefail + +_log_file="${TMP_DIR}/mock_calls.log" +args="$*" +echo "$args" >>"${_log_file}" + +if [[ "$args" == *"opentofu"* ]]; then + echo "$args" >>"${TMP_DIR}/opentofu_calls.log" + echo "Error from server (NotFound): opentofu removed" >&2 + exit 1 +fi + +if [[ "$args" == "config current-context" ]]; then + echo "gke_plenary-truck-485623-p7_us-west3_knoe-dev-0" + exit 0 +fi + +if [[ "$args" == "config view --minify -o jsonpath={.clusters[0].cluster.server}" ]]; then + echo "https://34.106.133.94" + exit 0 +fi + +if [[ "$args" == "-n test-ns get svc garage openbao" ]]; then + cat <<'EOF' +NAME TYPE CLUSTER-IP EXTERNAL-IP PORT(S) AGE +garage ClusterIP 34.118.228.37 3900/TCP,3901/TCP,3903/TCP 30m +openbao ClusterIP 34.118.231.12 8200/TCP 31m +EOF + exit 0 +fi + +if [[ "$args" == "-n test-ns get svc knoe-svc-kong" ]]; then + echo "NAME TYPE CLUSTER-IP EXTERNAL-IP PORT(S) AGE" + echo "knoe-svc-kong ClusterIP 34.118.226.103 8000/TCP 26m" + exit 0 +fi + +if [[ "$args" == "-n cert-manager get svc cert-manager cert-manager-webhook" ]]; then + echo "NAME TYPE CLUSTER-IP EXTERNAL-IP PORT(S) AGE" + echo "cert-manager ClusterIP 34.118.236.229 9402/TCP 30m" + echo "cert-manager-webhook ClusterIP 34.118.232.148 443/TCP 30m" + exit 0 +fi + +if [[ "$args" == "-n test-ns get statefulset openbao" ]]; then + echo "NAME READY AGE" + echo "openbao 1/1 31m" + exit 0 +fi + +if [[ "$args" == "-n test-ns get statefulset garage" ]]; then + echo "NAME READY AGE" + echo "garage 1/1 30m" + exit 0 +fi + +if [[ "$args" == "-n test-ns get deploy knoe-svc-kong" ]]; then + echo "NAME READY UP-TO-DATE AVAILABLE AGE" + echo "knoe-svc-kong 1/1 1 1 27m" + exit 0 +fi + +if [[ "$args" == "-n cert-manager get deploy cert-manager cert-manager-cainjector cert-manager-webhook" ]]; then + echo "NAME READY UP-TO-DATE AVAILABLE AGE" + echo "cert-manager 1/1 1 1 31m" + echo "cert-manager-cainjector 1/1 1 1 31m" + echo "cert-manager-webhook 1/1 1 1 30m" + exit 0 +fi + +# check_resource queries +if [[ "$args" == "-n test-ns get svc garage" || "$args" == "-n test-ns get svc openbao" || "$args" == "-n test-ns get svc knoe-svc-kong" ]]; then + exit 0 +fi +if [[ "$args" == "-n cert-manager get svc cert-manager" || "$args" == "-n cert-manager get svc cert-manager-webhook" ]]; then + exit 0 +fi +if [[ "$args" == "-n test-ns get statefulset openbao" || "$args" == "-n test-ns get statefulset garage" ]]; then + exit 0 +fi +if [[ "$args" == "-n test-ns get deploy knoe-svc-kong" ]]; then + exit 0 +fi +if [[ "$args" == "-n cert-manager get deploy cert-manager" || "$args" == "-n cert-manager get deploy cert-manager-cainjector" || "$args" == "-n cert-manager get deploy cert-manager-webhook" ]]; then + exit 0 +fi + +if [[ "$args" == "-n test-ns get pods --no-headers" ]]; then + cat <<'EOF' +garage-0 1/1 Running 0 30m +openbao-0 1/1 Running 0 32m +knoe-svc-kong-f685786f-rgtwl 1/1 Running 0 27m +EOF + exit 0 +fi + +if [[ "$args" == "-n cert-manager get pods --no-headers" ]]; then + cat <<'EOF' +cert-manager-7b67f8d7d8-gsf2t 1/1 Running 0 31m +cert-manager-cainjector-76f94d9655-blqd9 1/1 Running 0 31m +cert-manager-webhook-7b86d8b889-l88kr 1/1 Running 0 31m +EOF + exit 0 +fi + +exit 0 +K_EOF +chmod +x "$BIN_DIR/kubectl" + +CFG_DIR="$TMP_DIR/conf" +mkdir -p "$CFG_DIR" + +cat <"$CFG_DIR/prole.cfg" +[User] +NAMESPACE = test-ns +SERVICE_NAMESPACE = test-ns +KONG_NAMESPACE = test-ns +CERT_MANAGER_NAMESPACE = cert-manager + +[Global] +DEPLOYMENT_MODE = k8s +C_EOF + +export PATH="$BIN_DIR:$PATH" + +set +e +KONG_NAME=knoe-svc-kong bash "$SCRIPT_UNDER_TEST" -c "$CFG_DIR/prole.cfg" --mode k8s -n test-ns >"$TMP_DIR/stdout" 2>"$TMP_DIR/stderr" +RC=$? +set -e + +if [[ $RC -ne 0 ]]; then + echo "FAILURE: expected status_common_services.sh to succeed in k8s mode without OpenTofu" >&2 + sed -n '1,240p' "$TMP_DIR/stdout" >&2 || true + sed -n '1,240p' "$TMP_DIR/stderr" >&2 || true + exit 1 +fi + +if [[ -s "$TMP_DIR/opentofu_calls.log" ]]; then + echo "FAILURE: expected no OpenTofu kubectl calls in k8s mode" >&2 + sed -n '1,240p' "$TMP_DIR/opentofu_calls.log" >&2 || true + sed -n '1,240p' "$TMP_DIR/mock_calls.log" >&2 || true + exit 1 +fi + +if ! grep -q "skipping OpenTofu service check" "$TMP_DIR/stdout"; then + echo "FAILURE: expected informational OpenTofu skip message" >&2 + sed -n '1,240p' "$TMP_DIR/stdout" >&2 || true + exit 1 +fi + +echo "SUCCESS" diff --git a/tests/installer/test_actions_helpers.py b/tests/installer/test_actions_helpers.py index 6033ae8..5d65ee6 100644 --- a/tests/installer/test_actions_helpers.py +++ b/tests/installer/test_actions_helpers.py @@ -523,7 +523,9 @@ class TestCfgPathVariables: def test_load_inputs_from_cfg_keeps_legacy_system_environment_shell_vars(self, tmp_path): import configparser - cfg_path = tmp_path / "prole.cfg" + cfg_dir = tmp_path / "dev" + cfg_dir.mkdir(parents=True, exist_ok=True) + cfg_path = cfg_dir / "prole.cfg" cfg = configparser.ConfigParser(interpolation=None) cfg.optionxform = str cfg.add_section("System Environment") @@ -612,6 +614,20 @@ class TestScriptEnvForNamespace: assert env.get("CNPG_CLUSTER_NAME") == "knoe-db" assert env.get("CNPG_INSTANCES") == "3" + def test_resolves_db_kubecontext_from_selected_gke_context(self): + inst = _TestableInstaller(inputs={"init_cluster.cluster_env": "prole-prod-cluster"}) + inst.prole_cfg_data.setdefault("Global", {})["KUBECONTEXT"] = ( + "gke_plenary-truck-485623-p7_us-west3_knoe-dev-0" + ) + inst.prole_cfg_data.setdefault("Global", {})["DB_CLUSTER_NAME"] = "knoe-cnpg-0" + + env = inst._script_env_for_namespace("knoe-db-0", cluster_role="db") + + assert ( + env.get("KUBECONTEXT") + == "gke_plenary-truck-485623-p7_us-west3_knoe-cnpg-0" + ) + # --------------------------------------------------------------------------- # ensure_db_k8s_secrets @@ -810,6 +826,56 @@ class TestStepInitScripts: assert inst.prole_cfg_data["Initialization Scripts"]["STATUS"] == "Completed" assert inst._scripts_success is True + def test_verifies_critical_secrets_with_db_env(self, monkeypatch): + import knoe.core.actions as actions_mod + + inst = _TestableSilentInstaller( + inputs={ + "init_cluster.cluster_env": "service", + "init_password.db_namespace": "default", + "init_password.db_password": "pw", + "kerberos_config.enabled": "false", + "init_scripts.run_scripts": "true", + } + ) + + inst.controller.run_script.return_value = 0 + monkeypatch.setattr(inst, "_deployment_mode", lambda: "k3s") + + db_env = { + "NAMESPACE": "default", + "DATABASE_NAMESPACE": "default", + "CLUSTER_NAME": "knoe-db", + } + app_env = {"NAMESPACE": "knoe-system"} + + def _fake_script_env(_ns, cluster_role="db"): + return db_env if cluster_role == "db" else app_env + + monkeypatch.setattr(inst, "_script_env_for_namespace", _fake_script_env) + monkeypatch.setattr(inst, "ensure_db_k8s_secrets", lambda *a, **k: None) + monkeypatch.setattr(inst, "_ensure_cnpg_storage_provisioned", lambda *a, **k: None) + monkeypatch.setattr(inst, "_optional_workloads_policy", lambda _env: (False, 0, "skipped")) + monkeypatch.setattr(actions_mod, "cnpg_initialize", lambda **_kw: None) + monkeypatch.setattr(actions_mod.registry_ops, "update", lambda **_kw: None) + monkeypatch.setattr(actions_mod.openbao_ops, "update", lambda **_kw: None) + monkeypatch.setattr(actions_mod.garage_store_ops, "update", lambda **_kw: None) + monkeypatch.setattr(actions_mod.opentofu_ops, "update", lambda **_kw: None) + + secret_calls = [] + + def _fake_run_cmd(cmd, cwd=None, env=None, **_kwargs): + if isinstance(cmd, list) and cmd[:3] == ["kubectl", "get", "secret"]: + secret_calls.append((cmd, env)) + return 0 + + monkeypatch.setattr(inst, "_run_cmd", _fake_run_cmd, raising=False) + + inst._step_init_scripts() + + assert len(secret_calls) == 3 + assert all(call_env is db_env for _, call_env in secret_calls) + # --------------------------------------------------------------------------- # _step_cnpg_deploy @@ -817,7 +883,7 @@ class TestStepInitScripts: class TestStepCnpgDeploy: - def test_provisions_storage_before_cnpg_deploy(self, monkeypatch): + def test_deploy_invokes_cnpg_with_db_env(self, monkeypatch): import knoe.core.actions as actions_mod inst = _TestableSilentInstaller( @@ -830,31 +896,24 @@ class TestStepCnpgDeploy: inst.prole_cfg_data.setdefault("Deployment", {}) env = {"CNPG_CLUSTER_NAME": "cluster-a"} - monkeypatch.setattr(inst, "_script_env_for_namespace", lambda _ns: env) + monkeypatch.setattr( + inst, + "_script_env_for_namespace", + lambda _ns, cluster_role="db": env, + ) - calls = {"provision": 0, "deploy": 0} - - def _fake_provision(ns, env_arg): - calls["provision"] += 1 - assert ns == "knoe-db" - assert env_arg is env - env_arg["CNPG_DATA_SELECTOR_JSON"] = '{"synology.storage/role":"data"}' - env_arg["CNPG_WAL_SELECTOR_JSON"] = '{"synology.storage/role":"wal"}' + calls = {"deploy": 0} def _fake_cnpg_deploy(namespace, cluster_name, env=None, **_kwargs): calls["deploy"] += 1 assert namespace == "knoe-db" assert cluster_name == "cluster-a" - assert env is not None - assert env["CNPG_DATA_SELECTOR_JSON"] - assert env["CNPG_WAL_SELECTOR_JSON"] + assert env is env - monkeypatch.setattr(inst, "_ensure_cnpg_storage_provisioned", _fake_provision) monkeypatch.setattr(actions_mod, "cnpg_deploy", _fake_cnpg_deploy) inst._step_cnpg_deploy() - assert calls["provision"] == 1 assert calls["deploy"] == 1 assert inst._cnpg_success is True assert inst.prole_cfg_data["Deployment"]["STATUS"] == "Deployed" @@ -872,7 +931,7 @@ class TestStepCnpgDeploy: monkeypatch.setattr( inst, "_script_env_for_namespace", - lambda _ns: {"CNPG_CLUSTER_NAME": "cluster-a"}, + lambda _ns, cluster_role="db": {"CNPG_CLUSTER_NAME": "cluster-a"}, ) monkeypatch.setattr( inst, diff --git a/tests/installer/test_cnpg_barman_webhook.py b/tests/installer/test_cnpg_barman_webhook.py index b1bca09..aaf7c0b 100644 --- a/tests/installer/test_cnpg_barman_webhook.py +++ b/tests/installer/test_cnpg_barman_webhook.py @@ -142,8 +142,8 @@ def test_install_barman_plugin_succeeds_after_cert_manager_recovery(monkeypatch) _apply_calls = [0] _cert_mgr_called = [False] - def _fake_run(args, **kwargs): - if args[:2] == ["kubectl", "apply"]: + def _fake_kubectl(args, **_kwargs): + if args[:2] == ["apply", "-f"]: _apply_calls[0] += 1 if _apply_calls[0] == 1: return SimpleNamespace( @@ -152,12 +152,9 @@ def test_install_barman_plugin_succeeds_after_cert_manager_recovery(monkeypatch) stderr="webhook.cert-manager.io: failed to call webhook: tls: failed to verify certificate: x509: certificate signed by unknown authority", ) return SimpleNamespace(returncode=0, stdout="configured", stderr="") - # CA bundle poll - if "jsonpath={.webhooks[0].clientConfig.caBundle}" in args: - return SimpleNamespace(returncode=0, stdout="dGVzdA==", stderr="") return SimpleNamespace(returncode=0, stdout="", stderr="") - monkeypatch.setattr(cloudnative_pg.subprocess, "run", _fake_run) + monkeypatch.setattr(cloudnative_pg, "_kubectl", _fake_kubectl) monkeypatch.setattr(cloudnative_pg.time, "sleep", lambda _: None) def _fake_ensure(**_kw): @@ -165,7 +162,6 @@ def test_install_barman_plugin_succeeds_after_cert_manager_recovery(monkeypatch) return True # CA bundle confirmed ready monkeypatch.setattr(cloudnative_pg, "_ensure_cert_manager_for_barman", _fake_ensure) - monkeypatch.setattr(cloudnative_pg, "_kubectl", lambda *_a, **_kw: SimpleNamespace(returncode=0, stdout="", stderr="")) monkeypatch.setattr(cloudnative_pg, "_wait_for_barman_tls_secrets", lambda **_kw: True) env = {"BARMAN_PLUGIN_MANIFEST_URL": "https://example.com/barman.yaml"} @@ -181,8 +177,8 @@ def test_install_barman_plugin_logs_warn_when_ca_not_confirmed(monkeypatch): logs: list[str] = [] _apply_calls = [0] - def _fake_run(args, **kwargs): - if args[:2] == ["kubectl", "apply"]: + def _fake_kubectl(args, **_kwargs): + if args[:2] == ["apply", "-f"]: _apply_calls[0] += 1 if _apply_calls[0] == 1: return SimpleNamespace( @@ -191,10 +187,9 @@ def test_install_barman_plugin_logs_warn_when_ca_not_confirmed(monkeypatch): return SimpleNamespace(returncode=0, stdout="ok", stderr="") return SimpleNamespace(returncode=0, stdout="", stderr="") - monkeypatch.setattr(cloudnative_pg.subprocess, "run", _fake_run) + monkeypatch.setattr(cloudnative_pg, "_kubectl", _fake_kubectl) monkeypatch.setattr(cloudnative_pg.time, "sleep", lambda _: None) monkeypatch.setattr(cloudnative_pg, "_ensure_cert_manager_for_barman", lambda **_kw: False) - monkeypatch.setattr(cloudnative_pg, "_kubectl", lambda *_a, **_kw: SimpleNamespace(returncode=0, stdout="", stderr="")) monkeypatch.setattr(cloudnative_pg, "_wait_for_barman_tls_secrets", lambda **_kw: True) env = {"BARMAN_PLUGIN_MANIFEST_URL": "https://example.com/barman.yaml"} @@ -207,19 +202,20 @@ def test_install_barman_plugin_logs_warn_when_ca_not_confirmed(monkeypatch): def test_install_barman_plugin_raises_after_all_retries_exhausted(monkeypatch): """RuntimeError is raised if all 6 attempts fail with webhook CA errors.""" monkeypatch.setattr( - cloudnative_pg.subprocess, "run", - lambda *_a, **_kw: SimpleNamespace( + cloudnative_pg, + "_kubectl", + lambda args, **_kw: SimpleNamespace( returncode=1, stdout="", stderr="x509: certificate signed by unknown authority", - ), + ) if args[:2] == ["apply", "-f"] else SimpleNamespace(returncode=0, stdout="", stderr=""), ) monkeypatch.setattr(cloudnative_pg.time, "sleep", lambda _: None) monkeypatch.setattr( cloudnative_pg, "_ensure_cert_manager_for_barman", lambda **_kw: False, ) - monkeypatch.setattr(cloudnative_pg, "_kubectl", lambda *_a, **_kw: SimpleNamespace(returncode=0, stdout="", stderr="")) + monkeypatch.setattr(cloudnative_pg, "_wait_for_barman_tls_secrets", lambda **_kw: True) env = {"BARMAN_PLUGIN_MANIFEST_URL": "https://example.com/barman.yaml"} with pytest.raises(RuntimeError, match="Failed to apply Barman Cloud plugin after cert-manager restart"): @@ -372,6 +368,34 @@ def test_bootstrap_barman_tls_secrets_raises_on_apply_failure(monkeypatch): cloudnative_pg._bootstrap_barman_tls_secrets(env=None, log=None) +# --------------------------------------------------------------------------- +# _bootstrap_db_user_secrets +# --------------------------------------------------------------------------- +def test_bootstrap_db_user_secrets_honors_kubecontext(monkeypatch): + commands: list[list[str]] = [] + + def _fake_run(args, **_kwargs): + commands.append(list(args)) + if "get" in args and "secret" in args: + return SimpleNamespace(returncode=1, stdout="", stderr="not found") + if "--dry-run=client" in args: + return SimpleNamespace(returncode=0, stdout="apiVersion: v1\nkind: Secret\n", stderr="") + if "apply" in args: + return SimpleNamespace(returncode=0, stdout="created", stderr="") + return SimpleNamespace(returncode=0, stdout="", stderr="") + + monkeypatch.setattr(cloudnative_pg.subprocess, "run", _fake_run) + + cloudnative_pg._bootstrap_db_user_secrets( + namespace="knoe-db-0", + env={"KUBECONTEXT": "gke-db-context", "DB_PASSWORD": "pw"}, + log=None, + ) + + assert commands + assert all(cmd[:3] == ["kubectl", "--context", "gke-db-context"] for cmd in commands) + + # --------------------------------------------------------------------------- # _bootstrap_cnpg_admin_key # --------------------------------------------------------------------------- @@ -407,6 +431,31 @@ def test_bootstrap_cnpg_admin_key_creates_secret_when_missing(monkeypatch): assert any("Created secret 'cnpg-admin-key'" in m for m in logs) +def test_bootstrap_cnpg_admin_key_honors_kubecontext(monkeypatch): + commands: list[list[str]] = [] + + def _fake_run(args, **_kwargs): + commands.append(list(args)) + if "get" in args and "secret" in args: + return SimpleNamespace(returncode=1, stdout="", stderr="not found") + if "--dry-run=client" in args: + return SimpleNamespace(returncode=0, stdout="apiVersion: v1\nkind: Secret\n", stderr="") + if "apply" in args: + return SimpleNamespace(returncode=0, stdout="created", stderr="") + return SimpleNamespace(returncode=0, stdout="", stderr="") + + monkeypatch.setattr(cloudnative_pg.subprocess, "run", _fake_run) + + cloudnative_pg._bootstrap_cnpg_admin_key( + namespace="knoe-db-0", + env={"KUBECONTEXT": "gke-db-context"}, + log=None, + ) + + assert commands + assert all(cmd[:3] == ["kubectl", "--context", "gke-db-context"] for cmd in commands) + + def test_bootstrap_cnpg_admin_key_raises_on_apply_failure(monkeypatch): """Raises RuntimeError when kubectl apply fails.""" def _fake_run(args, **kwargs): @@ -633,6 +682,74 @@ def test_remove_legacy_barman_object_store_skips_when_plugin_not_present(monkeyp assert not any("patch" in args for args in calls) +def test_wait_cnpg_pods_relaxes_workload_selector_when_no_matching_nodes(monkeypatch): + calls: list[list[str]] = [] + readiness_polls = 0 + + def _fake_kubectl(args, env=None, timeout=30, check=False): + nonlocal readiness_polls + calls.append(list(args)) + + if "jsonpath={.spec.instances}" in args: + return SimpleNamespace(returncode=0, stdout="3", stderr="") + + if "jsonpath={range .items[*]}{.metadata.name}={.status.conditions[?(@.type==\"Ready\")].status}\\n{end}" in args: + readiness_polls += 1 + if readiness_polls == 1: + return SimpleNamespace( + returncode=0, + stdout="knoe-db-1=False\\nknoe-db-2=False\\nknoe-db-3=False\\n", + stderr="", + ) + return SimpleNamespace( + returncode=0, + stdout="knoe-db-1=True\\nknoe-db-2=True\\nknoe-db-3=True\\n", + stderr="", + ) + + if "jsonpath={range .items[*]}{.metadata.name}|phase=" in args: + return SimpleNamespace( + returncode=0, + stdout="knoe-db-1|phase=Pending|node=|ready=False|reason=Unschedulable\\n", + stderr="", + ) + + if "get" in args and "events" in args: + return SimpleNamespace( + returncode=0, + stdout="knoe-db-1|0/3 nodes are available: 3 node(s) didn't match Pod's node affinity/selector.\\n", + stderr="", + ) + + if "jsonpath={.spec.affinity.nodeSelector.workload}" in args: + return SimpleNamespace(returncode=0, stdout="db", stderr="") + + if args[:2] == ["get", "nodes"]: + return SimpleNamespace(returncode=0, stdout="", stderr="") + + if "patch" in args and "cluster" in args: + return SimpleNamespace(returncode=0, stdout="patched", stderr="") + + return SimpleNamespace(returncode=0, stdout="", stderr="") + + monotonic_values = iter([0.0, 1.1, 1.2, 1.3, 1.4]) + monkeypatch.setattr(cloudnative_pg, "_kubectl", _fake_kubectl) + monkeypatch.setattr(cloudnative_pg.time, "monotonic", lambda: next(monotonic_values)) + monkeypatch.setattr(cloudnative_pg.time, "sleep", lambda _: None) + + cloudnative_pg._wait_cnpg_pods( + namespace="knoe-db-0", + cluster_name="knoe-db", + env={"CNPG_SELECTOR_RELAX_GRACE_TIMEOUT": "60"}, + timeout=1, + log=None, + ) + + patch_calls = [c for c in calls if "patch" in c and "cluster" in c] + assert len(patch_calls) == 1 + assert '"nodeSelector": null' in patch_calls[0][-1] + + def test_init_cnpg_backup_cleans_legacy_backup_fields_before_plugin_enable(): script = (Path(__file__).resolve().parents[2] / "etc" / "init_cnpg_backup.sh").read_text(encoding="utf-8") diff --git a/tests/installer/test_cnpg_gke_autopilot_init.py b/tests/installer/test_cnpg_gke_autopilot_init.py index c7dfca1..d016fc4 100644 --- a/tests/installer/test_cnpg_gke_autopilot_init.py +++ b/tests/installer/test_cnpg_gke_autopilot_init.py @@ -8,6 +8,11 @@ def test_gke_cnpg_manifest_uses_autopilot_baseline_profile() -> None: manifest = (REPO_ROOT / "deploy" / "gcp" / "gke" / "knoe-db.yaml").read_text(encoding="utf-8") assert "podAntiAffinityType: preferred" in manifest + assert " nodeSelector:\n workload: db" in manifest + assert "\n nodeSelector:\n workload: db\n" not in manifest + assert 'cpu: "100m"' in manifest + assert 'memory: "128Mi"' in manifest + assert 'cpu: "500m"' in manifest assert 'memory: "512Mi"' in manifest assert "metadata:" in manifest and "namespace: knoe-db-0" in manifest assert "CREATE EXTENSION IF NOT EXISTS postgis_topology;" in manifest diff --git a/tests/installer/test_cnpg_tls_bootstrap.py b/tests/installer/test_cnpg_tls_bootstrap.py index dd785b4..d0574cd 100644 --- a/tests/installer/test_cnpg_tls_bootstrap.py +++ b/tests/installer/test_cnpg_tls_bootstrap.py @@ -71,3 +71,28 @@ def test_bootstrap_cnpg_tls_secrets_keeps_fqdn_cn_when_within_limit(monkeypatch) cn = cert.subject.get_attributes_for_oid(NameOID.COMMON_NAME)[0].value assert cn == f"{cluster_name}.{namespace}.svc" assert len(cn) <= 64 + + +def test_bootstrap_cnpg_tls_secrets_honors_kubecontext(monkeypatch): + namespace = "knoe-db-0" + cluster_name = "knoe-db" + commands: list[list[str]] = [] + + monkeypatch.setattr(cloudnative_pg, "_ensure_namespace", lambda *_args, **_kwargs: None) + monkeypatch.setattr(cloudnative_pg, "_kubectl_ok", lambda *_args, **_kwargs: False) + + def _fake_run(args, input=None, **_kwargs): + commands.append(list(args)) + assert input is not None + return SimpleNamespace(returncode=0, stdout="", stderr="") + + monkeypatch.setattr(cloudnative_pg.subprocess, "run", _fake_run) + + cloudnative_pg.bootstrap_cnpg_tls_secrets( + namespace=namespace, + cluster_name=cluster_name, + env={"KUBECONTEXT": "gke-db-context"}, + ) + + assert commands + assert all(cmd[:3] == ["kubectl", "--context", "gke-db-context"] for cmd in commands) diff --git a/tests/installer/test_core_classes.py b/tests/installer/test_core_classes.py index 3af2c32..5006e32 100644 --- a/tests/installer/test_core_classes.py +++ b/tests/installer/test_core_classes.py @@ -3,6 +3,7 @@ from __future__ import annotations import os +import sys from pathlib import Path from unittest import mock @@ -326,7 +327,14 @@ class TestKnoeController: c = KnoeController(tmp_path) installer = KnoeConsoleInstaller(c) - with mock.patch.object(installer, "_load_inputs_from_cfg", return_value={}), \ + with mock.patch.object( + installer, + "_load_inputs_from_cfg", + return_value={ + "init_password.db_password": "pw123", + "init_password.db_password_confirm": "pw123", + }, + ), \ mock.patch.object(installer, "_write_cfg"), \ mock.patch.object(installer, "_perform_cluster_reset"), \ mock.patch.object(installer, "_close_log_file"): @@ -377,6 +385,164 @@ class TestKnoeController: # No early cfg write for generated password; only the normal writes. assert write_cfg.call_count == 2 + def test_silent_installer_fails_when_password_missing_in_non_interactive_mode( + self, tmp_path + ): + c = KnoeController(tmp_path) + installer = KnoeConsoleInstaller(c) + + with ( + mock.patch.object(installer, "_load_inputs_from_cfg", return_value={}), + mock.patch.object( + installer, "_load_db_password_from_ansible_vault", return_value="" + ), + mock.patch.object(installer, "_write_cfg"), + mock.patch.object(installer, "_perform_cluster_reset"), + mock.patch.object(installer, "_close_log_file"), + mock.patch.dict(os.environ, {"CI": "1"}, clear=False), + ): + rc = installer.run() + + assert rc == 2 + + def test_silent_installer_loads_db_password_from_ansible_vault(self, tmp_path): + c = KnoeController(tmp_path) + installer = KnoeConsoleInstaller(c) + + with ( + mock.patch.object(installer, "_load_inputs_from_cfg", return_value={}), + mock.patch.object( + installer, + "_load_db_password_from_ansible_vault", + return_value="vaultpw123", + ), + mock.patch.object(installer, "_write_cfg") as write_cfg, + mock.patch.object(installer, "_perform_cluster_reset"), + mock.patch.object(installer, "_close_log_file"), + ): + captured = {} + + def fake_run_milestones(milestones, progress_callback=None): + captured["ids"] = [m.id for m in milestones] + + c.run_milestones = fake_run_milestones + rc = installer.run() + + assert rc == 0 + assert installer.inputs.get("init_password.db_password") == "vaultpw123" + assert installer.inputs.get("init_password.db_password_confirm") == "vaultpw123" + assert "deployment" in captured["ids"] + assert write_cfg.call_count == 2 + + def test_silent_installer_persists_bootstrap_password_to_ansible_vault(self, tmp_path): + c = KnoeController(tmp_path) + installer = KnoeConsoleInstaller(c) + fake_stdin = mock.Mock() + fake_stdin.isatty.return_value = True + fake_stdout = mock.Mock() + fake_stdout.isatty.return_value = True + env_no_pytest = dict(os.environ) + env_no_pytest.pop("PYTEST_CURRENT_TEST", None) + env_no_pytest.pop("CI", None) + + with ( + mock.patch.object(installer, "_load_inputs_from_cfg", return_value={}), + mock.patch.object( + installer, "_load_db_password_from_ansible_vault", return_value="" + ), + mock.patch.object( + installer, + "_prompt_for_master_password", + return_value="bootstrap_pw_123", + ), + mock.patch.object( + installer, "_persist_db_password_to_ansible_vault" + ) as persist_pw, + mock.patch.object(installer, "_write_cfg") as write_cfg, + mock.patch.object(installer, "_perform_cluster_reset"), + mock.patch.object(installer, "_close_log_file"), + mock.patch.dict("knoe.core.actions.os.environ", env_no_pytest, clear=True), + mock.patch.object(sys, "__stdin__", fake_stdin), + mock.patch.object(sys, "__stdout__", fake_stdout), + ): + c.run_milestones = lambda milestones, progress_callback=None: None + rc = installer.run() + + assert rc == 0 + persist_pw.assert_called_once_with("bootstrap_pw_123") + assert installer.inputs.get("init_password.db_password") == "bootstrap_pw_123" + assert installer.inputs.get("init_password.db_password_confirm") == "bootstrap_pw_123" + assert write_cfg.call_count == 3 + + def test_silent_installer_bootstraps_vault_from_env_password(self, tmp_path): + c = KnoeController(tmp_path) + installer = KnoeConsoleInstaller(c) + + with ( + mock.patch.object( + installer, + "_load_inputs_from_cfg", + return_value={ + "init_password.db_password": "env_pw_123", + "init_password.db_password_confirm": "env_pw_123", + }, + ), + mock.patch.object( + installer, "_load_db_password_from_ansible_vault", return_value="" + ), + mock.patch.object( + installer, "_persist_db_password_to_ansible_vault" + ) as persist_pw, + mock.patch.object(installer, "_write_cfg") as write_cfg, + mock.patch.object(installer, "_perform_cluster_reset"), + mock.patch.object(installer, "_close_log_file"), + mock.patch.dict( + "knoe.core.actions.os.environ", {"KNOE_DB_PASSWORD": "env_pw_123"}, clear=False + ), + ): + c.run_milestones = lambda milestones, progress_callback=None: None + rc = installer.run() + + assert rc == 0 + persist_pw.assert_called_once_with("env_pw_123") + assert write_cfg.call_count == 2 + + def test_silent_installer_prefers_vault_when_env_password_differs(self, tmp_path): + c = KnoeController(tmp_path) + installer = KnoeConsoleInstaller(c) + + with ( + mock.patch.object( + installer, + "_load_inputs_from_cfg", + return_value={ + "init_password.db_password": "env_pw_123", + "init_password.db_password_confirm": "env_pw_123", + }, + ), + mock.patch.object( + installer, + "_load_db_password_from_ansible_vault", + return_value="vault_pw_999", + ), + mock.patch.object( + installer, "_persist_db_password_to_ansible_vault" + ) as persist_pw, + mock.patch.object(installer, "_write_cfg"), + mock.patch.object(installer, "_perform_cluster_reset"), + mock.patch.object(installer, "_close_log_file"), + mock.patch.dict( + "knoe.core.actions.os.environ", {"KNOE_DB_PASSWORD": "env_pw_123"}, clear=False + ), + ): + c.run_milestones = lambda milestones, progress_callback=None: None + rc = installer.run() + + assert rc == 0 + persist_pw.assert_not_called() + assert installer.inputs.get("init_password.db_password") == "vault_pw_999" + assert installer.inputs.get("init_password.db_password_confirm") == "vault_pw_999" + # ===== Milestone concrete subclasses (from core/milestones.py) ===== diff --git a/tests/installer/test_gke_clusters.py b/tests/installer/test_gke_clusters.py index 961025c..e44c45b 100644 --- a/tests/installer/test_gke_clusters.py +++ b/tests/installer/test_gke_clusters.py @@ -55,12 +55,12 @@ def test_get_cluster_credentials_builds_expected_gcloud_command(monkeypatch): ] -def test_ensure_db_cluster_creates_standard_cluster_with_minimum_nodes(monkeypatch): +def test_ensure_db_cluster_creates_standard_cluster_with_db_node_labels(monkeypatch): commands = [] def fake_run(cmd, log=None): commands.append(cmd) - if "describe" in cmd: + if cmd[:4] == ["gcloud", "container", "clusters", "describe"]: return CompletedProcess(cmd, 1, stdout="", stderr="not found") return CompletedProcess(cmd, 0, stdout="created", stderr="") @@ -80,4 +80,41 @@ def test_ensure_db_cluster_creates_standard_cluster_with_minimum_nodes(monkeypat create_cmd = commands[-1] assert create_cmd[:5] == ["gcloud", "container", "clusters", "create", "knoe-cnpg-0"] assert "--num-nodes" in create_cmd - assert create_cmd[create_cmd.index("--num-nodes") + 1] == "3" + assert create_cmd[create_cmd.index("--num-nodes") + 1] == "1" + assert "--node-labels" in create_cmd + assert create_cmd[create_cmd.index("--node-labels") + 1] == "workload=db" + + +def test_ensure_db_cluster_existing_cluster_adds_labeled_db_pool(monkeypatch): + commands = [] + + def fake_run(cmd, log=None): + commands.append(cmd) + if cmd[:4] == ["gcloud", "container", "clusters", "describe"]: + return CompletedProcess(cmd, 0, stdout="knoe-cnpg-0", stderr="") + if cmd[:4] == ["gcloud", "container", "node-pools", "describe"]: + return CompletedProcess(cmd, 1, stdout="", stderr="not found") + return CompletedProcess(cmd, 0, stdout="ok", stderr="") + + monkeypatch.setattr(gke_clusters, "_run", fake_run) + + spec = gke_clusters.GkeClusterSpec( + name="knoe-cnpg-0", + mode="standard", + location="us-central1-a", + machine_type="e2-standard-2", + node_count=3, + node_pool_name="cnpg-db-pool", + ) + + gke_clusters.ensure_db_cluster(project_id="proj-1", spec=spec) + + create_pool_cmd = commands[-1] + assert create_pool_cmd[:4] == ["gcloud", "container", "node-pools", "create"] + assert create_pool_cmd[4] == "cnpg-db-pool" + assert "--machine-type" in create_pool_cmd + assert create_pool_cmd[create_pool_cmd.index("--machine-type") + 1] == "e2-standard-2" + assert "--num-nodes" in create_pool_cmd + assert create_pool_cmd[create_pool_cmd.index("--num-nodes") + 1] == "3" + assert "--node-labels" in create_pool_cmd + assert create_pool_cmd[create_pool_cmd.index("--node-labels") + 1] == "workload=db" diff --git a/tests/installer/test_milestones.py b/tests/installer/test_milestones.py index 998f890..5cb5c55 100644 --- a/tests/installer/test_milestones.py +++ b/tests/installer/test_milestones.py @@ -206,6 +206,7 @@ class TestMilestones(unittest.TestCase): mock_cnpg_init, ): self.state.controller.run_script.return_value = 0 + self.state.inputs["init_cluster.cluster_env"] = "prod" # Mock _run_cmd to return 0 (secrets exist) mock_run_cmd.return_value = 0 mock_opt_allowed.return_value = (True, 1, "ok") @@ -217,7 +218,7 @@ class TestMilestones(unittest.TestCase): mock_registry_update.assert_called() mock_openbao_update.assert_called() mock_garage_update.assert_called() - mock_opentofu_update.assert_called() + mock_opentofu_update.assert_not_called() # CNPG is now Python-owned — assert cnpg_initialize was called mock_cnpg_init.assert_called_once() # monitoring is now Python-owned in milestone flow @@ -228,6 +229,16 @@ class TestMilestones(unittest.TestCase): self.assertNotIn("init_monitoring.sh", called_scripts) self.assertNotIn("init_cloudnative_pg.sh", called_scripts) + secret_calls = [ + c + for c in mock_run_cmd.call_args_list + if isinstance(c.args[0], list) and c.args[0][:3] == ["kubectl", "get", "secret"] + ] + self.assertEqual(len(secret_calls), 3) + for c in secret_calls: + self.assertIn("env", c.kwargs) + self.assertEqual(c.kwargs["env"].get("DATABASE_NAMESPACE"), "default") + @patch("knoe.core.milestones.cnpg_deploy") def test_deployment_milestone(self, mock_cnpg_deploy): milestone = DeploymentMilestone() diff --git a/tests/installer/test_services_init_scripts.py b/tests/installer/test_services_init_scripts.py index cd2824c..672eb56 100644 --- a/tests/installer/test_services_init_scripts.py +++ b/tests/installer/test_services_init_scripts.py @@ -101,6 +101,11 @@ def test_tk_init_scripts_runner_invokes_init_certmgr(tmp_path, monkeypatch): def _deployment_mode(self): return "k3s" + def _cluster_kubecontext(self, cluster_role: str = "app") -> str: + if cluster_role == "db": + return "gke_db_ctx" + return "gke_app_ctx" + def _resolve_prole_logs_dir(self): return tmp_path @@ -247,6 +252,11 @@ def test_tk_init_scripts_common_services_prefers_service_namespace( def _deployment_mode(self): return "k3s" + def _cluster_kubecontext(self, cluster_role: str = "app") -> str: + if cluster_role == "db": + return "gke_db_ctx" + return "gke_app_ctx" + def _resolve_prole_logs_dir(self): return tmp_path @@ -1002,6 +1012,11 @@ def test_tk_init_scripts_k8s_sets_artifact_registry_env_for_cnpg(tmp_path, monke def _deployment_mode(self): return "k8s" + def _cluster_kubecontext(self, cluster_role: str = "app") -> str: + if cluster_role == "db": + return "gke_plenary-truck-485623-p7_us-west3_knoe-cnpg-0" + return "gke_plenary-truck-485623-p7_us-west3_knoe-dev-0" + def _resolve_prole_logs_dir(self): return tmp_path