prole/tests/etc/test_init_cloudnative_pg_instances.sh
chrisfu a069989315 Rename prole-db to knoe-db, add knoe-auth as cluster-internal KDC
Itemized changes:

1. knoe-auth: New cluster-internal KDC and SSO gateway service
   - Created etc/init_knoe_auth.sh based on init_kdc.sh with knoe-auth naming
   - Namespace defaults to SERVICE_NAMESPACE (knoe-system)
   - ConfigMap: knoe-auth-kdc-config, Secret: knoe-auth-secrets
   - Legacy cleanup removes old auth/dog/authority deployments

2. Orchestration: knoe-auth initializes before CloudNativePG
   - Updated prole.sh to insert init_knoe_auth.sh as step 2 (before CNPG)
   - Renumbered all subsequent initialization steps

3. Kong routing: Updated init_kong.sh to route to knoe-auth in SERVICE_NAMESPACE

4. Comment/reference updates for knoe-auth
   - Updated init_common_services.sh, init_service_layer.sh, init_kerberos.sh

5. prole-db renamed to knoe-db across the entire codebase
   - Renamed prole-db/ directory to knoe-db/
   - Renamed all prole-db Kubernetes manifests (deploy/opentofu, k8s/)
   - Renamed scripts: docker-root-knoe-db.sh, docker-run-knoe-db.sh, test-cnpg-knoe-db.sh
   - Renamed etc/init_prole-db-reset.sh to etc/init_knoe-db-reset.sh
   - Renamed etc/prole-db-passwwd.sh to etc/knoe-db-passwwd.sh
   - Renamed mock_val counterparts accordingly
   - Renamed tests/etc/test_init_prole-db-reset.sh to test_init_knoe-db-reset.sh
   - Renamed docs/prole-db-documentation-mcp-architecture.md to knoe-db variant
   - Renamed modes/k3d/prole-db/ to modes/k3d/knoe-db/
   - Renamed prole-db.iml to knoe-db.iml

6. Configuration updates
   - Updated conf/dev, conf/prod, conf/test, conf/service prole.cfg files
   - Updated conf/port-mapping.cfg
   - Updated etc/prole_cfg.sh and mock_val/prole_cfg.sh
   - Updated service/prole.cfg

7. Kubernetes manifests and deploy configuration
   - Updated deploy/opentofu/k3s ArgoCD application YAMLs
   - Updated kong-configmap.yaml and kustomization.yaml
   - Updated k3s/kong-config.yml and prole-resources.yaml
   - Updated prole-mssql-db deployment YAMLs
   - Updated supabase helm render and deploy scripts

8. Infrastructure and GCP Terraform
   - Updated deploy/gcp/terraform: folders, groups, IAM, service-projects

9. Python/installer code updates
   - Updated knoe/core: actions, build_context, controller, env, milestones
   - Updated knoe/milestone.py
   - Updated knoe/ui/screens: cfg, database, database_options, deploy, docker,
     navigation, security, services, validate
   - Updated knoe.spec, status.py

10. Shell script updates
    - Updated etc/: build_db, init_cloudnative_pg, init_cnpg_backup,
      init_db_manager, init_forgejo, init_gitlab, init_monitoring, init_openbao,
      init_port_forwards, init_postgrest, init_supabase_ports, status
    - Updated mock_val/ counterparts for all above scripts
    - Updated prole-net/init-prole-dns.sh
    - Updated bin/prole-kpf.sh, gitea/deploy.sh, supabase/deploy.sh

11. Test updates
    - Updated tests/etc/: test_init_cloudnative_pg*, test_init_cnpg_backup*,
      test_init_kdc*, test_init_kerberos*, test_init_kong*, test_prole_cfg*
    - Updated tests/installer/: test_actions_helpers, test_cfg_save_kubecontext,
      test_controller, test_core_classes, test_milestones, test_milestones_extended,
      test_namespace_propagation
    - Updated tests/: test_database_options, test_navigation,
      test_render_supabase_hostname, test_docker_build_fix,
      test_all_prole_home_fixes, silent_install_test, final_test

12. Documentation updates
    - Updated docs/: DOCKER-BUILD-FIX, PROLE-CFG-SECRETS, PROLE-HOME-DIRECTORY,
      build-system, patent
    - Updated scan/network_description.txt
    - Updated pom.xml

13. Miscellaneous script updates
    - Updated root-level: _adopt_replica_pvcs, _fix_replica_merlin, _import_pi,
      _patch_cluster, _prebind_pvcs, _rebind_d002, _rebind_d002b, test_resolve
    - Updated scripts/generate_spec.py

Co-authored-by: Junie <junie@jetbrains.com>
2026-03-22 22:16:21 -07:00

317 lines
9.8 KiB
Bash

#!/usr/bin/env bash
# Unit test for etc/init_cloudnative_pg.sh
# Verifies CNPG placement policy is label-based (no hostname pinning) and that
# bootstrap reconciles instances safely under degraded startup.
set -euo pipefail
SCRIPT_DIR=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)
PROLE_HOME=$(cd "$SCRIPT_DIR/../.." && pwd)
ETC_DIR="$PROLE_HOME/etc"
SCRIPT_UNDER_TEST="$ETC_DIR/init_cloudnative_pg.sh"
TMP_DIR=$(mktemp -d)
trap 'rm -rf "$TMP_DIR"' EXIT
BIN_DIR="$TMP_DIR/bin"
mkdir -p "$BIN_DIR"
mock_tool() {
cat <<M_EOF > "$BIN_DIR/$1"
#!/usr/bin/env bash
echo "Mocked $1 called with \$@" >> "$TMP_DIR/mock_calls.log"
exit 0
M_EOF
chmod +x "$BIN_DIR/$1"
}
cat <<'K_EOF' > "$BIN_DIR/kubectl"
#!/usr/bin/env bash
set -euo pipefail
_state_file="${TMP_DIR}/mock_cluster_instances"
_log_file="${TMP_DIR}/mock_calls.log"
echo "Mocked kubectl called with $*" >> "${_log_file}"
args="$*"
# API server readiness probes used by wait_for_apiserver_ready()
if [[ "${args}" == *"get --raw=/readyz"* || "${args}" == *"get --raw='/readyz'"* || "${args}" == *"get --raw=\"/readyz\""* ]]; then
echo "ok"
exit 0
fi
if [[ "${args}" == *"version --short"* ]]; then
echo "Client Version: v0.0.0"
echo "Server Version: v0.0.0"
exit 0
fi
# Short-circuit CNPG operator + webhook waits
if [[ "${args}" == *"get deployment"*"-n cnpg-system"*"cnpg-controller-manager"* ]]; then
exit 0
fi
if [[ "${args}" == *"get endpoints"*"cnpg-webhook-service"* ]]; then
echo "10.42.0.10"
exit 0
fi
# Avoid cert-manager installation waits
if [[ "${args}" == *"get crd"*"certificates.cert-manager.io"* ]]; then
exit 0
fi
if [[ "${args}" == *"-n cert-manager"*"get deploy"* ]]; then
exit 0
fi
# Avoid barman plugin waits
if [[ "${args}" == *"get crd"*"objectstores.barmancloud.cnpg.io"* ]]; then
exit 0
fi
if [[ "${args}" == *"-n cnpg-system"*"get secret"*"barman-cloud-client-tls"* ]]; then
exit 0
fi
if [[ "${args}" == *"-n cnpg-system"*"get secret"*"barman-cloud-server-tls"* ]]; then
exit 0
fi
if [[ "${args}" == *"-n cnpg-system"*"get deploy"*"barman-cloud"* ]]; then
exit 0
fi
if [[ "${args}" == *"-n cnpg-system"*"rollout status"*"deploy/barman-cloud"* ]]; then
exit 0
fi
if [[ "${args}" == *"-n cnpg-system"*"get svc"*"-l cnpg.io/pluginName=barman-cloud.cloudnative-pg.io"*"-o jsonpath="*"metadata.name"* ]]; then
echo "barman-cloud"
exit 0
fi
if [[ "${args}" == *"-n cnpg-system"*"get svc barman-cloud"*"-o jsonpath="*"metadata.name"* ]]; then
echo "barman-cloud"
exit 0
fi
if [[ "${args}" == *"-n cnpg-system"*"get svc barman-cloud"*"pluginClientSecret"* ]]; then
echo "barman-cloud-client-tls"
exit 0
fi
if [[ "${args}" == *"-n cnpg-system"*"get svc barman-cloud"*"pluginServerSecret"* ]]; then
echo "barman-cloud-server-tls"
exit 0
fi
if [[ "${args}" == *"-n cnpg-system"*"get svc barman-cloud"*"pluginPort"* ]]; then
echo "9090"
exit 0
fi
if [[ "${args}" == *"-n cnpg-system"*"get endpoints barman-cloud"* ]]; then
echo "10.42.0.11"
exit 0
fi
# Namespace check in ensure_namespace()
if [[ "${args}" == "get namespace test-ns" ]]; then
exit 0
fi
# Capture applied CNPG manifest (stdin or file)
if [[ "${args}" == apply* && "${args}" == *"-n"*"test-ns"* && "${args}" == *" -f "* ]]; then
f=""
if [[ "$#" -ge 2 ]]; then
for ((i=1; i<=$#; i++)); do
if [[ "${!i}" == "-f" ]]; then
j=$((i+1))
f="${!j:-}"
break
fi
done
fi
if [[ "${f}" == "-" ]]; then
cat > "${TMP_DIR}/applied.yaml"
elif [[ -n "${f}" && -f "${f}" ]]; then
cat "${f}" > "${TMP_DIR}/applied.yaml"
else
: > "${TMP_DIR}/applied.yaml"
fi
echo "cluster.postgresql.cnpg.io/knoe-db configured"
exit 0
fi
# Runtime storage validation (k3s) expects live Cluster + PVC/PV objects.
if [[ "${args}" == *"-n test-ns"*"get cluster"*"knoe-db"*"-o json"* ]]; then
cat <<'JSON'
{"apiVersion":"postgresql.cnpg.io/v1","kind":"Cluster","metadata":{"name":"knoe-db"},"spec":{"storage":{"pvcTemplate":{"storageClassName":"synology-iscsi","selector":{"matchLabels":{"synology.storage/role":"data"}}}},"walStorage":{"pvcTemplate":{"storageClassName":"synology-iscsi","selector":{"matchLabels":{"synology.storage/role":"wal"}}}}}}
JSON
exit 0
fi
if [[ "${args}" == *"-n test-ns"*"get pvc"*"cnpg.io/cluster=knoe-db"*"-o json"* ]]; then
cat <<'JSON'
{"items":[
{"metadata":{"name":"knoe-db-1"},"spec":{"storageClassName":"synology-iscsi","volumeName":"pv-data"}},
{"metadata":{"name":"knoe-db-1-wal"},"spec":{"storageClassName":"synology-iscsi","volumeName":"pv-wal"}}
]}
JSON
exit 0
fi
if [[ "${args}" == "get pv pv-data -o json"* ]]; then
echo '{"spec":{"storageClassName":"synology-iscsi","local":{"path":"/synology/d001/data"}}}'
exit 0
fi
if [[ "${args}" == "get pv pv-wal -o json"* ]]; then
echo '{"spec":{"storageClassName":"synology-iscsi","local":{"path":"/synology/d001/wal"}}}'
exit 0
fi
# Node inventory used by reconcile_cnpg_instances
if [[ "${args}" == "get nodes --no-headers" ]]; then
n="${MOCK_READY_NODES:-1}"
for i in $(seq 1 "$n"); do
echo "node-${i} Ready <none> 0d v0"
done
exit 0
fi
if [[ "${args}" == "get nodes -l prole.org/node-role=db --no-headers" ]]; then
n="${MOCK_DB_NODES:-1}"
for i in $(seq 1 "$n"); do
echo "dbnode-${i} Ready <none> 0d v0"
done
exit 0
fi
# Cluster instance state used by reconcile_cnpg_instances
if [[ "${args}" == *"-n test-ns"*"get cluster"*"knoe-db"*"-o jsonpath={.spec.instances}"* ]]; then
if [[ -f "${_state_file}" ]]; then
cat "${_state_file}"
else
echo "3"
fi
exit 0
fi
if [[ "${args}" == *"-n test-ns"*"patch cluster"*"knoe-db"*"\"instances\""* ]]; then
inst=$(printf '%s' "${args}" | sed -n 's/.*"instances":\([0-9][0-9]*\).*/\1/p' | head -n 1)
if [[ -n "${inst:-}" ]]; then
printf '%s' "${inst}" > "${_state_file}"
fi
echo "patched"
exit 0
fi
exit 0
K_EOF
chmod +x "$BIN_DIR/kubectl"
mock_tool curl
mock_tool docker
mock_tool k3d
mock_tool skopeo
mock_tool ansible-playbook
mock_tool tofu
mock_tool terraform
mock_tool ollama
export PATH="$BIN_DIR:$PATH"
export TMP_DIR
mkdir -p "$TMP_DIR/service/secrets"
printf '%s' "dummy-private-key" > "$TMP_DIR/service/secrets/admin.key"
printf '%s' "dummy-public-key" > "$TMP_DIR/service/secrets/admin.pub"
run_case() {
local label="$1"
local cnpg_instances_cfg="${2:-}"
local mock_ready_nodes="${3:-1}"
local mock_db_nodes="${4:-1}"
local conf_dir="$TMP_DIR/conf-${label}"
local env_home="$TMP_DIR/env-${label}"
mkdir -p "$conf_dir"
mkdir -p "$env_home"
cat <<C_EOF > "$conf_dir/prole.cfg"
[User]
NAMESPACE = test-ns
SERVICE_NAMESPACE = test-system
PROLE_HOME = $PROLE_HOME
PROLE_SERVICE = $TMP_DIR/service
[Global]
DEPLOYMENT_MODE = k3s
${cnpg_instances_cfg}
C_EOF
# `etc/prole_cfg.sh` sources `$PROLE_HOME/env.sh` unconditionally if present.
# The repo's generated `env.sh` hard-codes PROLE_CONF to the real conf dir,
# so for tests we provide a minimal env.sh that points PROLE_CONF at our temp config.
cat <<E_EOF > "$env_home/env.sh"
#!/usr/bin/env bash
export PROLE_HOME="$PROLE_HOME"
export PROLE_CONF="$conf_dir"
export PROLE_SERVICE="$TMP_DIR/service"
E_EOF
chmod +x "$env_home/env.sh"
rm -f "$TMP_DIR/applied.yaml"
rm -f "$TMP_DIR/mock_cluster_instances"
set +e
MOCK_READY_NODES="$mock_ready_nodes" MOCK_DB_NODES="$mock_db_nodes" \
BARMAN_CRD_TIMEOUT=5 BARMAN_TLS_TIMEOUT=5 BARMAN_DEPLOY_TIMEOUT=5 BARMAN_SERVICE_TIMEOUT=5 \
CNPG_STORAGE_VALIDATE_TIMEOUT=5 \
PROLE_HOME="$env_home" \
bash "$SCRIPT_UNDER_TEST" --mode k3s start >/dev/null 2>"$TMP_DIR/stderr-${label}"
local rc=$?
set -e
if [[ $rc -ne 0 ]]; then
echo "FAILURE: init_cloudnative_pg.sh returned rc=$rc for case '${label}'"
sed -n '1,200p' "$TMP_DIR/mock_calls.log" || true
sed -n '1,200p' "$TMP_DIR/stderr-${label}" || true
exit 1
fi
if [[ ! -f "$TMP_DIR/applied.yaml" ]]; then
echo "FAILURE: did not capture applied manifest for case '${label}'"
sed -n '1,200p' "$TMP_DIR/mock_calls.log" || true
sed -n '1,200p' "$TMP_DIR/stderr-${label}" || true
exit 1
fi
}
# Default: manifest must be label-based (no hostname pinning) and reconcile must scale down to 1
# when only one node is Ready/schedulable.
run_case "default" "" 1 1
if grep -qE '^\s*- key:\s*kubernetes\\.io/hostname\s*$|myrddin\\.prole\\.org' "$TMP_DIR/applied.yaml"; then
echo "FAILURE: expected no hostname pinning in applied manifest (default)"
sed -n '1,160p' "$TMP_DIR/applied.yaml" || true
exit 1
fi
if ! grep -qE '^\s*- key:\s*node\.kubernetes\.io/instance-type\s*$' "$TMP_DIR/applied.yaml"; then
echo "FAILURE: expected k3s instance-type node affinity in applied manifest (default)"
sed -n '1,160p' "$TMP_DIR/applied.yaml" || true
exit 1
fi
if ! grep -qE '^\s*- k3s\s*$' "$TMP_DIR/applied.yaml"; then
echo "FAILURE: expected node affinity value k3s in applied manifest (default)"
sed -n '1,160p' "$TMP_DIR/applied.yaml" || true
exit 1
fi
if ! grep -qE '^\s*podAntiAffinityType:\s*preferred\s*$' "$TMP_DIR/applied.yaml"; then
echo "FAILURE: expected podAntiAffinityType=preferred in applied manifest (default)"
sed -n '1,160p' "$TMP_DIR/applied.yaml" || true
exit 1
fi
if ! grep -qE '^\s*topologyKey:\s*kubernetes\.io/hostname\s*$' "$TMP_DIR/applied.yaml"; then
echo "FAILURE: expected topologyKey=kubernetes.io/hostname in applied manifest (default)"
sed -n '1,160p' "$TMP_DIR/applied.yaml" || true
exit 1
fi
if [[ "$(cat "$TMP_DIR/mock_cluster_instances" 2>/dev/null || true)" != "1" ]]; then
echo "FAILURE: expected reconcile to patch instances to 1 under degraded startup"
sed -n '1,200p' "$TMP_DIR/mock_calls.log" || true
exit 1
fi
# Override: when capacity allows (2 ready db nodes), CNPG_INSTANCES must be honored (cap=2)
run_case "override" "CNPG_INSTANCES = 2" 2 2
if [[ "$(cat "$TMP_DIR/mock_cluster_instances" 2>/dev/null || true)" != "2" ]]; then
echo "FAILURE: expected reconcile to patch instances to 2 when 2 db nodes are available"
sed -n '1,200p' "$TMP_DIR/mock_calls.log" || true
exit 1
fi
echo "SUCCESS"