From a0cf1e6cefadbee6d7e9ef7d26bb9d11d0483279 Mon Sep 17 00:00:00 2001 From: ArgoCD Setup Date: Sat, 9 May 2026 07:24:26 -0300 Subject: [PATCH] =?UTF-8?q?fix:=20corrigir=20tipos=20de=20servidor=20AMD?= =?UTF-8?q?=20(CPX=E2=86=92CX),=20autoscaler,=20stdin=20e=20cleanup?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - Corrigir tipos Hetzner: CPX não existe, usar CX (Intel shared vCPU) - CX23 (2 vCPU, 4GB) equivalente ao CAX11 - CX33 (4 vCPU, 8GB) equivalente ao CAX21 - CX43 (8 vCPU, 16GB) equivalente ao CAX31 - Corrigir location no autoscaler para amd64 (nbg1→fsn1) - Remover gitlab-pool do autoscaler (substituído pelo Gitea) - Aumentar worker-pool max para 30 - Corrigir Floating IP e Load Balancer hardcoded em nbg1 - Corrigir stdin em read interativo: usar < /dev/tty e read -p - Corrigir cleanup de namespace: lidar com Terminating e finalizers - Corrigir CNPG: nome do deployment, secret e endpointURL duplo - Aulas afetadas: 08, 09, 10, 11, 12, 14, 17 --- aula-08/cleanup.sh | 2 +- aula-08/cluster-autoscaler.yaml | 5 +---- aula-08/main.tf | 8 ++++---- aula-08/setup.sh | 12 ++++++------ aula-08/terraform.tfvars.example | 4 ++-- aula-08/variables.tf | 2 +- aula-09/setup.sh | 18 ++++++------------ aula-10/setup.sh | 21 +++++++-------------- aula-11/cleanup.sh | 8 +++++++- aula-11/setup.sh | 12 +++++++++++- aula-12/cleanup.sh | 24 ++++++++++++++++++++++-- aula-12/setup.sh | 22 +++++++++++++++++++++- aula-14/cleanup.sh | 16 ++++++++++++++-- aula-14/setup.sh | 15 ++++++++++++--- aula-17/README.md | 6 +++--- aula-17/cleanup.sh | 16 ++++++++++++++-- aula-17/cnpg/cluster.yaml | 28 +++++----------------------- aula-17/setup.sh | 26 ++++++++++++++++---------- 18 files changed, 153 insertions(+), 92 deletions(-) diff --git a/aula-08/cleanup.sh b/aula-08/cleanup.sh index a995dad..d0c128d 100755 --- a/aula-08/cleanup.sh +++ b/aula-08/cleanup.sh @@ -47,7 +47,7 @@ if [ -f "kubeconfig" ]; then # Detectar arch dos nodes para filtrar instance type correto NODE_ARCH=$(kubectl get nodes -o jsonpath='{.items[0].metadata.labels.kubernetes\.io/arch}' 2>/dev/null || echo "arm64") if [[ "$NODE_ARCH" == "amd64" ]]; then - INSTANCE_PATTERN="cpx" + INSTANCE_PATTERN="cx" else INSTANCE_PATTERN="cax" fi diff --git a/aula-08/cluster-autoscaler.yaml b/aula-08/cluster-autoscaler.yaml index e343009..acfd8b1 100644 --- a/aula-08/cluster-autoscaler.yaml +++ b/aula-08/cluster-autoscaler.yaml @@ -128,10 +128,7 @@ spec: # POOLS DE NODES: # # worker-pool: Workloads gerais (CAX21 = 4 vCPU, 8GB) - - --nodes=1:3:CAX21:nbg1:worker-pool - # - # gitlab-pool: Gitea e serviços pesados (CAX21) - - --nodes=1:2:CAX21:nbg1:gitlab-pool + - --nodes=1:30:CAX21:nbg1:worker-pool # # build-pool: Builds Docker (CAX31 = 8 vCPU, 16GB) # Escala 0-1 sob demanda, taint "dedicated=builds:NoSchedule" diff --git a/aula-08/main.tf b/aula-08/main.tf index c097cc8..fcad271 100644 --- a/aula-08/main.tf +++ b/aula-08/main.tf @@ -38,14 +38,14 @@ locals { # Endpoint: LoadBalancer IP if enabled, otherwise Floating IP cluster_endpoint_ip = var.enable_loadbalancer ? hcloud_load_balancer.cluster[0].ipv4 : hcloud_floating_ip.control_plane[0].ip_address - # Arquitetura: mapeia arm64→CAX (Nuremberg) ou amd64→CPX (Frankfurt) + # Arquitetura: mapeia arm64→CAX (Nuremberg) ou amd64→CX (Frankfurt) arch_config = { arm64 = { server_type_cp = "cax11" location = "nbg1" } amd64 = { - server_type_cp = "cpx21" + server_type_cp = "cx23" location = "fsn1" } } @@ -247,7 +247,7 @@ resource "hcloud_floating_ip" "control_plane" { count = var.enable_loadbalancer ? 0 : 1 type = "ipv4" name = "${local.cluster_name}-cp-ip" - home_location = "nbg1" + home_location = local.selected_arch.location labels = local.common_labels } @@ -265,7 +265,7 @@ resource "hcloud_load_balancer" "cluster" { count = var.enable_loadbalancer ? 1 : 0 name = "${local.cluster_name}-lb" load_balancer_type = "lb11" - location = "nbg1" + location = local.selected_arch.location labels = local.common_labels } diff --git a/aula-08/setup.sh b/aula-08/setup.sh index ab96655..9af6983 100755 --- a/aula-08/setup.sh +++ b/aula-08/setup.sh @@ -240,13 +240,13 @@ if [ "$SKIP_CREDENTIALS" != "true" ]; then echo "6. Arquitetura dos nodes" echo "" echo " arm64 (CAX) - Ampere Altra, melhor custo/benefício" - echo " amd64 (CPX) - AMD EPYC, mais disponível" + echo " amd64 (CX) - Intel, mais disponível" echo "" read -p " Arquitetura? (arm64/amd64) [arm64]: " arch_input ARCH="${arch_input:-arm64}" if [[ "$ARCH" == "amd64" ]]; then - log_success "AMD64 selecionado (CPX em fsn1)" + log_success "AMD64 selecionado (CX em fsn1)" else log_success "ARM64 selecionado (CAX em nbg1)" ARCH="arm64" @@ -317,8 +317,8 @@ ENABLE_LB_CONFIG=$(grep 'enable_loadbalancer' terraform.tfvars 2>/dev/null | gre # Custos estimados baseado na arquitetura selecionada ARCH_CONFIG=$(grep 'arch' terraform.tfvars 2>/dev/null | grep -o 'arm64\|amd64' || echo "arm64") if [[ "$ARCH_CONFIG" == "amd64" ]]; then - NODE_TYPE="CPX21" - NODE_COST="\$6.99" + NODE_TYPE="CX23" + NODE_COST="\$4.99" else NODE_TYPE="CAX11" NODE_COST="\$4.59" @@ -575,13 +575,13 @@ log_info "Aplicando manifesto do cluster-autoscaler..." # Ajustar server_types do autoscaler conforme arquitetura if [[ "$ARCH" == "amd64" ]]; then AUTOSCALER_FILE="${SCRIPT_DIR}/cluster-autoscaler.yaml" - # CPX equivalentes: CAX21→CPX31, CAX31→CPX41, CAX11→CPX21 + # CX equivalentes: CAX21→CX33, CAX31→CX43, CAX11→CX23 cat "$AUTOSCALER_FILE" | \ sed "s|\${TALOS_IMAGE_ID}|$TALOS_IMAGE_ID|g" | \ sed "s|\${NETWORK_NAME}|$CLUSTER_NAME-network|g" | \ sed "s|\${FIREWALL_NAME}|$CLUSTER_NAME-firewall|g" | \ sed "s|\${SSH_KEY_NAME}|$SSH_KEY_NAME|g" | \ - sed -e 's/CAX21/CPX31/g' -e 's/CAX31/CPX41/g' -e 's/CAX11/CPX21/g' | \ + sed -e 's/CAX21/CX33/g' -e 's/CAX31/CX43/g' -e 's/CAX11/CX23/g' -e 's/:nbg1:/:fsn1:/g' | \ kubectl apply -f - else cat cluster-autoscaler.yaml | \ diff --git a/aula-08/terraform.tfvars.example b/aula-08/terraform.tfvars.example index eb73c1d..d429f87 100644 --- a/aula-08/terraform.tfvars.example +++ b/aula-08/terraform.tfvars.example @@ -24,9 +24,9 @@ talos_image_id = 123456789 # Ambiente (prod, staging, dev) environment = "workshop" -# Arquitetura dos nodes: arm64 (CAX) ou amd64 (CPX) +# Arquitetura dos nodes: arm64 (CAX) ou amd64 (CX) # arm64 = CAX11/21/31 em Nuremberg (nbg1) - melhor custo -# amd64 = CPX21/31/41 em Frankfurt (fsn1) - mais disponível +# amd64 = CX23/33/43 em Frankfurt (fsn1) - mais disponível # arch = "arm64" # Versão do Talos OS (opcional - default: v1.11.2) diff --git a/aula-08/variables.tf b/aula-08/variables.tf index 2f5a4e5..c6c9633 100644 --- a/aula-08/variables.tf +++ b/aula-08/variables.tf @@ -74,7 +74,7 @@ variable "talos_version" { variable "arch" { type = string - description = "Arquitetura dos nodes: arm64 (CAX - Ampere Altra, melhor custo) ou amd64 (CPX - AMD EPYC, mais disponível)" + description = "Arquitetura dos nodes: arm64 (CAX - Ampere Altra, melhor custo) ou amd64 (CX - Intel, mais disponível)" default = "arm64" validation { diff --git a/aula-09/setup.sh b/aula-09/setup.sh index 3cc36dd..bc46627 100755 --- a/aula-09/setup.sh +++ b/aula-09/setup.sh @@ -99,8 +99,7 @@ collect_user_input() { echo -e "Deseja usar esta configuração?" echo -e " 1) Sim, continuar com a configuração existente" echo -e " 2) Não, reconfigurar" - echo -n "Escolha [1/2]: " - read -r choice + read -r -p "Escolha [1/2]: " choice < /dev/tty if [[ "$choice" == "1" ]]; then return 0 fi @@ -108,8 +107,7 @@ collect_user_input() { # Coletar hostname completo echo "" - echo -n "Digite o hostname do n8n (ex: n8n.kube.quest): " - read -r N8N_HOST + read -r -p "Digite o hostname do n8n (ex: n8n.kube.quest): " N8N_HOST < /dev/tty if [[ -z "$N8N_HOST" ]]; then log_error "Hostname não pode ser vazio" @@ -124,8 +122,7 @@ collect_user_input() { echo "Você usa CloudFlare para DNS?" echo " 1) Sim (com proxy/CDN ativado - ícone laranja)" echo " 2) Não" - echo -n "Escolha [1/2]: " - read -r choice + read -r -p "Escolha [1/2]: " choice < /dev/tty if [[ "$choice" == "1" ]]; then USE_CLOUDFLARE=true @@ -139,14 +136,12 @@ collect_user_input() { echo "Deseja ativar HTTPS com Let's Encrypt?" echo " 1) Sim (recomendado para produção)" echo " 2) Não (apenas HTTP)" - echo -n "Escolha [1/2]: " - read -r choice + read -r -p "Escolha [1/2]: " choice < /dev/tty if [[ "$choice" == "1" ]]; then USE_LETSENCRYPT=true echo "" - echo -n "Digite seu email para Let's Encrypt: " - read -r LETSENCRYPT_EMAIL + read -r -p "Digite seu email para Let's Encrypt: " LETSENCRYPT_EMAIL < /dev/tty if [[ -z "$LETSENCRYPT_EMAIL" ]]; then log_error "Email é obrigatório para Let's Encrypt" @@ -349,8 +344,7 @@ if [[ "$USE_LETSENCRYPT" == "true" ]]; then echo -e "${YELLOW} O Let's Encrypt precisa do DNS configurado para emitir o certificado.${NC}" fi echo "" -echo -n "Pressione ENTER quando o DNS estiver configurado..." -read -r +read -p "Pressione ENTER quando o DNS estiver configurado..." -r echo "" diff --git a/aula-10/setup.sh b/aula-10/setup.sh index 4d34a8f..caf71a5 100755 --- a/aula-10/setup.sh +++ b/aula-10/setup.sh @@ -105,8 +105,7 @@ collect_user_input() { echo -e "Deseja usar esta configuração?" echo -e " 1) Sim, continuar com a configuração existente" echo -e " 2) Não, reconfigurar" - echo -n "Escolha [1/2]: " - read -r choice + read -r -p "Escolha [1/2]: " choice < /dev/tty if [[ "$choice" == "1" ]]; then return 0 fi @@ -114,8 +113,7 @@ collect_user_input() { # Coletar hostname echo "" - echo -n "Digite o hostname do Gitea (ex: git.kube.quest): " - read -r GITEA_HOST + read -r -p "Digite o hostname do Gitea (ex: git.kube.quest): " GITEA_HOST < /dev/tty if [[ -z "$GITEA_HOST" ]]; then log_error "Hostname não pode ser vazio" @@ -134,8 +132,7 @@ collect_user_input() { echo -e "Deseja usar esta configuração de TLS?" echo -e " 1) Sim" echo -e " 2) Não, reconfigurar" - echo -n "Escolha [1/2]: " - read -r choice + read -r -p "Escolha [1/2]: " choice < /dev/tty if [[ "$choice" == "1" ]]; then save_config return 0 @@ -147,8 +144,7 @@ collect_user_input() { echo "Você usa CloudFlare para DNS?" echo " 1) Sim (com proxy/CDN ativado - ícone laranja)" echo " 2) Não" - echo -n "Escolha [1/2]: " - read -r choice + read -r -p "Escolha [1/2]: " choice < /dev/tty if [[ "$choice" == "1" ]]; then USE_CLOUDFLARE=true @@ -161,14 +157,12 @@ collect_user_input() { echo "Deseja ativar HTTPS com Let's Encrypt?" echo " 1) Sim (recomendado para produção)" echo " 2) Não (apenas HTTP)" - echo -n "Escolha [1/2]: " - read -r choice + read -r -p "Escolha [1/2]: " choice < /dev/tty if [[ "$choice" == "1" ]]; then USE_LETSENCRYPT=true echo "" - echo -n "Digite seu email para Let's Encrypt: " - read -r LETSENCRYPT_EMAIL + read -r -p "Digite seu email para Let's Encrypt: " LETSENCRYPT_EMAIL < /dev/tty if [[ -z "$LETSENCRYPT_EMAIL" ]]; then log_error "Email é obrigatório para Let's Encrypt" @@ -340,8 +334,7 @@ if [[ "$USE_LETSENCRYPT" == "true" ]]; then echo -e "${YELLOW} O Let's Encrypt precisa do DNS configurado para emitir o certificado.${NC}" fi echo "" -echo -n "Pressione ENTER quando o DNS estiver configurado..." -read -r +read -r -p "Pressione ENTER quando o DNS estiver configurado..." < /dev/tty echo "" diff --git a/aula-11/cleanup.sh b/aula-11/cleanup.sh index 987fa3f..ef34b54 100755 --- a/aula-11/cleanup.sh +++ b/aula-11/cleanup.sh @@ -42,7 +42,13 @@ fi # Remover namespace argocd log_info "Removendo namespace argocd..." -kubectl delete namespace argocd --timeout=60s 2>/dev/null || true +kubectl delete namespace argocd --timeout=120s 2>/dev/null || { + log_warn "Namespace preso em Terminating, forçando remoção..." + kubectl get namespace argocd -o json | \ + python3 -c "import sys,json; d=json.load(sys.stdin); d['metadata'].pop('finalizers',None); json.dump(d,sys.stdout)" | \ + kubectl replace --raw /api/v1/namespaces/argocd/finalize -f - 2>/dev/null || true + kubectl wait --for=delete namespace argocd --timeout=60s 2>/dev/null || sleep 5 +} # Limpar secrets residuais log_info "Limpando secrets residuais..." diff --git a/aula-11/setup.sh b/aula-11/setup.sh index 5f425a5..bb1332b 100755 --- a/aula-11/setup.sh +++ b/aula-11/setup.sh @@ -236,7 +236,17 @@ log_info "=== Instalando ArgoCD ===" helm repo add argo https://argoproj.github.io/argo-helm 2>/dev/null || true helm repo update -kubectl create namespace argocd 2>/dev/null || true +if kubectl get namespace argocd &>/dev/null; then + NS_PHASE=$(kubectl get namespace argocd -o jsonpath='{.status.phase}' 2>/dev/null) + if [[ "$NS_PHASE" == "Terminating" ]]; then + log_info "Namespace argocd em Terminating, forçando remoção..." + kubectl get namespace argocd -o json | \ + python3 -c "import sys,json; d=json.load(sys.stdin); d['metadata'].pop('finalizers',None); json.dump(d,sys.stdout)" | \ + kubectl replace --raw /api/v1/namespaces/argocd/finalize -f - 2>/dev/null || true + kubectl wait --for=delete namespace argocd --timeout=60s 2>/dev/null || sleep 5 + fi +fi +kubectl create namespace argocd --dry-run=client -o yaml | kubectl apply -f - if helm status argocd -n argocd &> /dev/null; then log_warn "ArgoCD já instalado, fazendo upgrade..." diff --git a/aula-12/cleanup.sh b/aula-12/cleanup.sh index 0844a8e..bdf7165 100755 --- a/aula-12/cleanup.sh +++ b/aula-12/cleanup.sh @@ -39,19 +39,39 @@ echo "" # Remover Helm release log_info "Removendo Victoria Metrics Stack..." if helm status monitoring -n monitoring &> /dev/null; then - helm uninstall monitoring -n monitoring --wait 2>/dev/null || true + helm uninstall monitoring -n monitoring --wait --timeout=120s 2>/dev/null || { + log_warn "Helm uninstall falhou, removendo com --no-hooks..." + helm uninstall monitoring -n monitoring --no-hooks 2>/dev/null || true + } log_success "Helm release removido" else log_info "Helm release não encontrado" fi +# Remover finalizers de CRDs do Victoria Metrics (podem prender o namespace) +log_info "Limpando finalizers de CRDs..." +for crd in vmagents vmalerts vmsingles vmclusters vmnodes vmrules vmauths; do + for resource in $(kubectl get $crd.operator.victoriametrics.com -n monitoring -o name 2>/dev/null); do + kubectl patch "$resource" -n monitoring --type=json \ + -p='[{"op":"remove","path":"/metadata/finalizers"}]' 2>/dev/null || true + done +done + # Remover PVCs log_info "Removendo PVCs..." kubectl delete pvc --all -n monitoring --wait=false 2>/dev/null || true # Remover namespace log_info "Removendo namespace..." -kubectl delete namespace monitoring --timeout=60s 2>/dev/null || true +if kubectl get namespace monitoring &>/dev/null; then + kubectl delete namespace monitoring --timeout=120s 2>/dev/null || { + log_warn "Namespace preso em Terminating, forçando remoção..." + kubectl get namespace monitoring -o json | \ + python3 -c "import sys,json; d=json.load(sys.stdin); d['metadata'].pop('finalizers',None); json.dump(d,sys.stdout)" | \ + kubectl replace --raw /api/v1/namespaces/monitoring/finalize -f - 2>/dev/null || true + kubectl wait --for=delete namespace monitoring --timeout=60s 2>/dev/null || sleep 5 + } +fi log_success "Namespace removido" # Remover .env diff --git a/aula-12/setup.sh b/aula-12/setup.sh index 3ee8e41..84d4c90 100755 --- a/aula-12/setup.sh +++ b/aula-12/setup.sh @@ -184,7 +184,27 @@ log_info "=== Instalando Victoria Metrics Stack ===" helm repo add vm https://victoriametrics.github.io/helm-charts/ 2>/dev/null || true helm repo update vm -kubectl create namespace monitoring 2>/dev/null || true +# Garantir namespace limpo +if kubectl get namespace monitoring &>/dev/null; then + NS_PHASE=$(kubectl get namespace monitoring -o jsonpath='{.status.phase}' 2>/dev/null) + if [[ "$NS_PHASE" == "Terminating" ]]; then + log_info "Namespace monitoring em Terminating, limpando finalizers..." + # Remover finalizers de CRDs do Victoria Metrics que prendem o namespace + for crd in vmagents vmalerts vmsingles vmclusters vmnodes vmrules vmauths; do + for resource in $(kubectl get $crd.operator.victoriametrics.com -n monitoring -o name 2>/dev/null); do + kubectl patch "$resource" -n monitoring --type=json \ + -p='[{"op":"remove","path":"/metadata/finalizers"}]' 2>/dev/null || true + done + done + kubectl wait --for=delete namespace monitoring --timeout=60s 2>/dev/null || { + kubectl get namespace monitoring -o json | \ + python3 -c "import sys,json; d=json.load(sys.stdin); d['metadata'].pop('finalizers',None); json.dump(d,sys.stdout)" | \ + kubectl replace --raw /api/v1/namespaces/monitoring/finalize -f - 2>/dev/null || true + kubectl wait --for=delete namespace monitoring --timeout=30s 2>/dev/null || sleep 5 + } + fi +fi +kubectl create namespace monitoring --dry-run=client -o yaml | kubectl apply -f - kubectl label namespace monitoring pod-security.kubernetes.io/enforce=privileged --overwrite if helm status monitoring -n monitoring &> /dev/null; then diff --git a/aula-14/cleanup.sh b/aula-14/cleanup.sh index e407a7a..769d90f 100755 --- a/aula-14/cleanup.sh +++ b/aula-14/cleanup.sh @@ -56,7 +56,13 @@ kubectl delete pod curl-test -n istio --ignore-not-found=true 2>/dev/null || tru # Remover namespace istio (aplicação demo) if kubectl get namespace istio &> /dev/null; then log_info "Removendo namespace istio..." - kubectl delete namespace istio --wait=false 2>/dev/null || true + kubectl delete namespace istio --timeout=120s 2>/dev/null || { + log_warn "Namespace preso em Terminating, forçando remoção..." + kubectl get namespace istio -o json | \ + python3 -c "import sys,json; d=json.load(sys.stdin); d['metadata'].pop('finalizers',None); json.dump(d,sys.stdout)" | \ + kubectl replace --raw /api/v1/namespaces/istio/finalize -f - 2>/dev/null || true + kubectl wait --for=delete namespace istio --timeout=60s 2>/dev/null || sleep 5 + } log_success "Namespace istio marcado para remoção" else log_info "Namespace istio não encontrado" @@ -94,7 +100,13 @@ fi # Remover namespace istio-system if kubectl get namespace istio-system &> /dev/null; then log_info "Removendo namespace istio-system..." - kubectl delete namespace istio-system --wait=false 2>/dev/null || true + kubectl delete namespace istio-system --timeout=120s 2>/dev/null || { + log_warn "Namespace preso em Terminating, forçando remoção..." + kubectl get namespace istio-system -o json | \ + python3 -c "import sys,json; d=json.load(sys.stdin); d['metadata'].pop('finalizers',None); json.dump(d,sys.stdout)" | \ + kubectl replace --raw /api/v1/namespaces/istio-system/finalize -f - 2>/dev/null || true + kubectl wait --for=delete namespace istio-system --timeout=60s 2>/dev/null || sleep 5 + } log_success "Namespace istio-system marcado para remoção" fi diff --git a/aula-14/setup.sh b/aula-14/setup.sh index fcba675..94a1ed6 100755 --- a/aula-14/setup.sh +++ b/aula-14/setup.sh @@ -222,7 +222,17 @@ install_istio() { helm repo add istio https://istio-release.storage.googleapis.com/charts 2>/dev/null || true helm repo update istio - kubectl create namespace istio-system 2>/dev/null || true + if kubectl get namespace istio-system &>/dev/null; then + NS_PHASE=$(kubectl get namespace istio-system -o jsonpath='{.status.phase}' 2>/dev/null) + if [[ "$NS_PHASE" == "Terminating" ]]; then + log_info "Namespace istio-system em Terminating, forçando remoção..." + kubectl get namespace istio-system -o json | \ + python3 -c "import sys,json; d=json.load(sys.stdin); d['metadata'].pop('finalizers',None); json.dump(d,sys.stdout)" | \ + kubectl replace --raw /api/v1/namespaces/istio-system/finalize -f - 2>/dev/null || true + kubectl wait --for=delete namespace istio-system --timeout=60s 2>/dev/null || sleep 5 + fi + fi + kubectl create namespace istio-system --dry-run=client -o yaml | kubectl apply -f - kubectl label namespace istio-system \ pod-security.kubernetes.io/enforce=privileged \ pod-security.kubernetes.io/warn=privileged \ @@ -577,8 +587,7 @@ main() { echo -e "${YELLOW}⚠ Configure o DNS agora antes de continuar.${NC}" fi echo "" - echo -n "Pressione ENTER quando o DNS estiver configurado..." - read -r + read -r -p "Pressione ENTER quando o DNS estiver configurado..." < /dev/tty setup_ingress build_and_push_images diff --git a/aula-17/README.md b/aula-17/README.md index 7fd022f..1ee6b10 100644 --- a/aula-17/README.md +++ b/aula-17/README.md @@ -14,7 +14,7 @@ Nesta aula, vamos instalar o CNPG do zero, criar um cluster HA com backup autom | Item | AWS RDS Multi-AZ | CNPG na Hetzner | |------|------------------|-----------------| -| Compute | db.r6g.large ~$185/mês | 3× CPX31 ~€36/mês (~$40) | +| Compute | db.r6g.large ~$185/mês | 3× CX33 ~€24/mês (~$27) | | Storage 200GB | ~$23/mês (gp3) | ~€9/mês (hcloud volumes) | | Backup | 7 dias (free) | ~€6/mês (Object Storage 1TB) | | **Total mensal** | **~$208/mês** | **~€51/mês (~$56)** | @@ -26,7 +26,7 @@ Nesta aula, vamos instalar o CNPG do zero, criar um cluster HA com backup autom ### Por que Hetzner? -- CPX31: 4 vCPU AMD, 8GB RAM, 160GB NVMe → **~€12/mês** +- CX33: 4 vCPU Intel, 8GB RAM, 80GB NVMe → **~€8/mês** - Object Storage: 1TB S3-compatible, 1TB egress free → **~€6/mês** - Sem surpresas: preço fixo, sem egress cost como AWS ($0.09/GB) @@ -48,7 +48,7 @@ Dois operadores PostgreSQL maduros para Kubernetes. Qual escolher? ### Por que CNPG para indie hackers? -1. **Menos recursos = custo menor**. StackGres consome 3-5× mais RAM em sidecars. Num CPX31 com 8GB, isso faz diferença. +1. **Menos recursos = custo menor**. StackGres consome 3-5× mais RAM em sidecars. Num CX33 com 8GB, isso faz diferença. 2. **Setup mais simples**. Um Cluster CRD resolve. StackGres precisa de SGCluster, SGPostgresConfig, SGPgBouncerConfig, SGBackupConfig, SGPoolingConfig... 3. **CNCF Sandbox**. Governança aberta, roadmap transparente, não depende de uma empresa. 4. **GitOps nativo**. CRDs foram desenhados pra serem declarativos. Funciona com ArgoCD/Flux sem hacks. diff --git a/aula-17/cleanup.sh b/aula-17/cleanup.sh index fc3f2d8..4bc79cd 100755 --- a/aula-17/cleanup.sh +++ b/aula-17/cleanup.sh @@ -73,13 +73,25 @@ log_success "Secrets removidos" # 5. Remover namespace log_info "Removendo namespace cnpg..." -kubectl delete namespace cnpg --ignore-not-found=true 2>/dev/null || true +kubectl delete namespace cnpg --timeout=120s 2>/dev/null || { + log_warn "Namespace preso em Terminating, forçando remoção..." + kubectl get namespace cnpg -o json | \ + python3 -c "import sys,json; d=json.load(sys.stdin); d['metadata'].pop('finalizers',None); json.dump(d,sys.stdout)" | \ + kubectl replace --raw /api/v1/namespaces/cnpg/finalize -f - 2>/dev/null || true + kubectl wait --for=delete namespace cnpg --timeout=60s 2>/dev/null || sleep 5 +} log_success "Namespace cnpg removido" # 6. Remover CNPG Operator log_info "Removendo CNPG operator..." helm uninstall cnpg -n cnpg-system --wait 2>/dev/null || true -kubectl delete namespace cnpg-system --ignore-not-found=true 2>/dev/null || true +kubectl delete namespace cnpg-system --timeout=120s 2>/dev/null || { + log_warn "Namespace preso em Terminating, forçando remoção..." + kubectl get namespace cnpg-system -o json | \ + python3 -c "import sys,json; d=json.load(sys.stdin); d['metadata'].pop('finalizers',None); json.dump(d,sys.stdout)" | \ + kubectl replace --raw /api/v1/namespaces/cnpg-system/finalize -f - 2>/dev/null || true + kubectl wait --for=delete namespace cnpg-system --timeout=60s 2>/dev/null || sleep 5 +} log_success "CNPG operator removido" # 7. Remover CRDs diff --git a/aula-17/cnpg/cluster.yaml b/aula-17/cnpg/cluster.yaml index aa7d994..a890121 100644 --- a/aula-17/cnpg/cluster.yaml +++ b/aula-17/cnpg/cluster.yaml @@ -12,29 +12,15 @@ spec: # topology.kubernetes.io/zone = fsn1 | nbg1 | hel1 (datacenter) # topology.kubernetes.io/region = eu-central | hel-southeast # - # Layer 1: podAntiAffinity on hostname (hard constraint) - # → No 2 PG pods on the same node. Ever. + # podAntiAffinityType: required → no 2 PG pods on same node (hard constraint) + # topologyKey: kubernetes.io/hostname → spread across nodes # - # Layer 2: podAntiAffinity on zone (soft constraint) - # → Prefer spreading across datacenters. With nodes in fsn1, nbg1, hel1, - # you get 1 pod per DC. With only eu-central nodes, they still spread - # across FSN and NBG (both in eu-central, ~1-2ms apart). - # - # Prerequisite for cross-region (eu-central + HEL): - # - Node pool in eu-central (FSN + NBG) → aula-08 - # - Node pool in hel-southeast (HEL) → add worker after aula-08 - # - Without HEL nodes, all 3 pods stay in eu-central (still HA within region) + # For cross-DC spread, CNPG relies on Kubernetes scheduler topology awareness. + # With nodes in multiple zones (fsn1, nbg1, hel1), the scheduler naturally + # spreads pods when combined with the node autoscaler. affinity: topologyKey: kubernetes.io/hostname podAntiAffinityType: required - podAntiAffinity: - preferredDuringSchedulingIgnoredDuringExecution: - - weight: 100 - podAffinityTerm: - topologyKey: topology.kubernetes.io/zone - labelSelector: - matchLabels: - cnpg.io/cluster: shared-postgres storage: size: 10Gi storageClass: hcloud-volumes @@ -60,10 +46,6 @@ spec: initdb: database: app owner: app - secret: - name: shared-postgres-superuser - superuserSecret: - name: shared-postgres-superuser backup: barmanObjectStore: destinationPath: s3://CHANGE_BUCKET_NAME/cnpg-backups/shared-postgres diff --git a/aula-17/setup.sh b/aula-17/setup.sh index 2a686f2..188e896 100755 --- a/aula-17/setup.sh +++ b/aula-17/setup.sh @@ -135,26 +135,22 @@ collect_config() { if [[ -z "$S3_ENDPOINT" ]]; then S3_ENDPOINT="nbg1.your-objectstorage.com" fi - echo -ne "Endpoint (${GREEN}${S3_ENDPOINT}${NC}): " - read -r input + read -r -p "Endpoint (${GREEN}${S3_ENDPOINT}${NC}): " input < /dev/tty [[ -n "$input" ]] && S3_ENDPOINT="$input" if [[ -z "$S3_BUCKET" ]]; then S3_BUCKET="cnpg-backups" fi - echo -ne "Bucket name (${GREEN}${S3_BUCKET}${NC}): " - read -r input + read -r -p "Bucket name (${GREEN}${S3_BUCKET}${NC}): " input < /dev/tty [[ -n "$input" ]] && S3_BUCKET="$input" - echo -ne "Access Key ID: " - read -r S3_ACCESS_KEY + read -r -p "Access Key ID: " S3_ACCESS_KEY < /dev/tty if [[ -z "$S3_ACCESS_KEY" ]]; then log_error "Access Key é obrigatória" exit 1 fi - echo -ne "Secret Access Key: " - read -rs S3_SECRET_KEY + read -rs -p "Secret Access Key: " S3_SECRET_KEY < /dev/tty echo "" if [[ -z "$S3_SECRET_KEY" ]]; then log_error "Secret Key é obrigatória" @@ -191,7 +187,7 @@ install_cnpg_operator() { log_success "CNPG operator instalado" log_info "Aguardando operator ficar ready..." - kubectl wait --for=condition=available deployment/cnpg-controller-manager -n cnpg-system --timeout=120s + kubectl wait --for=condition=available deployment/cnpg-cloudnative-pg -n cnpg-system --timeout=120s log_success "CNPG operator pronto" } @@ -205,6 +201,16 @@ setup_namespace_and_secrets() { echo -e "${CYAN} Criando namespace e secrets${NC}" echo -e "${CYAN}═══════════════════════════════════════════════════════════${NC}" + if kubectl get namespace cnpg &>/dev/null; then + NS_PHASE=$(kubectl get namespace cnpg -o jsonpath='{.status.phase}' 2>/dev/null) + if [[ "$NS_PHASE" == "Terminating" ]]; then + log_info "Namespace cnpg em Terminating, forçando remoção..." + kubectl get namespace cnpg -o json | \ + python3 -c "import sys,json; d=json.load(sys.stdin); d['metadata'].pop('finalizers',None); json.dump(d,sys.stdout)" | \ + kubectl replace --raw /api/v1/namespaces/cnpg/finalize -f - 2>/dev/null || true + kubectl wait --for=delete namespace cnpg --timeout=60s 2>/dev/null || sleep 5 + fi + fi kubectl create namespace cnpg --dry-run=client -o yaml | kubectl apply -f - log_info "Criando secret de credenciais S3..." @@ -228,7 +234,7 @@ create_cluster() { log_info "Gerando cluster.yaml com suas configurações..." sed -e "s|CHANGE_BUCKET_NAME|${S3_BUCKET}|g" \ - -e "s|CHANGE_ENDPOINT|https://${S3_ENDPOINT}|g" \ + -e "s|CHANGE_ENDPOINT|${S3_ENDPOINT}|g" \ "${SCRIPT_DIR}/cnpg/cluster.yaml" | kubectl apply -f - log_success "Cluster aplicado"