fix(cluster): 消除发布前门禁假阴性

发布前巡检在 pipefail 下直接将 K3s readyz 管给 grep -q,可能因 SIGPIPE 误报 etcd 失败;链路和 PostgreSQL 断言失败时也缺少可定位证据。

改为完整读取 readyz,输出六向链路指标,并为复制、归档、备份和配置投影增加明确失败原因;同时缩短 10 包探测间隔,不改变丢包与 RTT 门槛。

验证:bash -n、ShellCheck、生产 precutover 只读巡检。
This commit is contained in:
2026-07-31 23:34:22 +08:00
parent 709434a256
commit 744e4acd1f
2 changed files with 62 additions and 19 deletions
+61 -18
View File
@@ -14,24 +14,35 @@ load_cluster_env
hosts=("$CLUSTER_NINGBO_HOST" "$CLUSTER_HONGKONG_HOST" "$CLUSTER_LOS_ANGELES_HOST")
wireguard_ips=(10.77.0.1 10.77.0.2 10.77.0.3)
site_labels=(ningbo hongkong losangeles)
for source_index in 0 1 2; do
handshakes=$(cluster_ssh "${hosts[$source_index]}" \
"wg show wg0 latest-handshakes | awk '\$2 > 0 && systime() - \$2 < 180 { count++ } END { print count + 0 }'")
[[ $handshakes -eq 2 ]] || {
echo "stale WireGuard handshake on ${hosts[$source_index]}" >&2
echo "stale WireGuard handshake on ${site_labels[$source_index]}: recent_peers=$handshakes" >&2
exit 1
}
for target_index in 0 1 2; do
[[ $source_index -eq $target_index ]] && continue
ping_output=$(cluster_ssh "${hosts[$source_index]}" \
"ping -c 10 -W 2 ${wireguard_ips[$target_index]}")
"ping -i 0.2 -c 10 -W 2 ${wireguard_ips[$target_index]}")
packet_loss=$(sed -nE 's/.* ([0-9.]+)% packet loss.*/\1/p' <<<"$ping_output")
average_rtt=$(sed -nE 's#.* = [0-9.]+/([0-9.]+)/.*#\1#p' <<<"$ping_output")
awk -v loss="$packet_loss" 'BEGIN { exit !(loss < 1) }'
echo "wireguard_path=${site_labels[$source_index]}_to_${site_labels[$target_index]} loss_percent=$packet_loss average_rtt_ms=$average_rtt"
awk -v loss="$packet_loss" 'BEGIN { exit !(loss < 1) }' || {
echo "WireGuard packet loss gate failed on ${site_labels[$source_index]} to ${site_labels[$target_index]}: loss_percent=$packet_loss" >&2
exit 1
}
if [[ $source_index -eq 2 || $target_index -eq 2 ]]; then
awk -v rtt="$average_rtt" 'BEGIN { exit !(rtt < 300) }'
awk -v rtt="$average_rtt" 'BEGIN { exit !(rtt < 300) }' || {
echo "WireGuard RTT gate failed on ${site_labels[$source_index]} to ${site_labels[$target_index]}: average_rtt_ms=$average_rtt limit_ms=300" >&2
exit 1
}
else
awk -v rtt="$average_rtt" 'BEGIN { exit !(rtt < 80) }'
awk -v rtt="$average_rtt" 'BEGIN { exit !(rtt < 80) }' || {
echo "WireGuard RTT gate failed on ${site_labels[$source_index]} to ${site_labels[$target_index]}: average_rtt_ms=$average_rtt limit_ms=80" >&2
exit 1
}
fi
done
done
@@ -41,7 +52,8 @@ cluster_state=$(cluster_ssh "$CLUSTER_NINGBO_HOST" 'bash -s' <<'REMOTE'
set -euo pipefail
kubectl='k3s kubectl'
[[ $($kubectl get nodes --no-headers | awk '$2 == "Ready" { count++ } END { print count + 0 }') -eq 3 ]]
$kubectl get --raw='/readyz?verbose' | grep -q '\[+\]etcd ok'
ready_output=$($kubectl get --raw='/readyz?verbose')
grep -Fq '[+]etcd ok' <<<"$ready_output"
[[ $($kubectl get node easyai-los-angeles -o json |
jq '[.spec.taints[]? | select(.key=="easyai.io/witness" and .effect=="NoSchedule")] | length') -eq 1 ]]
[[ $($kubectl get pods -A -o json |
@@ -56,32 +68,63 @@ echo 'k3s_control_plane=PASS servers=3 witness_business_pods=0'
database_state=$(cluster_ssh "$CLUSTER_NINGBO_HOST" 'bash -s' <<'REMOTE'
set -euo pipefail
kubectl='k3s kubectl'
[[ $($kubectl get cluster easyai-postgres -n easyai -o jsonpath='{.status.readyInstances}') -eq 2 ]]
ready_instances=$($kubectl get cluster easyai-postgres -n easyai -o jsonpath='{.status.readyInstances}')
[[ $ready_instances -eq 2 ]] || {
echo "postgres ready instance gate failed: ready_instances=$ready_instances expected=2" >&2
exit 1
}
primary=$($kubectl get cluster easyai-postgres -n easyai -o jsonpath='{.status.currentPrimary}')
[[ $($kubectl get pod "$primary" -n easyai -o jsonpath='{.spec.nodeName}') == easyai-ningbo ||
$($kubectl get pod "$primary" -n easyai -o jsonpath='{.spec.nodeName}') == easyai-hongkong ]]
primary_node=$($kubectl get pod "$primary" -n easyai -o jsonpath='{.spec.nodeName}')
[[ $primary_node == easyai-ningbo || $primary_node == easyai-hongkong ]] || {
echo "postgres primary placement gate failed: primary_node=$primary_node" >&2
exit 1
}
replication=$($kubectl exec -n easyai "$primary" -c postgres -- \
psql -X -U postgres -d easyai_ai_gateway -At -c \
"SELECT count(*) FROM pg_stat_replication WHERE sync_state IN ('sync','quorum');")
[[ $replication -eq 1 ]]
[[ $replication -eq 1 ]] || {
echo "postgres synchronous replication gate failed: synchronous_replicas=$replication expected=1" >&2
exit 1
}
archived_count=$($kubectl exec -n easyai "$primary" -c postgres -- \
psql -X -U postgres -d postgres -At -c \
'SELECT archived_count FROM pg_stat_archiver;')
(( archived_count > 0 ))
[[ $($kubectl get cluster easyai-postgres -n easyai \
-o jsonpath='{.status.conditions[?(@.type=="ContinuousArchiving")].status}') == True ]]
(( archived_count > 0 )) || {
echo "postgres archive gate failed: archived_count=$archived_count" >&2
exit 1
}
continuous_archiving=$($kubectl get cluster easyai-postgres -n easyai \
-o jsonpath='{.status.conditions[?(@.type=="ContinuousArchiving")].status}')
[[ $continuous_archiving == True ]] || {
echo "postgres continuous archiving gate failed: status=$continuous_archiving" >&2
exit 1
}
completed_backups=$($kubectl get backups -n easyai -o json |
jq '[.items[] | select(
.spec.cluster.name == "easyai-postgres" and .status.phase == "completed"
)] | length')
(( completed_backups > 0 ))
(( completed_backups > 0 )) || {
echo "postgres backup gate failed: completed_backups=$completed_backups" >&2
exit 1
}
readable_backup_configs=0
for pod in $($kubectl get pods -n easyai -l cnpg.io/cluster=easyai-postgres \
-o jsonpath='{.items[*].metadata.name}'); do
$kubectl exec -n easyai "$pod" -c postgres -- \
test -r /projected/barman-aws/config
if $kubectl exec -n easyai "$pod" -c postgres -- \
test -r /projected/barman-aws/config; then
readable_backup_configs=$((readable_backup_configs + 1))
else
echo "postgres projected backup config gate failed: pod=$pod" >&2
exit 1
fi
done
printf 'primary=%s archived=%s backups=%s\n' \
"$primary" "$archived_count" "$completed_backups"
[[ $readable_backup_configs -eq 2 ]] || {
echo "postgres projected backup config count failed: readable=$readable_backup_configs expected=2" >&2
exit 1
}
printf 'primary=%s archived=%s backups=%s ready=%s synchronous=%s continuous_archiving=%s backup_configs=%s\n' \
"$primary" "$archived_count" "$completed_backups" "$ready_instances" \
"$replication" "$continuous_archiving" "$readable_backup_configs"
REMOTE
)
[[ $database_state == primary=easyai-postgres-* ]]