feat(acceptance): 建立同构验收与弹性容量体系

实现本地三节点 K3s 同构环境、脱敏生产快照、Gemini 图片和多参考图视频协议模拟、统一验收报告及故障注入。\n\n新增 Worker 容量控制器、资源与连接预算、任务恢复保护,并将生产验收拆分为 validation 执行和人工 CAS 放量。\n\n验证包括 Go 全量测试、PostgreSQL HTTP 集成测试、go vet、OpenAPI、ShellCheck、前端检查、迁移及发布脚本测试。
This commit is contained in:
2026-07-31 18:02:24 +08:00
parent 93d36d0e55
commit e05922b0f4
94 changed files with 12379 additions and 826 deletions
+43
View File
@@ -0,0 +1,43 @@
#!/usr/bin/env bash
set -euo pipefail
usage() {
cat <<'EOF'
Usage:
AI_GATEWAY_ACCEPTANCE_SNAPSHOT_DATABASE_URL='postgresql://...' \
scripts/acceptance/export-production-snapshot.sh \
--release-sha <full-sha> --output <snapshot.json>
This command is database read-only. The supplied role should have SELECT access
only. It never prints the database URL or candidate credentials.
EOF
}
script_dir=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)
repository_root=$(cd "$script_dir/../.." && pwd)
[[ ${1:-} == --release-sha && ${3:-} == --output && $# -eq 4 ]] || {
usage >&2
exit 64
}
release_sha=$2
output=$4
database_url=${AI_GATEWAY_ACCEPTANCE_SNAPSHOT_DATABASE_URL:-}
[[ $release_sha =~ ^[0-9a-f]{40}$ ]] || {
echo 'release SHA must be a full lowercase Git SHA' >&2
exit 64
}
[[ -n $database_url ]] || {
echo 'AI_GATEWAY_ACCEPTANCE_SNAPSHOT_DATABASE_URL is required' >&2
exit 64
}
umask 077
(
cd "$repository_root/apps/api"
AI_GATEWAY_DATABASE_URL="$database_url" \
go run ./cmd/acceptance-snapshot export \
--release-sha "$release_sha" \
--output "$output"
)
chmod 0600 "$output"
+556
View File
@@ -0,0 +1,556 @@
#!/usr/bin/env bash
set -euo pipefail
script_dir=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)
repository_root=$(cd "$script_dir/../.." && pwd)
manifest_root="$repository_root/deploy/kubernetes/local-acceptance"
lock_file="$manifest_root/dependencies.lock"
private_root="$repository_root/.local-secrets/acceptance"
tools_root="$private_root/tools"
state_root="$private_root/state"
media_root="$private_root/media"
cluster_name=easyai-acceptance-local
context="k3d-$cluster_name"
namespace=easyai
usage() {
cat <<'EOF'
Usage:
scripts/acceptance/local-cluster.sh install-tools
scripts/acceptance/local-cluster.sh preflight
scripts/acceptance/local-cluster.sh up --snapshot <acceptance-snapshot-v1.json>
scripts/acceptance/local-cluster.sh status
scripts/acceptance/local-cluster.sh down --confirm
The full cluster requires Docker Desktop memory >= 24 GiB. Failed `up` runs are
kept intact for diagnosis. `down` is the only destructive lifecycle action.
EOF
}
load_lock() {
[[ -f $lock_file && ! -L $lock_file ]] || {
echo "missing dependency lock: $lock_file" >&2
exit 1
}
# shellcheck source=/dev/null
source "$lock_file"
: "${K3D_VERSION:?}" "${K3S_IMAGE:?}" "${CNPG_MANIFEST_URL:?}" "${CNPG_MANIFEST_SHA256:?}"
}
require_command() {
command -v "$1" >/dev/null 2>&1 || {
echo "$1 is required" >&2
exit 1
}
}
private_file() {
local path=$1
[[ -f $path && ! -L $path ]] || return 1
local mode
if [[ $(uname -s) == Darwin ]]; then
mode=$(stat -f '%Lp' "$path")
else
mode=$(stat -c '%a' "$path")
fi
[[ $mode == 600 ]]
}
install_tools() {
load_lock
require_command curl
require_command shasum
install -d -m 0700 "$tools_root"
local os arch checksum url temporary
os=$(uname -s | tr '[:upper:]' '[:lower:]')
arch=$(uname -m)
[[ $os == darwin ]] || {
echo 'the pinned local acceptance installer currently supports macOS only' >&2
exit 1
}
case $arch in
arm64)
arch=arm64
checksum=$K3D_DARWIN_ARM64_SHA256
;;
x86_64)
arch=amd64
checksum=$K3D_DARWIN_AMD64_SHA256
;;
*)
echo "unsupported host architecture: $arch" >&2
exit 1
;;
esac
url="https://github.com/k3d-io/k3d/releases/download/$K3D_VERSION/k3d-darwin-$arch"
temporary="$tools_root/k3d.download"
curl -fsSL "$url" -o "$temporary"
[[ $(shasum -a 256 "$temporary" | awk '{print $1}') == "$checksum" ]] || {
echo 'k3d checksum mismatch' >&2
exit 1
}
chmod 0555 "$temporary"
mv "$temporary" "$tools_root/k3d"
echo "local_acceptance_tools=PASS k3d=$K3D_VERSION path=$tools_root/k3d"
}
k3d_binary() {
if [[ -x $tools_root/k3d ]]; then
printf '%s\n' "$tools_root/k3d"
return
fi
command -v k3d
}
docker_memory_bytes() {
docker info --format '{{.MemTotal}}'
}
preflight() {
load_lock
require_command docker
require_command kubectl
require_command jq
require_command openssl
require_command sed
require_command shasum
require_command go
require_command node
require_command curl
local k3d memory_bytes architecture
k3d=$(k3d_binary) || {
echo 'k3d is missing; run install-tools' >&2
exit 1
}
"$k3d" version | grep -Fq "$K3D_VERSION" || {
echo "k3d must be pinned to $K3D_VERSION" >&2
exit 1
}
docker info >/dev/null
memory_bytes=$(docker_memory_bytes)
[[ $memory_bytes =~ ^[0-9]+$ && $memory_bytes -ge 25769803776 ]] || {
memory_gib=$(awk -v bytes="${memory_bytes:-0}" 'BEGIN {printf "%.1f", bytes/1024/1024/1024}')
echo "Docker Desktop memory is ${memory_gib} GiB; local acceptance requires at least 24 GiB" >&2
exit 1
}
architecture=$(docker info --format '{{.Architecture}}')
[[ $architecture == aarch64 || $architecture == arm64 || $architecture == x86_64 ]] || {
echo "unsupported Docker architecture: $architecture" >&2
exit 1
}
docker buildx version >/dev/null
echo "local_acceptance_preflight=PASS docker_memory_bytes=$memory_bytes architecture=$architecture k3d=$K3D_VERSION"
}
ensure_private_material() {
install -d -m 0700 "$private_root" "$state_root" "$media_root"
local env_file="$state_root/local.env"
if [[ ! -e $env_file ]]; then
umask 077
{
printf 'LOCAL_CLUSTER_ID=%s\n' "easyai-local-$(openssl rand -hex 12)"
printf 'POSTGRES_PASSWORD=%s\n' "$(openssl rand -hex 32)"
printf 'CONFIG_JWT_SECRET=%s\n' "$(openssl rand -hex 32)"
printf 'SERVER_MAIN_INTERNAL_TOKEN=%s\n' "$(openssl rand -hex 32)"
printf 'SERVER_MAIN_INTERNAL_KEY=local-acceptance\n'
printf 'SERVER_MAIN_INTERNAL_SECRET=%s\n' "$(openssl rand -hex 32)"
} >"$env_file"
chmod 0600 "$env_file"
fi
private_file "$env_file" || {
echo "local acceptance state must be a 0600 regular file: $env_file" >&2
exit 1
}
}
generate_tls() {
local ca_key="$state_root/ca.key" ca_crt="$state_root/ca.crt"
local tls_key="$state_root/tls.key" tls_csr="$state_root/tls.csr"
local tls_crt="$state_root/tls.crt" extensions="$state_root/tls.ext"
if [[ ! -e $ca_key ]]; then
umask 077
openssl genrsa -out "$ca_key" 3072 >/dev/null 2>&1
openssl req -x509 -new -key "$ca_key" -sha256 -days 30 \
-subj '/CN=EasyAI local acceptance CA' -out "$ca_crt" >/dev/null 2>&1
openssl genrsa -out "$tls_key" 2048 >/dev/null 2>&1
openssl req -new -key "$tls_key" -subj '/CN=gateway.easyai.local' \
-out "$tls_csr" >/dev/null 2>&1
printf 'subjectAltName=DNS:gateway.easyai.local\nextendedKeyUsage=serverAuth\n' >"$extensions"
openssl x509 -req -in "$tls_csr" -CA "$ca_crt" -CAkey "$ca_key" \
-CAcreateserial -out "$tls_crt" -days 30 -sha256 -extfile "$extensions" \
>/dev/null 2>&1
chmod 0600 "$ca_key" "$ca_crt" "$tls_key" "$tls_csr" "$tls_crt" "$extensions"
fi
private_file "$ca_crt" && private_file "$tls_key" && private_file "$tls_crt"
}
render_k3d_config() {
local output=$1
[[ $media_root != *'|'* ]]
sed "s|EASYAI_ACCEPTANCE_MEDIA_DIR|$media_root|g" "$manifest_root/k3d.yaml" >"$output"
}
wait_rollout() {
kubectl --context "$context" -n "$namespace" rollout status "$1" --timeout="${2:-10m}"
}
apply_runtime_secrets() {
# shellcheck source=/dev/null
source "$state_root/local.env"
local database_ningbo database_hongkong database_direct
database_ningbo="postgresql://easyai:${POSTGRES_PASSWORD}@easyai-postgres-proxy-ningbo.easyai.svc.cluster.local:5432/easyai_ai_gateway?sslmode=require"
database_hongkong="postgresql://easyai:${POSTGRES_PASSWORD}@easyai-postgres-proxy-hongkong.easyai.svc.cluster.local:5432/easyai_ai_gateway?sslmode=require"
database_direct="postgresql://easyai:${POSTGRES_PASSWORD}@easyai-postgres-rw.easyai.svc.cluster.local:5432/easyai_ai_gateway?sslmode=require"
kubectl --context "$context" create namespace "$namespace" --dry-run=client -o yaml |
kubectl --context "$context" apply -f - >/dev/null
kubectl --context "$context" -n "$namespace" create secret generic easyai-postgres-app \
--type=kubernetes.io/basic-auth \
--from-literal=username=easyai --from-literal=password="$POSTGRES_PASSWORD" \
--dry-run=client -o yaml | kubectl --context "$context" apply -f - >/dev/null
kubectl --context "$context" -n "$namespace" create secret generic easyai-ai-gateway-runtime \
--from-literal=AI_GATEWAY_DATABASE_URL="$database_direct" \
--from-literal=CONFIG_JWT_SECRET="$CONFIG_JWT_SECRET" \
--from-literal=SERVER_MAIN_INTERNAL_TOKEN="$SERVER_MAIN_INTERNAL_TOKEN" \
--from-literal=SERVER_MAIN_INTERNAL_KEY="$SERVER_MAIN_INTERNAL_KEY" \
--from-literal=SERVER_MAIN_INTERNAL_SECRET="$SERVER_MAIN_INTERNAL_SECRET" \
--dry-run=client -o yaml | kubectl --context "$context" apply -f - >/dev/null
kubectl --context "$context" -n "$namespace" create secret generic easyai-database-ningbo \
--from-literal=AI_GATEWAY_DATABASE_URL="$database_ningbo" \
--dry-run=client -o yaml | kubectl --context "$context" apply -f - >/dev/null
kubectl --context "$context" -n "$namespace" create secret generic easyai-database-hongkong \
--from-literal=AI_GATEWAY_DATABASE_URL="$database_hongkong" \
--dry-run=client -o yaml | kubectl --context "$context" apply -f - >/dev/null
kubectl --context "$context" -n "$namespace" create secret generic easyai-oss-backup \
--from-literal=ACCESS_KEY_ID=local-disabled \
--from-literal=ACCESS_SECRET_KEY=local-disabled \
--dry-run=client -o yaml | kubectl --context "$context" apply -f - >/dev/null
kubectl --context "$context" -n "$namespace" create secret tls easyai-acceptance-tls \
--cert="$state_root/tls.crt" --key="$state_root/tls.key" \
--dry-run=client -o yaml | kubectl --context "$context" apply -f - >/dev/null
}
run_migrations() {
local api_image=$1
kubectl --context "$context" -n "$namespace" delete job easyai-local-migrate \
--ignore-not-found --wait=true >/dev/null
cat <<EOF | kubectl --context "$context" apply -f - >/dev/null
apiVersion: batch/v1
kind: Job
metadata:
name: easyai-local-migrate
namespace: easyai
spec:
backoffLimit: 0
template:
spec:
restartPolicy: Never
containers:
- name: migrate
image: $api_image
command: ["/bin/sh", "-ec", "cd /app && exec /app/easyai-ai-gateway-migrate"]
envFrom:
- secretRef:
name: easyai-ai-gateway-runtime
EOF
kubectl --context "$context" -n "$namespace" wait \
--for=condition=complete job/easyai-local-migrate --timeout=10m >/dev/null
}
mark_local_database() {
# shellcheck source=/dev/null
source "$state_root/local.env"
local primary
primary=$(kubectl --context "$context" -n "$namespace" get cluster easyai-postgres \
-o 'jsonpath={.status.currentPrimary}')
kubectl --context "$context" -n "$namespace" exec "$primary" -c postgres -- \
psql -X -v ON_ERROR_STOP=1 -U postgres -d easyai_ai_gateway \
-v cluster_id="$LOCAL_CLUSTER_ID" -c \
"INSERT INTO system_settings(setting_key,value)
VALUES('acceptance_local_cluster_id',jsonb_build_object('clusterId', :'cluster_id'))
ON CONFLICT(setting_key) DO UPDATE SET value=excluded.value,updated_at=now();" \
>/dev/null
}
import_snapshot() {
local api_image=$1 snapshot=$2
# shellcheck source=/dev/null
source "$state_root/local.env"
kubectl --context "$context" -n "$namespace" create configmap easyai-acceptance-snapshot \
--from-file=snapshot.json="$snapshot" --dry-run=client -o yaml |
kubectl --context "$context" apply -f - >/dev/null
kubectl --context "$context" -n "$namespace" delete job easyai-local-snapshot-import \
--ignore-not-found --wait=true >/dev/null
cat <<EOF | kubectl --context "$context" apply -f - >/dev/null
apiVersion: batch/v1
kind: Job
metadata:
name: easyai-local-snapshot-import
namespace: easyai
spec:
backoffLimit: 0
template:
spec:
restartPolicy: Never
containers:
- name: import
image: $api_image
command:
- /app/easyai-ai-gateway-acceptance-snapshot
- import
- --input
- /snapshot/snapshot.json
- --local-cluster-id
- $LOCAL_CLUSTER_ID
envFrom:
- secretRef:
name: easyai-ai-gateway-runtime
volumeMounts:
- name: snapshot
mountPath: /snapshot
readOnly: true
volumes:
- name: snapshot
configMap:
name: easyai-acceptance-snapshot
EOF
kubectl --context "$context" -n "$namespace" wait \
--for=condition=complete job/easyai-local-snapshot-import --timeout=5m >/dev/null
}
render_and_apply_application() {
local api_image=$1 web_image=$2 rendered=$state_root/application.rendered.yaml
sed \
-e "s|image: easyai-api|image: $api_image|g" \
-e "s|image: easyai-web|image: $web_image|g" \
-e 's/nodePort: 31088/nodePort: 32088/g' \
-e 's/nodePort: 31089/nodePort: 32089/g' \
"$repository_root/deploy/kubernetes/production/application.yaml" >"$rendered"
chmod 0600 "$rendered"
kubectl --context "$context" -n "$namespace" apply \
-f "$repository_root/deploy/kubernetes/production/service-account-rbac.yaml" >/dev/null
kubectl --context "$context" -n "$namespace" apply -f "$manifest_root/local-config.yaml" >/dev/null
kubectl --context "$context" -n "$namespace" apply -f "$rendered" >/dev/null
for workload in easyai-api-ningbo easyai-worker-ningbo; do
kubectl --context "$context" -n "$namespace" set env deployment/"$workload" \
--from=secret/easyai-database-ningbo >/dev/null
done
for workload in easyai-api-hongkong easyai-worker-hongkong; do
kubectl --context "$context" -n "$namespace" set env deployment/"$workload" \
--from=secret/easyai-database-hongkong >/dev/null
done
for workload in easyai-api-ningbo easyai-api-hongkong easyai-worker-ningbo easyai-worker-hongkong; do
kubectl --context "$context" -n "$namespace" patch deployment "$workload" --type=json \
-p='[{"op":"replace","path":"/spec/template/spec/volumes/0","value":{"name":"application-data","hostPath":{"path":"/var/lib/easyai-acceptance/media","type":"DirectoryOrCreate"}}}]' \
>/dev/null
done
kubectl --context "$context" -n "$namespace" scale \
deployment/easyai-web-ningbo deployment/easyai-web-hongkong --replicas=0 >/dev/null
}
bootstrap_runtime() {
local api_image=$1 api_digest=$2 snapshot=$3
# shellcheck source=/dev/null
source "$state_root/local.env"
local snapshot_hash release_sha runtime_file
snapshot_hash=$(jq -r '.snapshotSha256' "$snapshot")
release_sha=$(git -C "$repository_root" rev-parse HEAD)
runtime_file="$state_root/runtime-$snapshot_hash.json"
kubectl --context "$context" -n "$namespace" delete pod easyai-local-bootstrap \
--ignore-not-found --wait=true >/dev/null
cat <<EOF | kubectl --context "$context" apply -f - >/dev/null
apiVersion: v1
kind: Pod
metadata:
name: easyai-local-bootstrap
namespace: easyai
labels:
easyai.io/environment: local-acceptance
spec:
restartPolicy: Never
containers:
- name: bootstrap
image: $api_image
command:
- /app/easyai-ai-gateway-acceptance-bootstrap
- --local-cluster-id
- $LOCAL_CLUSTER_ID
- --release-sha
- $release_sha
- --api-image-digest
- $api_digest
- --worker-image-digest
- $api_digest
- --emulator-base-url
- http://easyai-acceptance-upstream-proxy.easyai.svc.cluster.local:8090
- --callback-url
- http://easyai-acceptance-callback-collector.easyai.svc.cluster.local:8091/callbacks
- --identity-shards
- "32"
- --output
- /tmp/runtime.json
envFrom:
- secretRef:
name: easyai-ai-gateway-runtime
EOF
kubectl --context "$context" -n "$namespace" wait \
--for=jsonpath='{.status.phase}'=Succeeded pod/easyai-local-bootstrap --timeout=5m >/dev/null
umask 077
kubectl --context "$context" -n "$namespace" cp \
easyai-local-bootstrap:/tmp/runtime.json "$runtime_file" >/dev/null
chmod 0600 "$runtime_file"
kubectl --context "$context" -n "$namespace" delete pod easyai-local-bootstrap \
--wait=true >/dev/null
printf '%s\n' "$runtime_file" >"$state_root/current-runtime-path"
chmod 0600 "$state_root/current-runtime-path"
}
up_cluster() {
[[ ${1:-} == --snapshot && $# -eq 2 ]] || {
usage >&2
exit 64
}
local snapshot=$2
[[ -f $snapshot && ! -L $snapshot ]] || {
echo 'snapshot must be a regular non-symlink file' >&2
exit 1
}
[[ -z $(git -C "$repository_root" status --short) ]] || {
echo 'local acceptance requires a clean source tree so report and image source are reproducible' >&2
exit 1
}
preflight
ensure_private_material
generate_tls
(
cd "$repository_root/apps/api"
go run ./cmd/acceptance-snapshot validate --input "$snapshot"
)
cp "$snapshot" "$state_root/snapshot.json"
chmod 0600 "$state_root/snapshot.json"
local k3d config native_arch release_sha api_image web_image netem_image api_digest
k3d=$(k3d_binary)
if "$k3d" cluster list --no-headers | awk '{print $1}' | grep -Fxq "$cluster_name"; then
echo "cluster $cluster_name already exists; use status or explicit down --confirm" >&2
exit 1
fi
native_arch=$(docker info --format '{{.Architecture}}')
case $native_arch in
aarch64|arm64) native_arch=arm64 ;;
x86_64) native_arch=amd64 ;;
esac
release_sha=$(git -C "$repository_root" rev-parse HEAD)
api_image="easyai-acceptance-api:$release_sha"
web_image="easyai-acceptance-web:$release_sha"
netem_image="easyai-acceptance-netem:$release_sha"
docker buildx build --load --platform "linux/$native_arch" --target api \
-t "$api_image" "$repository_root"
docker buildx build --load --platform "linux/$native_arch" --target web \
-t "$web_image" "$repository_root"
docker buildx build --load --platform "linux/$native_arch" --target acceptance-netem \
-t "$netem_image" "$repository_root"
api_digest=$(docker image inspect "$api_image" --format '{{.Id}}')
[[ $api_digest =~ ^sha256:[0-9a-f]{64}$ ]]
config="$state_root/k3d.rendered.yaml"
render_k3d_config "$config"
EASYAI_ACCEPTANCE_MEDIA_DIR="$media_root" "$k3d" cluster create --config "$config"
docker update --cpus 4 --memory 8g "${cluster_name}-server-0" >/dev/null
docker update --cpus 4 --memory 8g "${cluster_name}-server-1" >/dev/null
docker update --cpus 2 --memory 4g "${cluster_name}-server-2" >/dev/null
"$k3d" image import -c "$cluster_name" "$api_image" "$web_image" "$netem_image"
local cnpg_manifest="$state_root/cnpg.yaml"
curl -fsSL "$CNPG_MANIFEST_URL" -o "$cnpg_manifest"
[[ $(shasum -a 256 "$cnpg_manifest" | awk '{print $1}') == "$CNPG_MANIFEST_SHA256" ]] || {
echo 'CNPG manifest checksum mismatch' >&2
exit 1
}
chmod 0600 "$cnpg_manifest"
kubectl --context "$context" apply --server-side --force-conflicts -f "$cnpg_manifest" >/dev/null
kubectl --context "$context" -n cnpg-system rollout status \
deployment/cnpg-controller-manager --timeout=10m
apply_runtime_secrets
kubectl --context "$context" -n "$namespace" apply -f "$manifest_root/database.yaml" >/dev/null
kubectl --context "$context" -n "$namespace" wait --for=condition=Ready \
cluster/easyai-postgres --timeout=15m >/dev/null
run_migrations "$api_image"
mark_local_database
import_snapshot "$api_image" "$snapshot"
sed "s|image: easyai-api|image: $api_image|g; s|image: easyai-acceptance-netem|image: $netem_image|g" \
"$manifest_root/support-services.yaml" |
kubectl --context "$context" -n "$namespace" apply -f - >/dev/null
wait_rollout deployment/easyai-acceptance-emulator
wait_rollout deployment/easyai-acceptance-callback-collector
wait_rollout deployment/easyai-acceptance-upstream-proxy
wait_rollout deployment/easyai-postgres-proxy-ningbo
wait_rollout deployment/easyai-postgres-proxy-hongkong
render_and_apply_application "$api_image" "$web_image"
sed "s|image: easyai-web|image: $web_image|g" "$manifest_root/tls-edges.yaml" |
kubectl --context "$context" -n "$namespace" apply -f - >/dev/null
wait_rollout deployment/easyai-api-ningbo
wait_rollout deployment/easyai-api-hongkong
wait_rollout deployment/easyai-worker-ningbo
wait_rollout deployment/easyai-worker-hongkong
wait_rollout deployment/easyai-capacity-controller
wait_rollout deployment/easyai-acceptance-edge-ningbo
wait_rollout deployment/easyai-acceptance-edge-hongkong
bootstrap_runtime "$api_image" "$api_digest" "$snapshot"
"$script_dir/network-fault.sh" baseline
echo "local_acceptance_cluster=PASS context=$context gateways=https://127.0.0.1:18443,https://127.0.0.1:19443 ca_file=$state_root/ca.crt"
}
status_cluster() {
local k3d
k3d=$(k3d_binary) || {
echo 'k3d is unavailable' >&2
exit 1
}
"$k3d" cluster list --no-headers | awk '{print $1}' | grep -Fxq "$cluster_name" || {
echo "cluster $cluster_name does not exist" >&2
exit 1
}
kubectl --context "$context" get nodes -o wide
kubectl --context "$context" -n "$namespace" get cluster,pod,deployment
echo "local_acceptance_status=PASS context=$context"
}
down_cluster() {
[[ ${1:-} == --confirm && $# -eq 1 ]] || {
echo 'down requires the literal --confirm flag' >&2
exit 64
}
local k3d
k3d=$(k3d_binary)
"$k3d" cluster delete "$cluster_name"
echo "local_acceptance_down=PASS preserved_private_root=$private_root"
}
case ${1:-} in
install-tools)
[[ $# -eq 1 ]] || { usage >&2; exit 64; }
install_tools
;;
preflight)
[[ $# -eq 1 ]] || { usage >&2; exit 64; }
preflight
;;
up)
shift
up_cluster "$@"
;;
status)
[[ $# -eq 1 ]] || { usage >&2; exit 64; }
status_cluster
;;
down)
shift
down_cluster "$@"
;;
*)
usage >&2
exit 64
;;
esac
+17
View File
@@ -0,0 +1,17 @@
#!/bin/sh
set -eu
listen_port=${NETEM_LISTEN_PORT:-15432}
target_host=${NETEM_TARGET_HOST:-}
target_port=${NETEM_TARGET_PORT:-}
case "$listen_port:$target_port" in
*[!0-9:]*|:*|*:) echo "NETEM_LISTEN_PORT and NETEM_TARGET_PORT must be numeric" >&2; exit 64 ;;
esac
case "$target_host" in
""|*[!A-Za-z0-9.-]*) echo "NETEM_TARGET_HOST must be a DNS name or address" >&2; exit 64 ;;
esac
exec socat \
"TCP-LISTEN:${listen_port},fork,reuseaddr,keepalive" \
"TCP:${target_host}:${target_port},connect-timeout=10"
+96
View File
@@ -0,0 +1,96 @@
#!/usr/bin/env bash
set -euo pipefail
usage() {
cat <<'EOF'
Usage:
scripts/acceptance/network-fault.sh baseline
scripts/acceptance/network-fault.sh weak-link
scripts/acceptance/network-fault.sh upstream-outage
scripts/acceptance/network-fault.sh database-outage <ningbo|hongkong>
scripts/acceptance/network-fault.sh reset
Only operates on Pods carrying easyai.io/environment=local-acceptance in the
easyai-acceptance-local k3d context.
EOF
}
profile=${1:-}
site=${2:-}
namespace=${AI_GATEWAY_LOCAL_ACCEPTANCE_NAMESPACE:-easyai}
context=${AI_GATEWAY_LOCAL_ACCEPTANCE_CONTEXT:-k3d-easyai-acceptance-local}
command -v kubectl >/dev/null 2>&1 || {
echo 'kubectl is required' >&2
exit 1
}
[[ $(kubectl config get-contexts "$context" -o name) == "$context" ]] || {
echo "refusing netem because context $context does not exist" >&2
exit 1
}
pod_for() {
local name=$1
kubectl --context "$context" -n "$namespace" get pod \
-l "app.kubernetes.io/name=$name,easyai.io/environment=local-acceptance" \
-o 'jsonpath={.items[0].metadata.name}'
}
apply_qdisc() {
local pod=$1
shift
[[ -n $pod ]] || {
echo 'local acceptance proxy Pod was not found' >&2
exit 1
}
kubectl --context "$context" -n "$namespace" exec "$pod" -- \
tc qdisc replace dev eth0 root netem "$@"
}
reset_qdisc() {
local pod=$1
[[ -z $pod ]] || kubectl --context "$context" -n "$namespace" exec "$pod" -- \
tc qdisc del dev eth0 root >/dev/null 2>&1 || true
}
upstream_pod=$(pod_for easyai-acceptance-upstream-proxy)
ningbo_db_pod=$(kubectl --context "$context" -n "$namespace" get pod \
-l 'app.kubernetes.io/name=easyai-postgres-proxy,easyai.io/site=ningbo,easyai.io/environment=local-acceptance' \
-o 'jsonpath={.items[0].metadata.name}')
hongkong_db_pod=$(kubectl --context "$context" -n "$namespace" get pod \
-l 'app.kubernetes.io/name=easyai-postgres-proxy,easyai.io/site=hongkong,easyai.io/environment=local-acceptance' \
-o 'jsonpath={.items[0].metadata.name}')
case $profile in
baseline)
apply_qdisc "$upstream_pod" delay 20ms
apply_qdisc "$ningbo_db_pod" delay 20ms
apply_qdisc "$hongkong_db_pod" delay 20ms
;;
weak-link)
apply_qdisc "$upstream_pod" delay 20ms 10ms distribution normal loss 0.5%
apply_qdisc "$ningbo_db_pod" delay 20ms 10ms distribution normal loss 0.5%
apply_qdisc "$hongkong_db_pod" delay 20ms 10ms distribution normal loss 0.5%
;;
upstream-outage)
apply_qdisc "$upstream_pod" loss 100%
;;
database-outage)
case $site in
ningbo) apply_qdisc "$ningbo_db_pod" loss 100% ;;
hongkong) apply_qdisc "$hongkong_db_pod" loss 100% ;;
*) usage >&2; exit 64 ;;
esac
;;
reset)
reset_qdisc "$upstream_pod"
reset_qdisc "$ningbo_db_pod"
reset_qdisc "$hongkong_db_pod"
;;
*)
usage >&2
exit 64
;;
esac
echo "local_acceptance_netem=PASS profile=$profile${site:+ site=$site}"
+316
View File
@@ -0,0 +1,316 @@
#!/usr/bin/env node
import { createHash } from 'node:crypto';
import { lstat, mkdir, readFile, readdir, writeFile } from 'node:fs/promises';
import { dirname, resolve } from 'node:path';
const shaPattern = /^[0-9a-f]{40}$/;
const digestPattern = /^sha256:[0-9a-f]{64}$/;
const hashPattern = /^[0-9a-f]{64}$/;
const unsafeKeyPattern = /(password|secret|token|credential|authorization|api.?key|access.?key|private.?key|connection.?string|database.?url|proxy)/i;
function parseArgs(argv) {
const command = argv.shift();
const values = {};
while (argv.length > 0) {
const key = argv.shift();
if (!key?.startsWith('--') || argv.length === 0) throw new Error('invalid arguments');
values[key.slice(2)] = argv.shift();
}
return { command, values };
}
async function regularJSON(path) {
const absolute = resolve(path);
const info = await lstat(absolute);
if (!info.isFile() || info.isSymbolicLink()) throw new Error(`${path} must be a regular non-symlink file`);
return JSON.parse(await readFile(absolute, 'utf8'));
}
function assertSecretSafe(value, path = '$') {
if (Array.isArray(value)) {
value.forEach((item, index) => assertSecretSafe(item, `${path}[${index}]`));
return;
}
if (value && typeof value === 'object') {
for (const [key, nested] of Object.entries(value)) {
if (key !== 'secretSafe' && unsafeKeyPattern.test(key)) {
throw new Error(`unsafe report key at ${path}.${key}`);
}
assertSecretSafe(nested, `${path}.${key}`);
}
return;
}
if (typeof value === 'string' && /(?:postgres(?:ql)?|https?):\/\/[^\s]+/i.test(value)) {
throw new Error(`URL-like value is forbidden at ${path}`);
}
}
function validate(report) {
if (report.schemaVersion !== 'acceptance-report/v1' || report.secretSafe !== true) {
throw new Error('unsupported or non-secret-safe acceptance report');
}
if (!report.runId || !shaPattern.test(report.release?.sourceSha ?? '') ||
!digestPattern.test(report.release?.apiImageDigest ?? '') ||
!digestPattern.test(report.release?.workerImageDigest ?? '') ||
!shaPattern.test(report.snapshot?.sourceReleaseSha ?? '') ||
!hashPattern.test(report.snapshot?.configHash ?? '') ||
!hashPattern.test(report.snapshot?.sha256 ?? '')) {
throw new Error('acceptance report identity fields are invalid');
}
if (!report.stages || !Array.isArray(report.gates) ||
report.promotion?.requiresManualConfirmation !== true ||
!['validation', 'live', 'not-applicable'].includes(report.promotion?.trafficMode)) {
throw new Error('acceptance report stages, gates, or promotion are invalid');
}
for (const [name, stage] of Object.entries(report.stages)) {
if (!['passed', 'failed', 'not-run'].includes(stage?.status)) {
throw new Error(`invalid stage status for ${name}`);
}
}
assertSecretSafe(report);
}
async function loadReports(directory) {
const entries = (await readdir(resolve(directory), { withFileTypes: true }))
.filter((entry) => entry.isFile() && entry.name.endsWith('.json'))
.map((entry) => entry.name)
.sort();
const reports = [];
for (const name of entries) {
const report = await regularJSON(resolve(directory, name));
if (report.schemaVersion === 'acceptance-load-report/v1') {
if (report.secretSafe !== true) throw new Error(`load report ${name} is not secret-safe`);
reports.push({
file: name,
profile: report.profile,
passed: report.passed,
phases: report.phases,
startedAt: report.startedAt,
finishedAt: report.finishedAt
});
}
}
return reports;
}
async function buildLocal(values) {
for (const key of ['runtime', 'snapshot', 'reports', 'artifact', 'output']) {
if (!values[key]) throw new Error(`--${key} is required`);
}
const runtime = await regularJSON(values.runtime);
const snapshot = await regularJSON(values.snapshot);
const artifact = await regularJSON(values.artifact);
const loads = await loadReports(values.reports);
if (runtime.schemaVersion !== 'acceptance-runtime/v1' ||
snapshot.schemaVersion !== 'acceptance-snapshot/v1' ||
artifact.schemaVersion !== 'acceptance-artifact-smoke/v1') {
throw new Error('local report inputs have incompatible schema versions');
}
if (runtime.releaseSha !== artifact.releaseSha ||
artifact.apiImageDigest !== values['api-digest'] ||
runtime.snapshotConfigHash !== snapshot.source.configHash ||
runtime.snapshotSha256 !== snapshot.snapshotSha256) {
throw new Error('local report CAS identity mismatch');
}
const localPassed = loads.length > 0 && loads.every((item) => item.passed === true);
const gates = [
{ id: 'local_load_reports', passed: localPassed, detail: `${loads.length} load reports` },
{ id: 'amd64_artifact_smoke', passed: artifact.passed === true }
];
const report = {
schemaVersion: 'acceptance-report/v1',
runId: runtime.runId,
release: {
sourceSha: artifact.releaseSha,
apiImageDigest: artifact.apiImageDigest,
workerImageDigest: artifact.apiImageDigest
},
snapshot: {
sourceReleaseSha: snapshot.source.releaseSha,
configHash: snapshot.source.configHash,
sha256: snapshot.snapshotSha256
},
stages: {
localNative: { status: localPassed ? 'passed' : 'failed', loadReports: loads },
amd64Artifact: { status: artifact.passed ? 'passed' : 'failed', ...artifact },
onlineSimulation: { status: 'not-run' },
realCanary: { status: 'not-run' }
},
certifiedProfile: values['certified-profile'] ? {
name: values['certified-profile'],
source: 'local-candidate-only'
} : null,
gates,
promotion: {
ready: false,
requiresManualConfirmation: true,
trafficMode: 'not-applicable'
},
createdAt: new Date().toISOString(),
secretSafe: true
};
validate(report);
await mkdir(dirname(resolve(values.output)), { recursive: true, mode: 0o700 });
await writeFile(resolve(values.output), `${JSON.stringify(report, null, 2)}\n`, { mode: 0o600 });
process.stdout.write(`acceptance_report_local=PASS sha256=${createHash('sha256').update(JSON.stringify(report)).digest('hex')}\n`);
}
async function buildLocalPartial(values) {
for (const key of ['runtime', 'snapshot', 'reports', 'output', 'failure-phase']) {
if (!values[key]) throw new Error(`--${key} is required`);
}
const runtime = await regularJSON(values.runtime);
const snapshot = await regularJSON(values.snapshot);
const loads = await loadReports(values.reports);
if (runtime.schemaVersion !== 'acceptance-runtime/v1' ||
snapshot.schemaVersion !== 'acceptance-snapshot/v1' ||
!shaPattern.test(runtime.releaseSha ?? '') ||
!digestPattern.test(runtime.apiImageDigest ?? '') ||
!digestPattern.test(runtime.workerImageDigest ?? '') ||
runtime.snapshotConfigHash !== snapshot.source.configHash ||
runtime.snapshotSha256 !== snapshot.snapshotSha256) {
throw new Error('partial local report inputs have incompatible CAS identity');
}
const completedLoadsPassed = loads.every((item) => item.passed === true);
const report = {
schemaVersion: 'acceptance-report/v1',
runId: runtime.runId,
release: {
sourceSha: runtime.releaseSha,
apiImageDigest: runtime.apiImageDigest,
workerImageDigest: runtime.workerImageDigest
},
snapshot: {
sourceReleaseSha: snapshot.source.releaseSha,
configHash: snapshot.source.configHash,
sha256: snapshot.snapshotSha256
},
stages: {
localNative: {
status: 'failed',
failurePhase: values['failure-phase'],
completedLoadReportsPassed: completedLoadsPassed,
loadReports: loads
},
amd64Artifact: { status: 'not-run' },
onlineSimulation: { status: 'not-run' },
realCanary: { status: 'not-run' }
},
certifiedProfile: values['certified-profile'] ? {
name: values['certified-profile'],
source: 'last-local-stable-candidate'
} : null,
gates: [
{
id: 'local_execution_complete',
passed: false,
detail: `stopped during ${values['failure-phase']}`
},
{
id: 'completed_load_reports',
passed: completedLoadsPassed,
detail: `${loads.length} load reports`
}
],
promotion: {
ready: false,
requiresManualConfirmation: true,
trafficMode: 'not-applicable'
},
createdAt: new Date().toISOString(),
secretSafe: true
};
validate(report);
await mkdir(dirname(resolve(values.output)), { recursive: true, mode: 0o700 });
await writeFile(resolve(values.output), `${JSON.stringify(report, null, 2)}\n`, { mode: 0o600 });
process.stdout.write('acceptance_report_local_partial=PASS\n');
}
async function mergeProduction(values) {
for (const key of ['local-report', 'production-summary', 'output']) {
if (!values[key]) throw new Error(`--${key} is required`);
}
const local = await regularJSON(values['local-report']);
const production = await regularJSON(values['production-summary']);
validate(local);
if (local.release.sourceSha !== production.releaseSha ||
local.release.apiImageDigest !== production.apiImageDigest ||
local.snapshot.configHash !== production.snapshotConfigHash ||
local.snapshot.sha256 !== production.snapshotSha256) {
throw new Error('production summary does not match local acceptance CAS fields');
}
local.stages.onlineSimulation = {
status: production.passed === true ? 'passed' : 'failed',
stableCapacityProfile: production.stableCapacityProfile,
tasks: production.tasks,
geminiTasks: production.geminiTasks,
videoTasks: production.videoTasks
};
local.stages.realCanary = {
status: production.realCanaryPassed === true ? 'passed' : 'failed'
};
local.certifiedProfile = production.certifiedProfile ?? local.certifiedProfile;
local.gates.push(...(production.gates ?? []));
local.promotion = {
ready: production.passed === true && production.realCanaryPassed === true &&
local.gates.every((gate) => gate.passed === true),
requiresManualConfirmation: true,
trafficMode: 'validation'
};
local.createdAt = new Date().toISOString();
validate(local);
await mkdir(dirname(resolve(values.output)), { recursive: true, mode: 0o700 });
await writeFile(resolve(values.output), `${JSON.stringify(local, null, 2)}\n`, { mode: 0o600 });
process.stdout.write(`acceptance_report_production=PASS promotion_ready=${local.promotion.ready}\n`);
}
const { command, values } = parseArgs(process.argv.slice(2));
if (command === 'validate') {
const report = await regularJSON(values.input ?? '');
validate(report);
process.stdout.write('acceptance_report_validate=PASS\n');
} else if (command === 'build-local') {
await buildLocal(values);
} else if (command === 'build-local-partial') {
await buildLocalPartial(values);
} else if (command === 'merge-production') {
await mergeProduction(values);
} else if (command === 'mark-promoted') {
const input = values.input ?? '';
const output = values.output ?? input;
const report = await regularJSON(input);
validate(report);
if (report.promotion.ready !== true || report.promotion.trafficMode !== 'validation') {
throw new Error('only a ready validation report can be marked promoted');
}
report.promotion.trafficMode = 'live';
report.promotion.promotedAt = new Date().toISOString();
await writeFile(resolve(output), `${JSON.stringify(report, null, 2)}\n`, { mode: 0o600 });
process.stdout.write('acceptance_report_promoted=PASS\n');
} else if (command === 'attach-monitor') {
const input = values.input ?? '';
const output = values.output ?? input;
const report = await regularJSON(input);
const monitor = await regularJSON(values.monitor ?? '');
validate(report);
if (monitor.schemaVersion !== 'acceptance-monitor-report/v1' ||
monitor.runId !== report.runId || monitor.releaseSha !== report.release.sourceSha ||
monitor.secretSafe !== true) {
throw new Error('monitor report does not match overall acceptance report');
}
report.stages.postPromotionMonitor = {
status: monitor.passed === true ? 'passed' : 'failed',
startedAt: monitor.startedAt,
finishedAt: monitor.finishedAt,
samples: monitor.samples,
failureGateId: monitor.failureGateId
};
if (monitor.passed !== true) {
report.promotion.ready = false;
report.promotion.trafficMode = 'validation';
}
await writeFile(resolve(output), `${JSON.stringify(report, null, 2)}\n`, { mode: 0o600 });
process.stdout.write(`acceptance_report_monitor=PASS monitor_passed=${monitor.passed === true}\n`);
} else {
throw new Error('usage: report.mjs {validate|build-local|build-local-partial|merge-production|mark-promoted|attach-monitor} [options]');
}
+536
View File
@@ -0,0 +1,536 @@
#!/usr/bin/env bash
set -euo pipefail
script_dir=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)
repository_root=$(cd "$script_dir/../.." && pwd)
private_root="$repository_root/.local-secrets/acceptance"
state_root="$private_root/state"
context=k3d-easyai-acceptance-local
namespace=easyai
runtime_path_file="$state_root/current-runtime-path"
snapshot="$state_root/snapshot.json"
load_binary="$state_root/easyai-ai-gateway-acceptance-load"
active_load_pid=
netem_active=false
current_phase=initialization
stable_profile=
usage() {
cat <<'EOF'
Usage:
scripts/acceptance/run-local-acceptance.sh quick
scripts/acceptance/run-local-acceptance.sh full --release-manifest dist/releases/<SHA>.json
scripts/acceptance/run-local-acceptance.sh artifact-smoke --release-manifest dist/releases/<SHA>.json
`full` executes P24/P28/P32 three times, the fault matrix, autoscaling/drain,
80% soak, 120% overload, and exact linux/amd64 artifact smoke. The load process
runs outside K3s and splits requests 50/50 across both TLS entrances.
EOF
}
private_file() {
local path=$1 mode
[[ -f $path && ! -L $path ]] || return 1
if [[ $(uname -s) == Darwin ]]; then
mode=$(stat -f '%Lp' "$path")
else
mode=$(stat -c '%a' "$path")
fi
[[ $mode == 600 ]]
}
cleanup() {
local status=$?
trap - EXIT
if [[ -n $active_load_pid ]]; then
kill "$active_load_pid" >/dev/null 2>&1 || true
wait "$active_load_pid" >/dev/null 2>&1 || true
fi
if [[ $netem_active == true ]]; then
"$script_dir/network-fault.sh" reset >/dev/null 2>&1 || true
fi
if [[ $status -ne 0 && -n ${runtime:-} && -n ${report_root:-} && -f ${runtime:-} && -f $snapshot ]]; then
restore_profile_best_effort "${stable_profile:-P24}"
local partial_args=(
"$script_dir/report.mjs" build-local-partial
--runtime "$runtime" \
--snapshot "$snapshot" \
--reports "$report_root" \
--failure-phase "$current_phase" \
--output "$report_root/acceptance-report.partial.json"
)
if [[ -n $stable_profile ]]; then
partial_args+=(--certified-profile "$stable_profile")
fi
if node "${partial_args[@]}" >/dev/null 2>&1; then
echo "local_acceptance=FAILED phase=$current_phase partial_report=$report_root/acceptance-report.partial.json" >&2
else
echo "local_acceptance=FAILED phase=$current_phase partial_report=unavailable" >&2
fi
fi
exit "$status"
}
trap cleanup EXIT
trap 'current_phase=signal_interrupted; exit 130' HUP INT TERM
require_local_cluster() {
command -v kubectl >/dev/null 2>&1
command -v jq >/dev/null 2>&1
command -v node >/dev/null 2>&1
[[ $(kubectl config get-contexts "$context" -o name) == "$context" ]]
private_file "$runtime_path_file" && private_file "$snapshot"
runtime=$(<"$runtime_path_file")
private_file "$runtime" || {
echo 'local acceptance runtime file is missing or not mode 0600' >&2
exit 1
}
jq -e '.schemaVersion == "acceptance-runtime/v1"' "$runtime" >/dev/null
run_id=$(jq -r '.runId' "$runtime")
release_sha=$(jq -r '.releaseSha' "$runtime")
report_root="$repository_root/dist/acceptance/local/$run_id"
install -d -m 0700 "$report_root"
(
cd "$repository_root/apps/api"
env -u AI_GATEWAY_TEST_DATABASE_URL go build -trimpath \
-o "$load_binary" ./cmd/acceptance-load
)
}
database_query() {
local sql=$1 primary
primary=$(kubectl --context "$context" -n "$namespace" get cluster easyai-postgres \
-o 'jsonpath={.status.currentPrimary}')
kubectl --context "$context" -n "$namespace" exec "$primary" -c postgres -- \
psql -X -v ON_ERROR_STOP=1 -U postgres -d easyai_ai_gateway -At -c "$sql"
}
load_environment() {
acceptance_api_keys=$(jq -r '.apiKeys | join(",")' "$runtime")
acceptance_run_token=$(jq -r '.runToken' "$runtime")
acceptance_gemini_model=$(jq -r '.geminiModel' "$runtime")
acceptance_video_model=$(jq -r '.videoModel' "$runtime")
acceptance_emulator_url=$(jq -r '.emulatorBaseUrl' "$runtime")
}
run_load() {
local profile=$1 report_path=$2
shift 2
AI_GATEWAY_ACCEPTANCE_GATEWAYS='https://127.0.0.1:18443,https://127.0.0.1:19443' \
AI_GATEWAY_ACCEPTANCE_GATEWAY_TLS_SERVER_NAME=gateway.easyai.local \
AI_GATEWAY_ACCEPTANCE_GATEWAY_CA_FILE="$state_root/ca.crt" \
AI_GATEWAY_ACCEPTANCE_EMULATOR_URL="$acceptance_emulator_url" \
AI_GATEWAY_ACCEPTANCE_API_KEYS="$acceptance_api_keys" \
AI_GATEWAY_ACCEPTANCE_RUN_ID="$run_id" \
AI_GATEWAY_ACCEPTANCE_RUN_TOKEN="$acceptance_run_token" \
AI_GATEWAY_ACCEPTANCE_GEMINI_MODEL="$acceptance_gemini_model" \
AI_GATEWAY_ACCEPTANCE_VIDEO_MODEL="$acceptance_video_model" \
"$load_binary" -profile "$profile" -report "$report_path" "$@" \
>"$report_path.stdout"
chmod 0600 "$report_path" "$report_path.stdout"
jq -e '.schemaVersion == "acceptance-load-report/v1" and .secretSafe == true and .passed == true' \
"$report_path" >/dev/null
}
sample_resources() {
local output=$1 stop_file=$2
printf 'timestamp,scope,name,cpu,memory\n' >"$output"
while [[ ! -e $stop_file ]]; do
kubectl --context "$context" top nodes --no-headers 2>/dev/null |
awk -v timestamp="$(date -u '+%Y-%m-%dT%H:%M:%SZ')" \
'{print timestamp ",node," $1 "," $3 "," $5}' >>"$output" || true
kubectl --context "$context" -n "$namespace" top pods --no-headers 2>/dev/null |
awk -v timestamp="$(date -u '+%Y-%m-%dT%H:%M:%SZ')" \
'{print timestamp ",pod," $1 "," $2 "," $3}' >>"$output" || true
sleep 5
done
}
verify_hard_gates() {
local unhealthy ready sync_state connections max_connections queue duplicate_remote duplicate_billing
ready=$(kubectl --context "$context" get nodes -o json |
jq '[.items[] | select(any(.status.conditions[]; .type=="Ready" and .status=="True"))] | length')
[[ $ready == 3 ]]
[[ $(kubectl --context "$context" get nodes -o json |
jq '[.items[] | select(any(.status.conditions[]; .type=="MemoryPressure" and .status=="True"))] | length') == 0 ]]
[[ $(kubectl --context "$context" -n "$namespace" get cluster easyai-postgres \
-o 'jsonpath={.status.readyInstances}') == 2 ]]
sync_state=$(database_query "SELECT COALESCE(string_agg(sync_state,','),'') FROM pg_stat_replication;")
[[ ",$sync_state," == *,sync,* || ",$sync_state," == *,quorum,* ]]
unhealthy=$(kubectl --context "$context" -n "$namespace" get pods -o json |
jq '[.items[]
| select(.metadata.labels["app.kubernetes.io/name"] == "easyai-api" or
.metadata.labels["app.kubernetes.io/name"] == "easyai-worker")
| .status.containerStatuses[]?
| select(.restartCount > 0 or .lastState.terminated.reason == "OOMKilled" or
.state.terminated.reason == "OOMKilled")] | length')
[[ $unhealthy == 0 ]]
while read -r _node _cpu _cpu_percent _memory memory_percent; do
memory_percent=${memory_percent%\%}
[[ $memory_percent =~ ^[0-9]+$ ]]
(( memory_percent < 80 ))
done < <(kubectl --context "$context" top nodes --no-headers)
while read -r _pod _cpu memory; do
case $memory in
*Gi) memory=$(awk -v value="${memory%Gi}" 'BEGIN {printf "%.0f", value*1024}') ;;
*Mi) memory=${memory%Mi} ;;
*Ki) memory=$(awk -v value="${memory%Ki}" 'BEGIN {printf "%.0f", value/1024}') ;;
*) return 1 ;;
esac
(( memory < 1536 ))
done < <(kubectl --context "$context" -n "$namespace" top pods \
-l 'app.kubernetes.io/part-of=easyai-ai-gateway' --no-headers)
connections=$(database_query "SELECT count(*) FROM pg_stat_activity WHERE backend_type='client backend';")
max_connections=$(database_query "SELECT setting::int FROM pg_settings WHERE name='max_connections';")
(( connections < 150 && connections * 4 < max_connections * 3 ))
queue=$(database_query "SELECT count(*) FILTER (WHERE status='queued')||':'||count(*) FILTER (WHERE status='running') FROM gateway_tasks WHERE acceptance_run_id='$run_id'::uuid;")
[[ $queue == 0:0 ]]
duplicate_remote=$(database_query "SELECT count(*) FROM (SELECT remote_task_id FROM gateway_tasks WHERE acceptance_run_id='$run_id'::uuid AND remote_task_id IS NOT NULL GROUP BY remote_task_id HAVING count(*)>1) d;")
duplicate_billing=$(database_query "SELECT count(*) FROM (SELECT reference_id,transaction_type FROM gateway_wallet_transactions WHERE reference_type='gateway_task' AND reference_id IN (SELECT id::text FROM gateway_tasks WHERE acceptance_run_id='$run_id'::uuid) GROUP BY reference_id,transaction_type HAVING count(*)>1) d;")
[[ $duplicate_remote == 0 && $duplicate_billing == 0 ]]
kubectl --context "$context" -n "$namespace" exec deployment/easyai-acceptance-callback-collector -- \
wget -qO- http://127.0.0.1:8091/report |
jq -e '.duplicates == 0 and .invalid == 0' >/dev/null
}
apply_profile() {
local profile=$1 slots pool
case $profile in
P24) slots=24; pool=32 ;;
P28) slots=28; pool=36 ;;
P32) slots=32; pool=40 ;;
*) return 64 ;;
esac
local global=$((slots * 2))
kubectl --context "$context" -n "$namespace" patch configmap easyai-ai-gateway-config \
--type=merge -p "$(jq -cn \
--arg slots "$slots" --arg global "$global" \
'{data:{
AI_GATEWAY_WORKER_AUTOSCALING_ENABLED:"false",
AI_GATEWAY_ASYNC_WORKER_INSTANCE_HARD_LIMIT:$slots,
AI_GATEWAY_ASYNC_WORKER_HARD_LIMIT:$slots,
AI_GATEWAY_ASYNC_WORKER_GLOBAL_HARD_LIMIT:$global,
AI_GATEWAY_WORKER_TARGET_OUTSTANDING_PER_REPLICA:$global
}}')" >/dev/null
kubectl --context "$context" -n "$namespace" scale \
deployment/easyai-worker-ningbo deployment/easyai-worker-hongkong --replicas=1 >/dev/null
for deployment in easyai-worker-ningbo easyai-worker-hongkong; do
kubectl --context "$context" -n "$namespace" set env deployment/"$deployment" \
AI_GATEWAY_ASYNC_WORKER_INSTANCE_HARD_LIMIT="$slots" \
AI_GATEWAY_DATABASE_MAX_CONNS="$pool" \
AI_GATEWAY_MEDIA_MATERIALIZATION_CONCURRENCY="$slots" \
AI_GATEWAY_MEDIA_REQUEST_CONCURRENCY="$slots" >/dev/null
kubectl --context "$context" -n "$namespace" rollout status deployment/"$deployment" --timeout=10m
done
for deployment in easyai-api-ningbo easyai-api-hongkong easyai-capacity-controller; do
kubectl --context "$context" -n "$namespace" rollout restart deployment/"$deployment" >/dev/null
kubectl --context "$context" -n "$namespace" rollout status deployment/"$deployment" --timeout=10m
done
}
restore_profile_best_effort() {
local profile=$1 slots pool global
case $profile in
P24) slots=24; pool=32 ;;
P28) slots=28; pool=36 ;;
P32) slots=32; pool=40 ;;
*) return ;;
esac
global=$((slots * 2))
kubectl --context "$context" -n "$namespace" patch configmap easyai-ai-gateway-config \
--type=merge -p "$(jq -cn \
--arg slots "$slots" --arg global "$global" \
'{data:{
AI_GATEWAY_WORKER_AUTOSCALING_ENABLED:"false",
AI_GATEWAY_ASYNC_WORKER_INSTANCE_HARD_LIMIT:$slots,
AI_GATEWAY_ASYNC_WORKER_HARD_LIMIT:$slots,
AI_GATEWAY_ASYNC_WORKER_GLOBAL_HARD_LIMIT:$global,
AI_GATEWAY_WORKER_TARGET_OUTSTANDING_PER_REPLICA:$global
}}')" >/dev/null 2>&1 || true
kubectl --context "$context" -n "$namespace" scale \
deployment/easyai-worker-ningbo deployment/easyai-worker-hongkong \
--replicas=1 >/dev/null 2>&1 || true
for deployment in easyai-worker-ningbo easyai-worker-hongkong; do
kubectl --context "$context" -n "$namespace" set env deployment/"$deployment" \
AI_GATEWAY_ASYNC_WORKER_INSTANCE_HARD_LIMIT="$slots" \
AI_GATEWAY_DATABASE_MAX_CONNS="$pool" \
AI_GATEWAY_MEDIA_MATERIALIZATION_CONCURRENCY="$slots" \
AI_GATEWAY_MEDIA_REQUEST_CONCURRENCY="$slots" >/dev/null 2>&1 || true
done
}
run_profile_round() {
local profile=$1 repetition=$2 prefix
prefix="$report_root/${profile,,}-$repetition"
local stop_file="$prefix.resources.stop"
sample_resources "$prefix.resources.csv" "$stop_file" &
local sampler=$!
run_load gemini-baseline "$prefix-gemini-baseline.json"
run_load gemini-large "$prefix-gemini-large.json"
run_load gemini-peak "$prefix-gemini-peak.json"
run_load video-throughput "$prefix-video-throughput.json"
run_recovery "$prefix-video-recovery.json"
touch "$stop_file"
wait "$sampler"
verify_hard_gates
}
wait_for_recovery_owner() {
local deadline=$((SECONDS + 120)) owner
while (( SECONDS < deadline )); do
owner=$(database_query "
SELECT task.id::text||','||worker.pod_name||','||worker.site
FROM gateway_tasks task
JOIN gateway_worker_instances worker ON worker.instance_id=task.locked_by
WHERE task.acceptance_run_id='$run_id'::uuid
AND task.status='running'
AND task.request::text LIKE '%acceptance-long-recovery%'
AND worker.status='active'
ORDER BY task.updated_at
LIMIT 1;")
[[ -z $owner ]] || { printf '%s\n' "$owner"; return; }
sleep 1
done
return 1
}
run_recovery() {
local report_path=$1 owner task_id pod site
run_load video-recovery "$report_path" &
active_load_pid=$!
owner=$(wait_for_recovery_owner)
IFS=',' read -r task_id pod site <<<"$owner"
[[ $task_id =~ ^[0-9a-f-]{36}$ && $pod == easyai-worker-"$site"-* ]]
kubectl --context "$context" -n "$namespace" delete pod "$pod" \
--force --grace-period=0 --wait=false >/dev/null
kubectl --context "$context" -n "$namespace" rollout status \
deployment/easyai-worker-"$site" --timeout=10m
wait "$active_load_pid"
active_load_pid=
}
run_fault_matrix() {
local duration=${AI_GATEWAY_LOCAL_WEAK_LINK_DURATION:-5m}
"$script_dir/network-fault.sh" weak-link
netem_active=true
run_load mixed-soak "$report_root/fault-weak-link.json" \
-duration "$duration" -image-rate 1 -video-rate 1
"$script_dir/network-fault.sh" reset
netem_active=false
run_load video-throughput "$report_root/fault-upstream-outage.json" &
active_load_pid=$!
sleep 5
"$script_dir/network-fault.sh" upstream-outage
netem_active=true
sleep 10
"$script_dir/network-fault.sh" reset
netem_active=false
wait "$active_load_pid"
active_load_pid=
run_load video-recovery "$report_root/fault-database-outage.json" &
active_load_pid=$!
sleep 10
"$script_dir/network-fault.sh" database-outage hongkong
netem_active=true
sleep 30
"$script_dir/network-fault.sh" reset
netem_active=false
wait "$active_load_pid"
active_load_pid=
local leader
leader=$(kubectl --context "$context" -n "$namespace" get pods \
-l app.kubernetes.io/name=easyai-capacity-controller -o name |
while read -r pod; do
status=$(kubectl --context "$context" -n "$namespace" exec "$pod" -- \
wget -qO- http://127.0.0.1:8088/status)
[[ $(jq -r '.leader' <<<"$status") == true ]] && { printf '%s\n' "${pod#pod/}"; break; }
done)
[[ -n $leader ]]
kubectl --context "$context" -n "$namespace" delete pod "$leader" --wait=false >/dev/null
kubectl --context "$context" -n "$namespace" rollout status \
deployment/easyai-capacity-controller --timeout=5m
verify_hard_gates
}
run_autoscaling() {
local profile=$1 slots
slots=${profile#P}
kubectl --context "$context" -n "$namespace" patch configmap easyai-ai-gateway-config \
--type=merge -p "$(jq -cn --arg global "$((slots * 4))" \
'{data:{
AI_GATEWAY_WORKER_AUTOSCALING_ENABLED:"true",
AI_GATEWAY_WORKER_MIN_REPLICAS_NINGBO:"1",
AI_GATEWAY_WORKER_MIN_REPLICAS_HONGKONG:"1",
AI_GATEWAY_WORKER_MAX_REPLICAS_NINGBO:"2",
AI_GATEWAY_WORKER_MAX_REPLICAS_HONGKONG:"2",
AI_GATEWAY_ASYNC_WORKER_GLOBAL_HARD_LIMIT:$global
}}')" >/dev/null
kubectl --context "$context" -n "$namespace" rollout restart \
deployment/easyai-capacity-controller >/dev/null
kubectl --context "$context" -n "$namespace" rollout status \
deployment/easyai-capacity-controller --timeout=5m
run_load video-throughput "$report_root/autoscaling-load.json" &
active_load_pid=$!
local deadline=$((SECONDS + 420)) total=2
while (( SECONDS < deadline )); do
total=$(kubectl --context "$context" -n "$namespace" get deployment \
easyai-worker-ningbo easyai-worker-hongkong \
-o json | jq '[.items[].spec.replicas] | add')
(( total >= 3 )) && break
sleep 5
done
(( total >= 3 ))
wait "$active_load_pid"
active_load_pid=
deadline=$((SECONDS + 780))
while (( SECONDS < deadline )); do
total=$(kubectl --context "$context" -n "$namespace" get deployment \
easyai-worker-ningbo easyai-worker-hongkong \
-o json | jq '[.items[].spec.replicas] | add')
(( total == 2 )) && break
sleep 10
done
(( total == 2 ))
verify_hard_gates
}
artifact_smoke() {
local manifest=$1
node "$repository_root/scripts/release-manifest.mjs" validate "$manifest" >/dev/null
local manifest_sha api_image web_image api_digest artifact_report
manifest_sha=$(node "$repository_root/scripts/release-manifest.mjs" get "$manifest" sourceSha)
api_image=$(node "$repository_root/scripts/release-manifest.mjs" get "$manifest" images.api)
web_image=$(node "$repository_root/scripts/release-manifest.mjs" get "$manifest" images.web)
api_digest=${api_image##*@}
[[ $manifest_sha == "$release_sha" && $api_digest =~ ^sha256:[0-9a-f]{64}$ ]]
docker pull --platform linux/amd64 "$api_image" >/dev/null
docker pull --platform linux/amd64 "$web_image" >/dev/null
"$private_root/tools/k3d" image import -c easyai-acceptance-local "$api_image" "$web_image"
for deployment in easyai-api-ningbo easyai-api-hongkong easyai-worker-ningbo \
easyai-worker-hongkong easyai-capacity-controller easyai-acceptance-emulator \
easyai-acceptance-callback-collector; do
kubectl --context "$context" -n "$namespace" set image deployment/"$deployment" \
"*=$api_image" >/dev/null
kubectl --context "$context" -n "$namespace" rollout status deployment/"$deployment" --timeout=15m
done
for deployment in easyai-acceptance-edge-ningbo easyai-acceptance-edge-hongkong; do
kubectl --context "$context" -n "$namespace" set image deployment/"$deployment" \
"*=$web_image" >/dev/null
kubectl --context "$context" -n "$namespace" rollout status deployment/"$deployment" --timeout=10m
done
kubectl --context "$context" -n "$namespace" delete job easyai-artifact-migrate \
--ignore-not-found --wait=true >/dev/null
cat <<EOF | kubectl --context "$context" apply -f - >/dev/null
apiVersion: batch/v1
kind: Job
metadata:
name: easyai-artifact-migrate
namespace: easyai
spec:
backoffLimit: 0
template:
spec:
restartPolicy: Never
containers:
- name: migrate
image: $api_image
command: ["/bin/sh", "-ec", "cd /app && exec /app/easyai-ai-gateway-migrate"]
envFrom:
- secretRef:
name: easyai-ai-gateway-runtime
EOF
kubectl --context "$context" -n "$namespace" wait \
--for=condition=complete job/easyai-artifact-migrate --timeout=10m >/dev/null
run_load simulated-smoke "$report_root/artifact-simulated-smoke.json"
verify_hard_gates
artifact_report="$report_root/artifact-smoke.json"
jq -n \
--arg releaseSha "$manifest_sha" \
--arg apiImageDigest "$api_digest" \
--arg webImageDigest "${web_image##*@}" \
'{
schemaVersion:"acceptance-artifact-smoke/v1",
releaseSha:$releaseSha,
apiImageDigest:$apiImageDigest,
webImageDigest:$webImageDigest,
architecture:"linux/amd64",
migrationSmoke:true,
startupSmoke:true,
mediaSmoke:true,
passed:true,
secretSafe:true
}' >"$artifact_report"
chmod 0600 "$artifact_report"
}
quick() {
current_phase=quick
run_load simulated-smoke "$report_root/quick.json"
verify_hard_gates
echo "local_acceptance_quick=PASS run_id=$run_id report=$report_root/quick.json"
}
full() {
local manifest=$1 profile repetition
quick
for profile in P24 P28 P32; do
current_phase="capacity_${profile,,}_apply"
apply_profile "$profile"
for repetition in 1 2 3; do
current_phase="capacity_${profile,,}_round_$repetition"
run_profile_round "$profile" "$repetition"
done
stable_profile=$profile
done
current_phase=fault_matrix
run_fault_matrix
current_phase=autoscaling_and_drain
run_autoscaling "$stable_profile"
current_phase=certified_soak
run_load mixed-soak "$report_root/mixed-soak.json" \
-duration "${AI_GATEWAY_LOCAL_SOAK_DURATION:-2h}" \
-image-rate "${AI_GATEWAY_LOCAL_CERTIFIED_IMAGE_RATE:-1}" \
-video-rate "${AI_GATEWAY_LOCAL_CERTIFIED_VIDEO_RATE:-1}"
current_phase=overload_shedding
run_load mixed-overload "$report_root/mixed-overload.json" \
-duration "${AI_GATEWAY_LOCAL_OVERLOAD_DURATION:-10m}" \
-image-rate "${AI_GATEWAY_LOCAL_OVERLOAD_IMAGE_RATE:-2}" \
-video-rate "${AI_GATEWAY_LOCAL_OVERLOAD_VIDEO_RATE:-2}"
current_phase=amd64_artifact_smoke
artifact_smoke "$manifest"
current_phase=final_report
node "$script_dir/report.mjs" build-local \
--runtime "$runtime" \
--snapshot "$snapshot" \
--reports "$report_root" \
--artifact "$report_root/artifact-smoke.json" \
--api-digest "$(node "$repository_root/scripts/release-manifest.mjs" get "$manifest" images.api | sed 's/.*@//')" \
--certified-profile "$stable_profile" \
--output "$report_root/acceptance-report.json"
echo "local_acceptance_full=PASS run_id=$run_id certified_profile=$stable_profile report=$report_root/acceptance-report.json"
}
command=${1:-}
shift || true
require_local_cluster
load_environment
case $command in
quick)
[[ $# -eq 0 ]] || { usage >&2; exit 64; }
quick
;;
artifact-smoke)
[[ ${1:-} == --release-manifest && $# -eq 2 ]] || { usage >&2; exit 64; }
artifact_smoke "$2"
;;
full)
[[ ${1:-} == --release-manifest && $# -eq 2 ]] || { usage >&2; exit 64; }
full "$2"
;;
*)
usage >&2
exit 64
;;
esac