feat(acceptance): 建立同构验收与弹性容量体系
实现本地三节点 K3s 同构环境、脱敏生产快照、Gemini 图片和多参考图视频协议模拟、统一验收报告及故障注入。\n\n新增 Worker 容量控制器、资源与连接预算、任务恢复保护,并将生产验收拆分为 validation 执行和人工 CAS 放量。\n\n验证包括 Go 全量测试、PostgreSQL HTTP 集成测试、go vet、OpenAPI、ShellCheck、前端检查、迁移及发布脚本测试。
This commit is contained in:
+43
@@ -0,0 +1,43 @@
|
||||
#!/usr/bin/env bash
|
||||
set -euo pipefail
|
||||
|
||||
usage() {
|
||||
cat <<'EOF'
|
||||
Usage:
|
||||
AI_GATEWAY_ACCEPTANCE_SNAPSHOT_DATABASE_URL='postgresql://...' \
|
||||
scripts/acceptance/export-production-snapshot.sh \
|
||||
--release-sha <full-sha> --output <snapshot.json>
|
||||
|
||||
This command is database read-only. The supplied role should have SELECT access
|
||||
only. It never prints the database URL or candidate credentials.
|
||||
EOF
|
||||
}
|
||||
|
||||
script_dir=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)
|
||||
repository_root=$(cd "$script_dir/../.." && pwd)
|
||||
|
||||
[[ ${1:-} == --release-sha && ${3:-} == --output && $# -eq 4 ]] || {
|
||||
usage >&2
|
||||
exit 64
|
||||
}
|
||||
release_sha=$2
|
||||
output=$4
|
||||
database_url=${AI_GATEWAY_ACCEPTANCE_SNAPSHOT_DATABASE_URL:-}
|
||||
|
||||
[[ $release_sha =~ ^[0-9a-f]{40}$ ]] || {
|
||||
echo 'release SHA must be a full lowercase Git SHA' >&2
|
||||
exit 64
|
||||
}
|
||||
[[ -n $database_url ]] || {
|
||||
echo 'AI_GATEWAY_ACCEPTANCE_SNAPSHOT_DATABASE_URL is required' >&2
|
||||
exit 64
|
||||
}
|
||||
umask 077
|
||||
(
|
||||
cd "$repository_root/apps/api"
|
||||
AI_GATEWAY_DATABASE_URL="$database_url" \
|
||||
go run ./cmd/acceptance-snapshot export \
|
||||
--release-sha "$release_sha" \
|
||||
--output "$output"
|
||||
)
|
||||
chmod 0600 "$output"
|
||||
Executable
+556
@@ -0,0 +1,556 @@
|
||||
#!/usr/bin/env bash
|
||||
set -euo pipefail
|
||||
|
||||
script_dir=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)
|
||||
repository_root=$(cd "$script_dir/../.." && pwd)
|
||||
manifest_root="$repository_root/deploy/kubernetes/local-acceptance"
|
||||
lock_file="$manifest_root/dependencies.lock"
|
||||
private_root="$repository_root/.local-secrets/acceptance"
|
||||
tools_root="$private_root/tools"
|
||||
state_root="$private_root/state"
|
||||
media_root="$private_root/media"
|
||||
cluster_name=easyai-acceptance-local
|
||||
context="k3d-$cluster_name"
|
||||
namespace=easyai
|
||||
|
||||
usage() {
|
||||
cat <<'EOF'
|
||||
Usage:
|
||||
scripts/acceptance/local-cluster.sh install-tools
|
||||
scripts/acceptance/local-cluster.sh preflight
|
||||
scripts/acceptance/local-cluster.sh up --snapshot <acceptance-snapshot-v1.json>
|
||||
scripts/acceptance/local-cluster.sh status
|
||||
scripts/acceptance/local-cluster.sh down --confirm
|
||||
|
||||
The full cluster requires Docker Desktop memory >= 24 GiB. Failed `up` runs are
|
||||
kept intact for diagnosis. `down` is the only destructive lifecycle action.
|
||||
EOF
|
||||
}
|
||||
|
||||
load_lock() {
|
||||
[[ -f $lock_file && ! -L $lock_file ]] || {
|
||||
echo "missing dependency lock: $lock_file" >&2
|
||||
exit 1
|
||||
}
|
||||
# shellcheck source=/dev/null
|
||||
source "$lock_file"
|
||||
: "${K3D_VERSION:?}" "${K3S_IMAGE:?}" "${CNPG_MANIFEST_URL:?}" "${CNPG_MANIFEST_SHA256:?}"
|
||||
}
|
||||
|
||||
require_command() {
|
||||
command -v "$1" >/dev/null 2>&1 || {
|
||||
echo "$1 is required" >&2
|
||||
exit 1
|
||||
}
|
||||
}
|
||||
|
||||
private_file() {
|
||||
local path=$1
|
||||
[[ -f $path && ! -L $path ]] || return 1
|
||||
local mode
|
||||
if [[ $(uname -s) == Darwin ]]; then
|
||||
mode=$(stat -f '%Lp' "$path")
|
||||
else
|
||||
mode=$(stat -c '%a' "$path")
|
||||
fi
|
||||
[[ $mode == 600 ]]
|
||||
}
|
||||
|
||||
install_tools() {
|
||||
load_lock
|
||||
require_command curl
|
||||
require_command shasum
|
||||
install -d -m 0700 "$tools_root"
|
||||
local os arch checksum url temporary
|
||||
os=$(uname -s | tr '[:upper:]' '[:lower:]')
|
||||
arch=$(uname -m)
|
||||
[[ $os == darwin ]] || {
|
||||
echo 'the pinned local acceptance installer currently supports macOS only' >&2
|
||||
exit 1
|
||||
}
|
||||
case $arch in
|
||||
arm64)
|
||||
arch=arm64
|
||||
checksum=$K3D_DARWIN_ARM64_SHA256
|
||||
;;
|
||||
x86_64)
|
||||
arch=amd64
|
||||
checksum=$K3D_DARWIN_AMD64_SHA256
|
||||
;;
|
||||
*)
|
||||
echo "unsupported host architecture: $arch" >&2
|
||||
exit 1
|
||||
;;
|
||||
esac
|
||||
url="https://github.com/k3d-io/k3d/releases/download/$K3D_VERSION/k3d-darwin-$arch"
|
||||
temporary="$tools_root/k3d.download"
|
||||
curl -fsSL "$url" -o "$temporary"
|
||||
[[ $(shasum -a 256 "$temporary" | awk '{print $1}') == "$checksum" ]] || {
|
||||
echo 'k3d checksum mismatch' >&2
|
||||
exit 1
|
||||
}
|
||||
chmod 0555 "$temporary"
|
||||
mv "$temporary" "$tools_root/k3d"
|
||||
echo "local_acceptance_tools=PASS k3d=$K3D_VERSION path=$tools_root/k3d"
|
||||
}
|
||||
|
||||
k3d_binary() {
|
||||
if [[ -x $tools_root/k3d ]]; then
|
||||
printf '%s\n' "$tools_root/k3d"
|
||||
return
|
||||
fi
|
||||
command -v k3d
|
||||
}
|
||||
|
||||
docker_memory_bytes() {
|
||||
docker info --format '{{.MemTotal}}'
|
||||
}
|
||||
|
||||
preflight() {
|
||||
load_lock
|
||||
require_command docker
|
||||
require_command kubectl
|
||||
require_command jq
|
||||
require_command openssl
|
||||
require_command sed
|
||||
require_command shasum
|
||||
require_command go
|
||||
require_command node
|
||||
require_command curl
|
||||
local k3d memory_bytes architecture
|
||||
k3d=$(k3d_binary) || {
|
||||
echo 'k3d is missing; run install-tools' >&2
|
||||
exit 1
|
||||
}
|
||||
"$k3d" version | grep -Fq "$K3D_VERSION" || {
|
||||
echo "k3d must be pinned to $K3D_VERSION" >&2
|
||||
exit 1
|
||||
}
|
||||
docker info >/dev/null
|
||||
memory_bytes=$(docker_memory_bytes)
|
||||
[[ $memory_bytes =~ ^[0-9]+$ && $memory_bytes -ge 25769803776 ]] || {
|
||||
memory_gib=$(awk -v bytes="${memory_bytes:-0}" 'BEGIN {printf "%.1f", bytes/1024/1024/1024}')
|
||||
echo "Docker Desktop memory is ${memory_gib} GiB; local acceptance requires at least 24 GiB" >&2
|
||||
exit 1
|
||||
}
|
||||
architecture=$(docker info --format '{{.Architecture}}')
|
||||
[[ $architecture == aarch64 || $architecture == arm64 || $architecture == x86_64 ]] || {
|
||||
echo "unsupported Docker architecture: $architecture" >&2
|
||||
exit 1
|
||||
}
|
||||
docker buildx version >/dev/null
|
||||
echo "local_acceptance_preflight=PASS docker_memory_bytes=$memory_bytes architecture=$architecture k3d=$K3D_VERSION"
|
||||
}
|
||||
|
||||
ensure_private_material() {
|
||||
install -d -m 0700 "$private_root" "$state_root" "$media_root"
|
||||
local env_file="$state_root/local.env"
|
||||
if [[ ! -e $env_file ]]; then
|
||||
umask 077
|
||||
{
|
||||
printf 'LOCAL_CLUSTER_ID=%s\n' "easyai-local-$(openssl rand -hex 12)"
|
||||
printf 'POSTGRES_PASSWORD=%s\n' "$(openssl rand -hex 32)"
|
||||
printf 'CONFIG_JWT_SECRET=%s\n' "$(openssl rand -hex 32)"
|
||||
printf 'SERVER_MAIN_INTERNAL_TOKEN=%s\n' "$(openssl rand -hex 32)"
|
||||
printf 'SERVER_MAIN_INTERNAL_KEY=local-acceptance\n'
|
||||
printf 'SERVER_MAIN_INTERNAL_SECRET=%s\n' "$(openssl rand -hex 32)"
|
||||
} >"$env_file"
|
||||
chmod 0600 "$env_file"
|
||||
fi
|
||||
private_file "$env_file" || {
|
||||
echo "local acceptance state must be a 0600 regular file: $env_file" >&2
|
||||
exit 1
|
||||
}
|
||||
}
|
||||
|
||||
generate_tls() {
|
||||
local ca_key="$state_root/ca.key" ca_crt="$state_root/ca.crt"
|
||||
local tls_key="$state_root/tls.key" tls_csr="$state_root/tls.csr"
|
||||
local tls_crt="$state_root/tls.crt" extensions="$state_root/tls.ext"
|
||||
if [[ ! -e $ca_key ]]; then
|
||||
umask 077
|
||||
openssl genrsa -out "$ca_key" 3072 >/dev/null 2>&1
|
||||
openssl req -x509 -new -key "$ca_key" -sha256 -days 30 \
|
||||
-subj '/CN=EasyAI local acceptance CA' -out "$ca_crt" >/dev/null 2>&1
|
||||
openssl genrsa -out "$tls_key" 2048 >/dev/null 2>&1
|
||||
openssl req -new -key "$tls_key" -subj '/CN=gateway.easyai.local' \
|
||||
-out "$tls_csr" >/dev/null 2>&1
|
||||
printf 'subjectAltName=DNS:gateway.easyai.local\nextendedKeyUsage=serverAuth\n' >"$extensions"
|
||||
openssl x509 -req -in "$tls_csr" -CA "$ca_crt" -CAkey "$ca_key" \
|
||||
-CAcreateserial -out "$tls_crt" -days 30 -sha256 -extfile "$extensions" \
|
||||
>/dev/null 2>&1
|
||||
chmod 0600 "$ca_key" "$ca_crt" "$tls_key" "$tls_csr" "$tls_crt" "$extensions"
|
||||
fi
|
||||
private_file "$ca_crt" && private_file "$tls_key" && private_file "$tls_crt"
|
||||
}
|
||||
|
||||
render_k3d_config() {
|
||||
local output=$1
|
||||
[[ $media_root != *'|'* ]]
|
||||
sed "s|EASYAI_ACCEPTANCE_MEDIA_DIR|$media_root|g" "$manifest_root/k3d.yaml" >"$output"
|
||||
}
|
||||
|
||||
wait_rollout() {
|
||||
kubectl --context "$context" -n "$namespace" rollout status "$1" --timeout="${2:-10m}"
|
||||
}
|
||||
|
||||
apply_runtime_secrets() {
|
||||
# shellcheck source=/dev/null
|
||||
source "$state_root/local.env"
|
||||
local database_ningbo database_hongkong database_direct
|
||||
database_ningbo="postgresql://easyai:${POSTGRES_PASSWORD}@easyai-postgres-proxy-ningbo.easyai.svc.cluster.local:5432/easyai_ai_gateway?sslmode=require"
|
||||
database_hongkong="postgresql://easyai:${POSTGRES_PASSWORD}@easyai-postgres-proxy-hongkong.easyai.svc.cluster.local:5432/easyai_ai_gateway?sslmode=require"
|
||||
database_direct="postgresql://easyai:${POSTGRES_PASSWORD}@easyai-postgres-rw.easyai.svc.cluster.local:5432/easyai_ai_gateway?sslmode=require"
|
||||
|
||||
kubectl --context "$context" create namespace "$namespace" --dry-run=client -o yaml |
|
||||
kubectl --context "$context" apply -f - >/dev/null
|
||||
kubectl --context "$context" -n "$namespace" create secret generic easyai-postgres-app \
|
||||
--type=kubernetes.io/basic-auth \
|
||||
--from-literal=username=easyai --from-literal=password="$POSTGRES_PASSWORD" \
|
||||
--dry-run=client -o yaml | kubectl --context "$context" apply -f - >/dev/null
|
||||
kubectl --context "$context" -n "$namespace" create secret generic easyai-ai-gateway-runtime \
|
||||
--from-literal=AI_GATEWAY_DATABASE_URL="$database_direct" \
|
||||
--from-literal=CONFIG_JWT_SECRET="$CONFIG_JWT_SECRET" \
|
||||
--from-literal=SERVER_MAIN_INTERNAL_TOKEN="$SERVER_MAIN_INTERNAL_TOKEN" \
|
||||
--from-literal=SERVER_MAIN_INTERNAL_KEY="$SERVER_MAIN_INTERNAL_KEY" \
|
||||
--from-literal=SERVER_MAIN_INTERNAL_SECRET="$SERVER_MAIN_INTERNAL_SECRET" \
|
||||
--dry-run=client -o yaml | kubectl --context "$context" apply -f - >/dev/null
|
||||
kubectl --context "$context" -n "$namespace" create secret generic easyai-database-ningbo \
|
||||
--from-literal=AI_GATEWAY_DATABASE_URL="$database_ningbo" \
|
||||
--dry-run=client -o yaml | kubectl --context "$context" apply -f - >/dev/null
|
||||
kubectl --context "$context" -n "$namespace" create secret generic easyai-database-hongkong \
|
||||
--from-literal=AI_GATEWAY_DATABASE_URL="$database_hongkong" \
|
||||
--dry-run=client -o yaml | kubectl --context "$context" apply -f - >/dev/null
|
||||
kubectl --context "$context" -n "$namespace" create secret generic easyai-oss-backup \
|
||||
--from-literal=ACCESS_KEY_ID=local-disabled \
|
||||
--from-literal=ACCESS_SECRET_KEY=local-disabled \
|
||||
--dry-run=client -o yaml | kubectl --context "$context" apply -f - >/dev/null
|
||||
kubectl --context "$context" -n "$namespace" create secret tls easyai-acceptance-tls \
|
||||
--cert="$state_root/tls.crt" --key="$state_root/tls.key" \
|
||||
--dry-run=client -o yaml | kubectl --context "$context" apply -f - >/dev/null
|
||||
}
|
||||
|
||||
run_migrations() {
|
||||
local api_image=$1
|
||||
kubectl --context "$context" -n "$namespace" delete job easyai-local-migrate \
|
||||
--ignore-not-found --wait=true >/dev/null
|
||||
cat <<EOF | kubectl --context "$context" apply -f - >/dev/null
|
||||
apiVersion: batch/v1
|
||||
kind: Job
|
||||
metadata:
|
||||
name: easyai-local-migrate
|
||||
namespace: easyai
|
||||
spec:
|
||||
backoffLimit: 0
|
||||
template:
|
||||
spec:
|
||||
restartPolicy: Never
|
||||
containers:
|
||||
- name: migrate
|
||||
image: $api_image
|
||||
command: ["/bin/sh", "-ec", "cd /app && exec /app/easyai-ai-gateway-migrate"]
|
||||
envFrom:
|
||||
- secretRef:
|
||||
name: easyai-ai-gateway-runtime
|
||||
EOF
|
||||
kubectl --context "$context" -n "$namespace" wait \
|
||||
--for=condition=complete job/easyai-local-migrate --timeout=10m >/dev/null
|
||||
}
|
||||
|
||||
mark_local_database() {
|
||||
# shellcheck source=/dev/null
|
||||
source "$state_root/local.env"
|
||||
local primary
|
||||
primary=$(kubectl --context "$context" -n "$namespace" get cluster easyai-postgres \
|
||||
-o 'jsonpath={.status.currentPrimary}')
|
||||
kubectl --context "$context" -n "$namespace" exec "$primary" -c postgres -- \
|
||||
psql -X -v ON_ERROR_STOP=1 -U postgres -d easyai_ai_gateway \
|
||||
-v cluster_id="$LOCAL_CLUSTER_ID" -c \
|
||||
"INSERT INTO system_settings(setting_key,value)
|
||||
VALUES('acceptance_local_cluster_id',jsonb_build_object('clusterId', :'cluster_id'))
|
||||
ON CONFLICT(setting_key) DO UPDATE SET value=excluded.value,updated_at=now();" \
|
||||
>/dev/null
|
||||
}
|
||||
|
||||
import_snapshot() {
|
||||
local api_image=$1 snapshot=$2
|
||||
# shellcheck source=/dev/null
|
||||
source "$state_root/local.env"
|
||||
kubectl --context "$context" -n "$namespace" create configmap easyai-acceptance-snapshot \
|
||||
--from-file=snapshot.json="$snapshot" --dry-run=client -o yaml |
|
||||
kubectl --context "$context" apply -f - >/dev/null
|
||||
kubectl --context "$context" -n "$namespace" delete job easyai-local-snapshot-import \
|
||||
--ignore-not-found --wait=true >/dev/null
|
||||
cat <<EOF | kubectl --context "$context" apply -f - >/dev/null
|
||||
apiVersion: batch/v1
|
||||
kind: Job
|
||||
metadata:
|
||||
name: easyai-local-snapshot-import
|
||||
namespace: easyai
|
||||
spec:
|
||||
backoffLimit: 0
|
||||
template:
|
||||
spec:
|
||||
restartPolicy: Never
|
||||
containers:
|
||||
- name: import
|
||||
image: $api_image
|
||||
command:
|
||||
- /app/easyai-ai-gateway-acceptance-snapshot
|
||||
- import
|
||||
- --input
|
||||
- /snapshot/snapshot.json
|
||||
- --local-cluster-id
|
||||
- $LOCAL_CLUSTER_ID
|
||||
envFrom:
|
||||
- secretRef:
|
||||
name: easyai-ai-gateway-runtime
|
||||
volumeMounts:
|
||||
- name: snapshot
|
||||
mountPath: /snapshot
|
||||
readOnly: true
|
||||
volumes:
|
||||
- name: snapshot
|
||||
configMap:
|
||||
name: easyai-acceptance-snapshot
|
||||
EOF
|
||||
kubectl --context "$context" -n "$namespace" wait \
|
||||
--for=condition=complete job/easyai-local-snapshot-import --timeout=5m >/dev/null
|
||||
}
|
||||
|
||||
render_and_apply_application() {
|
||||
local api_image=$1 web_image=$2 rendered=$state_root/application.rendered.yaml
|
||||
sed \
|
||||
-e "s|image: easyai-api|image: $api_image|g" \
|
||||
-e "s|image: easyai-web|image: $web_image|g" \
|
||||
-e 's/nodePort: 31088/nodePort: 32088/g' \
|
||||
-e 's/nodePort: 31089/nodePort: 32089/g' \
|
||||
"$repository_root/deploy/kubernetes/production/application.yaml" >"$rendered"
|
||||
chmod 0600 "$rendered"
|
||||
kubectl --context "$context" -n "$namespace" apply \
|
||||
-f "$repository_root/deploy/kubernetes/production/service-account-rbac.yaml" >/dev/null
|
||||
kubectl --context "$context" -n "$namespace" apply -f "$manifest_root/local-config.yaml" >/dev/null
|
||||
kubectl --context "$context" -n "$namespace" apply -f "$rendered" >/dev/null
|
||||
|
||||
for workload in easyai-api-ningbo easyai-worker-ningbo; do
|
||||
kubectl --context "$context" -n "$namespace" set env deployment/"$workload" \
|
||||
--from=secret/easyai-database-ningbo >/dev/null
|
||||
done
|
||||
for workload in easyai-api-hongkong easyai-worker-hongkong; do
|
||||
kubectl --context "$context" -n "$namespace" set env deployment/"$workload" \
|
||||
--from=secret/easyai-database-hongkong >/dev/null
|
||||
done
|
||||
for workload in easyai-api-ningbo easyai-api-hongkong easyai-worker-ningbo easyai-worker-hongkong; do
|
||||
kubectl --context "$context" -n "$namespace" patch deployment "$workload" --type=json \
|
||||
-p='[{"op":"replace","path":"/spec/template/spec/volumes/0","value":{"name":"application-data","hostPath":{"path":"/var/lib/easyai-acceptance/media","type":"DirectoryOrCreate"}}}]' \
|
||||
>/dev/null
|
||||
done
|
||||
kubectl --context "$context" -n "$namespace" scale \
|
||||
deployment/easyai-web-ningbo deployment/easyai-web-hongkong --replicas=0 >/dev/null
|
||||
}
|
||||
|
||||
bootstrap_runtime() {
|
||||
local api_image=$1 api_digest=$2 snapshot=$3
|
||||
# shellcheck source=/dev/null
|
||||
source "$state_root/local.env"
|
||||
local snapshot_hash release_sha runtime_file
|
||||
snapshot_hash=$(jq -r '.snapshotSha256' "$snapshot")
|
||||
release_sha=$(git -C "$repository_root" rev-parse HEAD)
|
||||
runtime_file="$state_root/runtime-$snapshot_hash.json"
|
||||
kubectl --context "$context" -n "$namespace" delete pod easyai-local-bootstrap \
|
||||
--ignore-not-found --wait=true >/dev/null
|
||||
cat <<EOF | kubectl --context "$context" apply -f - >/dev/null
|
||||
apiVersion: v1
|
||||
kind: Pod
|
||||
metadata:
|
||||
name: easyai-local-bootstrap
|
||||
namespace: easyai
|
||||
labels:
|
||||
easyai.io/environment: local-acceptance
|
||||
spec:
|
||||
restartPolicy: Never
|
||||
containers:
|
||||
- name: bootstrap
|
||||
image: $api_image
|
||||
command:
|
||||
- /app/easyai-ai-gateway-acceptance-bootstrap
|
||||
- --local-cluster-id
|
||||
- $LOCAL_CLUSTER_ID
|
||||
- --release-sha
|
||||
- $release_sha
|
||||
- --api-image-digest
|
||||
- $api_digest
|
||||
- --worker-image-digest
|
||||
- $api_digest
|
||||
- --emulator-base-url
|
||||
- http://easyai-acceptance-upstream-proxy.easyai.svc.cluster.local:8090
|
||||
- --callback-url
|
||||
- http://easyai-acceptance-callback-collector.easyai.svc.cluster.local:8091/callbacks
|
||||
- --identity-shards
|
||||
- "32"
|
||||
- --output
|
||||
- /tmp/runtime.json
|
||||
envFrom:
|
||||
- secretRef:
|
||||
name: easyai-ai-gateway-runtime
|
||||
EOF
|
||||
kubectl --context "$context" -n "$namespace" wait \
|
||||
--for=jsonpath='{.status.phase}'=Succeeded pod/easyai-local-bootstrap --timeout=5m >/dev/null
|
||||
umask 077
|
||||
kubectl --context "$context" -n "$namespace" cp \
|
||||
easyai-local-bootstrap:/tmp/runtime.json "$runtime_file" >/dev/null
|
||||
chmod 0600 "$runtime_file"
|
||||
kubectl --context "$context" -n "$namespace" delete pod easyai-local-bootstrap \
|
||||
--wait=true >/dev/null
|
||||
printf '%s\n' "$runtime_file" >"$state_root/current-runtime-path"
|
||||
chmod 0600 "$state_root/current-runtime-path"
|
||||
}
|
||||
|
||||
up_cluster() {
|
||||
[[ ${1:-} == --snapshot && $# -eq 2 ]] || {
|
||||
usage >&2
|
||||
exit 64
|
||||
}
|
||||
local snapshot=$2
|
||||
[[ -f $snapshot && ! -L $snapshot ]] || {
|
||||
echo 'snapshot must be a regular non-symlink file' >&2
|
||||
exit 1
|
||||
}
|
||||
[[ -z $(git -C "$repository_root" status --short) ]] || {
|
||||
echo 'local acceptance requires a clean source tree so report and image source are reproducible' >&2
|
||||
exit 1
|
||||
}
|
||||
preflight
|
||||
ensure_private_material
|
||||
generate_tls
|
||||
(
|
||||
cd "$repository_root/apps/api"
|
||||
go run ./cmd/acceptance-snapshot validate --input "$snapshot"
|
||||
)
|
||||
cp "$snapshot" "$state_root/snapshot.json"
|
||||
chmod 0600 "$state_root/snapshot.json"
|
||||
local k3d config native_arch release_sha api_image web_image netem_image api_digest
|
||||
k3d=$(k3d_binary)
|
||||
if "$k3d" cluster list --no-headers | awk '{print $1}' | grep -Fxq "$cluster_name"; then
|
||||
echo "cluster $cluster_name already exists; use status or explicit down --confirm" >&2
|
||||
exit 1
|
||||
fi
|
||||
native_arch=$(docker info --format '{{.Architecture}}')
|
||||
case $native_arch in
|
||||
aarch64|arm64) native_arch=arm64 ;;
|
||||
x86_64) native_arch=amd64 ;;
|
||||
esac
|
||||
release_sha=$(git -C "$repository_root" rev-parse HEAD)
|
||||
api_image="easyai-acceptance-api:$release_sha"
|
||||
web_image="easyai-acceptance-web:$release_sha"
|
||||
netem_image="easyai-acceptance-netem:$release_sha"
|
||||
docker buildx build --load --platform "linux/$native_arch" --target api \
|
||||
-t "$api_image" "$repository_root"
|
||||
docker buildx build --load --platform "linux/$native_arch" --target web \
|
||||
-t "$web_image" "$repository_root"
|
||||
docker buildx build --load --platform "linux/$native_arch" --target acceptance-netem \
|
||||
-t "$netem_image" "$repository_root"
|
||||
api_digest=$(docker image inspect "$api_image" --format '{{.Id}}')
|
||||
[[ $api_digest =~ ^sha256:[0-9a-f]{64}$ ]]
|
||||
|
||||
config="$state_root/k3d.rendered.yaml"
|
||||
render_k3d_config "$config"
|
||||
EASYAI_ACCEPTANCE_MEDIA_DIR="$media_root" "$k3d" cluster create --config "$config"
|
||||
docker update --cpus 4 --memory 8g "${cluster_name}-server-0" >/dev/null
|
||||
docker update --cpus 4 --memory 8g "${cluster_name}-server-1" >/dev/null
|
||||
docker update --cpus 2 --memory 4g "${cluster_name}-server-2" >/dev/null
|
||||
"$k3d" image import -c "$cluster_name" "$api_image" "$web_image" "$netem_image"
|
||||
|
||||
local cnpg_manifest="$state_root/cnpg.yaml"
|
||||
curl -fsSL "$CNPG_MANIFEST_URL" -o "$cnpg_manifest"
|
||||
[[ $(shasum -a 256 "$cnpg_manifest" | awk '{print $1}') == "$CNPG_MANIFEST_SHA256" ]] || {
|
||||
echo 'CNPG manifest checksum mismatch' >&2
|
||||
exit 1
|
||||
}
|
||||
chmod 0600 "$cnpg_manifest"
|
||||
kubectl --context "$context" apply --server-side --force-conflicts -f "$cnpg_manifest" >/dev/null
|
||||
kubectl --context "$context" -n cnpg-system rollout status \
|
||||
deployment/cnpg-controller-manager --timeout=10m
|
||||
apply_runtime_secrets
|
||||
kubectl --context "$context" -n "$namespace" apply -f "$manifest_root/database.yaml" >/dev/null
|
||||
kubectl --context "$context" -n "$namespace" wait --for=condition=Ready \
|
||||
cluster/easyai-postgres --timeout=15m >/dev/null
|
||||
run_migrations "$api_image"
|
||||
mark_local_database
|
||||
import_snapshot "$api_image" "$snapshot"
|
||||
|
||||
sed "s|image: easyai-api|image: $api_image|g; s|image: easyai-acceptance-netem|image: $netem_image|g" \
|
||||
"$manifest_root/support-services.yaml" |
|
||||
kubectl --context "$context" -n "$namespace" apply -f - >/dev/null
|
||||
wait_rollout deployment/easyai-acceptance-emulator
|
||||
wait_rollout deployment/easyai-acceptance-callback-collector
|
||||
wait_rollout deployment/easyai-acceptance-upstream-proxy
|
||||
wait_rollout deployment/easyai-postgres-proxy-ningbo
|
||||
wait_rollout deployment/easyai-postgres-proxy-hongkong
|
||||
render_and_apply_application "$api_image" "$web_image"
|
||||
sed "s|image: easyai-web|image: $web_image|g" "$manifest_root/tls-edges.yaml" |
|
||||
kubectl --context "$context" -n "$namespace" apply -f - >/dev/null
|
||||
wait_rollout deployment/easyai-api-ningbo
|
||||
wait_rollout deployment/easyai-api-hongkong
|
||||
wait_rollout deployment/easyai-worker-ningbo
|
||||
wait_rollout deployment/easyai-worker-hongkong
|
||||
wait_rollout deployment/easyai-capacity-controller
|
||||
wait_rollout deployment/easyai-acceptance-edge-ningbo
|
||||
wait_rollout deployment/easyai-acceptance-edge-hongkong
|
||||
bootstrap_runtime "$api_image" "$api_digest" "$snapshot"
|
||||
"$script_dir/network-fault.sh" baseline
|
||||
echo "local_acceptance_cluster=PASS context=$context gateways=https://127.0.0.1:18443,https://127.0.0.1:19443 ca_file=$state_root/ca.crt"
|
||||
}
|
||||
|
||||
status_cluster() {
|
||||
local k3d
|
||||
k3d=$(k3d_binary) || {
|
||||
echo 'k3d is unavailable' >&2
|
||||
exit 1
|
||||
}
|
||||
"$k3d" cluster list --no-headers | awk '{print $1}' | grep -Fxq "$cluster_name" || {
|
||||
echo "cluster $cluster_name does not exist" >&2
|
||||
exit 1
|
||||
}
|
||||
kubectl --context "$context" get nodes -o wide
|
||||
kubectl --context "$context" -n "$namespace" get cluster,pod,deployment
|
||||
echo "local_acceptance_status=PASS context=$context"
|
||||
}
|
||||
|
||||
down_cluster() {
|
||||
[[ ${1:-} == --confirm && $# -eq 1 ]] || {
|
||||
echo 'down requires the literal --confirm flag' >&2
|
||||
exit 64
|
||||
}
|
||||
local k3d
|
||||
k3d=$(k3d_binary)
|
||||
"$k3d" cluster delete "$cluster_name"
|
||||
echo "local_acceptance_down=PASS preserved_private_root=$private_root"
|
||||
}
|
||||
|
||||
case ${1:-} in
|
||||
install-tools)
|
||||
[[ $# -eq 1 ]] || { usage >&2; exit 64; }
|
||||
install_tools
|
||||
;;
|
||||
preflight)
|
||||
[[ $# -eq 1 ]] || { usage >&2; exit 64; }
|
||||
preflight
|
||||
;;
|
||||
up)
|
||||
shift
|
||||
up_cluster "$@"
|
||||
;;
|
||||
status)
|
||||
[[ $# -eq 1 ]] || { usage >&2; exit 64; }
|
||||
status_cluster
|
||||
;;
|
||||
down)
|
||||
shift
|
||||
down_cluster "$@"
|
||||
;;
|
||||
*)
|
||||
usage >&2
|
||||
exit 64
|
||||
;;
|
||||
esac
|
||||
Executable
+17
@@ -0,0 +1,17 @@
|
||||
#!/bin/sh
|
||||
set -eu
|
||||
|
||||
listen_port=${NETEM_LISTEN_PORT:-15432}
|
||||
target_host=${NETEM_TARGET_HOST:-}
|
||||
target_port=${NETEM_TARGET_PORT:-}
|
||||
|
||||
case "$listen_port:$target_port" in
|
||||
*[!0-9:]*|:*|*:) echo "NETEM_LISTEN_PORT and NETEM_TARGET_PORT must be numeric" >&2; exit 64 ;;
|
||||
esac
|
||||
case "$target_host" in
|
||||
""|*[!A-Za-z0-9.-]*) echo "NETEM_TARGET_HOST must be a DNS name or address" >&2; exit 64 ;;
|
||||
esac
|
||||
|
||||
exec socat \
|
||||
"TCP-LISTEN:${listen_port},fork,reuseaddr,keepalive" \
|
||||
"TCP:${target_host}:${target_port},connect-timeout=10"
|
||||
Executable
+96
@@ -0,0 +1,96 @@
|
||||
#!/usr/bin/env bash
|
||||
set -euo pipefail
|
||||
|
||||
usage() {
|
||||
cat <<'EOF'
|
||||
Usage:
|
||||
scripts/acceptance/network-fault.sh baseline
|
||||
scripts/acceptance/network-fault.sh weak-link
|
||||
scripts/acceptance/network-fault.sh upstream-outage
|
||||
scripts/acceptance/network-fault.sh database-outage <ningbo|hongkong>
|
||||
scripts/acceptance/network-fault.sh reset
|
||||
|
||||
Only operates on Pods carrying easyai.io/environment=local-acceptance in the
|
||||
easyai-acceptance-local k3d context.
|
||||
EOF
|
||||
}
|
||||
|
||||
profile=${1:-}
|
||||
site=${2:-}
|
||||
namespace=${AI_GATEWAY_LOCAL_ACCEPTANCE_NAMESPACE:-easyai}
|
||||
context=${AI_GATEWAY_LOCAL_ACCEPTANCE_CONTEXT:-k3d-easyai-acceptance-local}
|
||||
|
||||
command -v kubectl >/dev/null 2>&1 || {
|
||||
echo 'kubectl is required' >&2
|
||||
exit 1
|
||||
}
|
||||
[[ $(kubectl config get-contexts "$context" -o name) == "$context" ]] || {
|
||||
echo "refusing netem because context $context does not exist" >&2
|
||||
exit 1
|
||||
}
|
||||
|
||||
pod_for() {
|
||||
local name=$1
|
||||
kubectl --context "$context" -n "$namespace" get pod \
|
||||
-l "app.kubernetes.io/name=$name,easyai.io/environment=local-acceptance" \
|
||||
-o 'jsonpath={.items[0].metadata.name}'
|
||||
}
|
||||
|
||||
apply_qdisc() {
|
||||
local pod=$1
|
||||
shift
|
||||
[[ -n $pod ]] || {
|
||||
echo 'local acceptance proxy Pod was not found' >&2
|
||||
exit 1
|
||||
}
|
||||
kubectl --context "$context" -n "$namespace" exec "$pod" -- \
|
||||
tc qdisc replace dev eth0 root netem "$@"
|
||||
}
|
||||
|
||||
reset_qdisc() {
|
||||
local pod=$1
|
||||
[[ -z $pod ]] || kubectl --context "$context" -n "$namespace" exec "$pod" -- \
|
||||
tc qdisc del dev eth0 root >/dev/null 2>&1 || true
|
||||
}
|
||||
|
||||
upstream_pod=$(pod_for easyai-acceptance-upstream-proxy)
|
||||
ningbo_db_pod=$(kubectl --context "$context" -n "$namespace" get pod \
|
||||
-l 'app.kubernetes.io/name=easyai-postgres-proxy,easyai.io/site=ningbo,easyai.io/environment=local-acceptance' \
|
||||
-o 'jsonpath={.items[0].metadata.name}')
|
||||
hongkong_db_pod=$(kubectl --context "$context" -n "$namespace" get pod \
|
||||
-l 'app.kubernetes.io/name=easyai-postgres-proxy,easyai.io/site=hongkong,easyai.io/environment=local-acceptance' \
|
||||
-o 'jsonpath={.items[0].metadata.name}')
|
||||
|
||||
case $profile in
|
||||
baseline)
|
||||
apply_qdisc "$upstream_pod" delay 20ms
|
||||
apply_qdisc "$ningbo_db_pod" delay 20ms
|
||||
apply_qdisc "$hongkong_db_pod" delay 20ms
|
||||
;;
|
||||
weak-link)
|
||||
apply_qdisc "$upstream_pod" delay 20ms 10ms distribution normal loss 0.5%
|
||||
apply_qdisc "$ningbo_db_pod" delay 20ms 10ms distribution normal loss 0.5%
|
||||
apply_qdisc "$hongkong_db_pod" delay 20ms 10ms distribution normal loss 0.5%
|
||||
;;
|
||||
upstream-outage)
|
||||
apply_qdisc "$upstream_pod" loss 100%
|
||||
;;
|
||||
database-outage)
|
||||
case $site in
|
||||
ningbo) apply_qdisc "$ningbo_db_pod" loss 100% ;;
|
||||
hongkong) apply_qdisc "$hongkong_db_pod" loss 100% ;;
|
||||
*) usage >&2; exit 64 ;;
|
||||
esac
|
||||
;;
|
||||
reset)
|
||||
reset_qdisc "$upstream_pod"
|
||||
reset_qdisc "$ningbo_db_pod"
|
||||
reset_qdisc "$hongkong_db_pod"
|
||||
;;
|
||||
*)
|
||||
usage >&2
|
||||
exit 64
|
||||
;;
|
||||
esac
|
||||
|
||||
echo "local_acceptance_netem=PASS profile=$profile${site:+ site=$site}"
|
||||
Executable
+316
@@ -0,0 +1,316 @@
|
||||
#!/usr/bin/env node
|
||||
import { createHash } from 'node:crypto';
|
||||
import { lstat, mkdir, readFile, readdir, writeFile } from 'node:fs/promises';
|
||||
import { dirname, resolve } from 'node:path';
|
||||
|
||||
const shaPattern = /^[0-9a-f]{40}$/;
|
||||
const digestPattern = /^sha256:[0-9a-f]{64}$/;
|
||||
const hashPattern = /^[0-9a-f]{64}$/;
|
||||
const unsafeKeyPattern = /(password|secret|token|credential|authorization|api.?key|access.?key|private.?key|connection.?string|database.?url|proxy)/i;
|
||||
|
||||
function parseArgs(argv) {
|
||||
const command = argv.shift();
|
||||
const values = {};
|
||||
while (argv.length > 0) {
|
||||
const key = argv.shift();
|
||||
if (!key?.startsWith('--') || argv.length === 0) throw new Error('invalid arguments');
|
||||
values[key.slice(2)] = argv.shift();
|
||||
}
|
||||
return { command, values };
|
||||
}
|
||||
|
||||
async function regularJSON(path) {
|
||||
const absolute = resolve(path);
|
||||
const info = await lstat(absolute);
|
||||
if (!info.isFile() || info.isSymbolicLink()) throw new Error(`${path} must be a regular non-symlink file`);
|
||||
return JSON.parse(await readFile(absolute, 'utf8'));
|
||||
}
|
||||
|
||||
function assertSecretSafe(value, path = '$') {
|
||||
if (Array.isArray(value)) {
|
||||
value.forEach((item, index) => assertSecretSafe(item, `${path}[${index}]`));
|
||||
return;
|
||||
}
|
||||
if (value && typeof value === 'object') {
|
||||
for (const [key, nested] of Object.entries(value)) {
|
||||
if (key !== 'secretSafe' && unsafeKeyPattern.test(key)) {
|
||||
throw new Error(`unsafe report key at ${path}.${key}`);
|
||||
}
|
||||
assertSecretSafe(nested, `${path}.${key}`);
|
||||
}
|
||||
return;
|
||||
}
|
||||
if (typeof value === 'string' && /(?:postgres(?:ql)?|https?):\/\/[^\s]+/i.test(value)) {
|
||||
throw new Error(`URL-like value is forbidden at ${path}`);
|
||||
}
|
||||
}
|
||||
|
||||
function validate(report) {
|
||||
if (report.schemaVersion !== 'acceptance-report/v1' || report.secretSafe !== true) {
|
||||
throw new Error('unsupported or non-secret-safe acceptance report');
|
||||
}
|
||||
if (!report.runId || !shaPattern.test(report.release?.sourceSha ?? '') ||
|
||||
!digestPattern.test(report.release?.apiImageDigest ?? '') ||
|
||||
!digestPattern.test(report.release?.workerImageDigest ?? '') ||
|
||||
!shaPattern.test(report.snapshot?.sourceReleaseSha ?? '') ||
|
||||
!hashPattern.test(report.snapshot?.configHash ?? '') ||
|
||||
!hashPattern.test(report.snapshot?.sha256 ?? '')) {
|
||||
throw new Error('acceptance report identity fields are invalid');
|
||||
}
|
||||
if (!report.stages || !Array.isArray(report.gates) ||
|
||||
report.promotion?.requiresManualConfirmation !== true ||
|
||||
!['validation', 'live', 'not-applicable'].includes(report.promotion?.trafficMode)) {
|
||||
throw new Error('acceptance report stages, gates, or promotion are invalid');
|
||||
}
|
||||
for (const [name, stage] of Object.entries(report.stages)) {
|
||||
if (!['passed', 'failed', 'not-run'].includes(stage?.status)) {
|
||||
throw new Error(`invalid stage status for ${name}`);
|
||||
}
|
||||
}
|
||||
assertSecretSafe(report);
|
||||
}
|
||||
|
||||
async function loadReports(directory) {
|
||||
const entries = (await readdir(resolve(directory), { withFileTypes: true }))
|
||||
.filter((entry) => entry.isFile() && entry.name.endsWith('.json'))
|
||||
.map((entry) => entry.name)
|
||||
.sort();
|
||||
const reports = [];
|
||||
for (const name of entries) {
|
||||
const report = await regularJSON(resolve(directory, name));
|
||||
if (report.schemaVersion === 'acceptance-load-report/v1') {
|
||||
if (report.secretSafe !== true) throw new Error(`load report ${name} is not secret-safe`);
|
||||
reports.push({
|
||||
file: name,
|
||||
profile: report.profile,
|
||||
passed: report.passed,
|
||||
phases: report.phases,
|
||||
startedAt: report.startedAt,
|
||||
finishedAt: report.finishedAt
|
||||
});
|
||||
}
|
||||
}
|
||||
return reports;
|
||||
}
|
||||
|
||||
async function buildLocal(values) {
|
||||
for (const key of ['runtime', 'snapshot', 'reports', 'artifact', 'output']) {
|
||||
if (!values[key]) throw new Error(`--${key} is required`);
|
||||
}
|
||||
const runtime = await regularJSON(values.runtime);
|
||||
const snapshot = await regularJSON(values.snapshot);
|
||||
const artifact = await regularJSON(values.artifact);
|
||||
const loads = await loadReports(values.reports);
|
||||
if (runtime.schemaVersion !== 'acceptance-runtime/v1' ||
|
||||
snapshot.schemaVersion !== 'acceptance-snapshot/v1' ||
|
||||
artifact.schemaVersion !== 'acceptance-artifact-smoke/v1') {
|
||||
throw new Error('local report inputs have incompatible schema versions');
|
||||
}
|
||||
if (runtime.releaseSha !== artifact.releaseSha ||
|
||||
artifact.apiImageDigest !== values['api-digest'] ||
|
||||
runtime.snapshotConfigHash !== snapshot.source.configHash ||
|
||||
runtime.snapshotSha256 !== snapshot.snapshotSha256) {
|
||||
throw new Error('local report CAS identity mismatch');
|
||||
}
|
||||
const localPassed = loads.length > 0 && loads.every((item) => item.passed === true);
|
||||
const gates = [
|
||||
{ id: 'local_load_reports', passed: localPassed, detail: `${loads.length} load reports` },
|
||||
{ id: 'amd64_artifact_smoke', passed: artifact.passed === true }
|
||||
];
|
||||
const report = {
|
||||
schemaVersion: 'acceptance-report/v1',
|
||||
runId: runtime.runId,
|
||||
release: {
|
||||
sourceSha: artifact.releaseSha,
|
||||
apiImageDigest: artifact.apiImageDigest,
|
||||
workerImageDigest: artifact.apiImageDigest
|
||||
},
|
||||
snapshot: {
|
||||
sourceReleaseSha: snapshot.source.releaseSha,
|
||||
configHash: snapshot.source.configHash,
|
||||
sha256: snapshot.snapshotSha256
|
||||
},
|
||||
stages: {
|
||||
localNative: { status: localPassed ? 'passed' : 'failed', loadReports: loads },
|
||||
amd64Artifact: { status: artifact.passed ? 'passed' : 'failed', ...artifact },
|
||||
onlineSimulation: { status: 'not-run' },
|
||||
realCanary: { status: 'not-run' }
|
||||
},
|
||||
certifiedProfile: values['certified-profile'] ? {
|
||||
name: values['certified-profile'],
|
||||
source: 'local-candidate-only'
|
||||
} : null,
|
||||
gates,
|
||||
promotion: {
|
||||
ready: false,
|
||||
requiresManualConfirmation: true,
|
||||
trafficMode: 'not-applicable'
|
||||
},
|
||||
createdAt: new Date().toISOString(),
|
||||
secretSafe: true
|
||||
};
|
||||
validate(report);
|
||||
await mkdir(dirname(resolve(values.output)), { recursive: true, mode: 0o700 });
|
||||
await writeFile(resolve(values.output), `${JSON.stringify(report, null, 2)}\n`, { mode: 0o600 });
|
||||
process.stdout.write(`acceptance_report_local=PASS sha256=${createHash('sha256').update(JSON.stringify(report)).digest('hex')}\n`);
|
||||
}
|
||||
|
||||
async function buildLocalPartial(values) {
|
||||
for (const key of ['runtime', 'snapshot', 'reports', 'output', 'failure-phase']) {
|
||||
if (!values[key]) throw new Error(`--${key} is required`);
|
||||
}
|
||||
const runtime = await regularJSON(values.runtime);
|
||||
const snapshot = await regularJSON(values.snapshot);
|
||||
const loads = await loadReports(values.reports);
|
||||
if (runtime.schemaVersion !== 'acceptance-runtime/v1' ||
|
||||
snapshot.schemaVersion !== 'acceptance-snapshot/v1' ||
|
||||
!shaPattern.test(runtime.releaseSha ?? '') ||
|
||||
!digestPattern.test(runtime.apiImageDigest ?? '') ||
|
||||
!digestPattern.test(runtime.workerImageDigest ?? '') ||
|
||||
runtime.snapshotConfigHash !== snapshot.source.configHash ||
|
||||
runtime.snapshotSha256 !== snapshot.snapshotSha256) {
|
||||
throw new Error('partial local report inputs have incompatible CAS identity');
|
||||
}
|
||||
const completedLoadsPassed = loads.every((item) => item.passed === true);
|
||||
const report = {
|
||||
schemaVersion: 'acceptance-report/v1',
|
||||
runId: runtime.runId,
|
||||
release: {
|
||||
sourceSha: runtime.releaseSha,
|
||||
apiImageDigest: runtime.apiImageDigest,
|
||||
workerImageDigest: runtime.workerImageDigest
|
||||
},
|
||||
snapshot: {
|
||||
sourceReleaseSha: snapshot.source.releaseSha,
|
||||
configHash: snapshot.source.configHash,
|
||||
sha256: snapshot.snapshotSha256
|
||||
},
|
||||
stages: {
|
||||
localNative: {
|
||||
status: 'failed',
|
||||
failurePhase: values['failure-phase'],
|
||||
completedLoadReportsPassed: completedLoadsPassed,
|
||||
loadReports: loads
|
||||
},
|
||||
amd64Artifact: { status: 'not-run' },
|
||||
onlineSimulation: { status: 'not-run' },
|
||||
realCanary: { status: 'not-run' }
|
||||
},
|
||||
certifiedProfile: values['certified-profile'] ? {
|
||||
name: values['certified-profile'],
|
||||
source: 'last-local-stable-candidate'
|
||||
} : null,
|
||||
gates: [
|
||||
{
|
||||
id: 'local_execution_complete',
|
||||
passed: false,
|
||||
detail: `stopped during ${values['failure-phase']}`
|
||||
},
|
||||
{
|
||||
id: 'completed_load_reports',
|
||||
passed: completedLoadsPassed,
|
||||
detail: `${loads.length} load reports`
|
||||
}
|
||||
],
|
||||
promotion: {
|
||||
ready: false,
|
||||
requiresManualConfirmation: true,
|
||||
trafficMode: 'not-applicable'
|
||||
},
|
||||
createdAt: new Date().toISOString(),
|
||||
secretSafe: true
|
||||
};
|
||||
validate(report);
|
||||
await mkdir(dirname(resolve(values.output)), { recursive: true, mode: 0o700 });
|
||||
await writeFile(resolve(values.output), `${JSON.stringify(report, null, 2)}\n`, { mode: 0o600 });
|
||||
process.stdout.write('acceptance_report_local_partial=PASS\n');
|
||||
}
|
||||
|
||||
async function mergeProduction(values) {
|
||||
for (const key of ['local-report', 'production-summary', 'output']) {
|
||||
if (!values[key]) throw new Error(`--${key} is required`);
|
||||
}
|
||||
const local = await regularJSON(values['local-report']);
|
||||
const production = await regularJSON(values['production-summary']);
|
||||
validate(local);
|
||||
if (local.release.sourceSha !== production.releaseSha ||
|
||||
local.release.apiImageDigest !== production.apiImageDigest ||
|
||||
local.snapshot.configHash !== production.snapshotConfigHash ||
|
||||
local.snapshot.sha256 !== production.snapshotSha256) {
|
||||
throw new Error('production summary does not match local acceptance CAS fields');
|
||||
}
|
||||
local.stages.onlineSimulation = {
|
||||
status: production.passed === true ? 'passed' : 'failed',
|
||||
stableCapacityProfile: production.stableCapacityProfile,
|
||||
tasks: production.tasks,
|
||||
geminiTasks: production.geminiTasks,
|
||||
videoTasks: production.videoTasks
|
||||
};
|
||||
local.stages.realCanary = {
|
||||
status: production.realCanaryPassed === true ? 'passed' : 'failed'
|
||||
};
|
||||
local.certifiedProfile = production.certifiedProfile ?? local.certifiedProfile;
|
||||
local.gates.push(...(production.gates ?? []));
|
||||
local.promotion = {
|
||||
ready: production.passed === true && production.realCanaryPassed === true &&
|
||||
local.gates.every((gate) => gate.passed === true),
|
||||
requiresManualConfirmation: true,
|
||||
trafficMode: 'validation'
|
||||
};
|
||||
local.createdAt = new Date().toISOString();
|
||||
validate(local);
|
||||
await mkdir(dirname(resolve(values.output)), { recursive: true, mode: 0o700 });
|
||||
await writeFile(resolve(values.output), `${JSON.stringify(local, null, 2)}\n`, { mode: 0o600 });
|
||||
process.stdout.write(`acceptance_report_production=PASS promotion_ready=${local.promotion.ready}\n`);
|
||||
}
|
||||
|
||||
const { command, values } = parseArgs(process.argv.slice(2));
|
||||
if (command === 'validate') {
|
||||
const report = await regularJSON(values.input ?? '');
|
||||
validate(report);
|
||||
process.stdout.write('acceptance_report_validate=PASS\n');
|
||||
} else if (command === 'build-local') {
|
||||
await buildLocal(values);
|
||||
} else if (command === 'build-local-partial') {
|
||||
await buildLocalPartial(values);
|
||||
} else if (command === 'merge-production') {
|
||||
await mergeProduction(values);
|
||||
} else if (command === 'mark-promoted') {
|
||||
const input = values.input ?? '';
|
||||
const output = values.output ?? input;
|
||||
const report = await regularJSON(input);
|
||||
validate(report);
|
||||
if (report.promotion.ready !== true || report.promotion.trafficMode !== 'validation') {
|
||||
throw new Error('only a ready validation report can be marked promoted');
|
||||
}
|
||||
report.promotion.trafficMode = 'live';
|
||||
report.promotion.promotedAt = new Date().toISOString();
|
||||
await writeFile(resolve(output), `${JSON.stringify(report, null, 2)}\n`, { mode: 0o600 });
|
||||
process.stdout.write('acceptance_report_promoted=PASS\n');
|
||||
} else if (command === 'attach-monitor') {
|
||||
const input = values.input ?? '';
|
||||
const output = values.output ?? input;
|
||||
const report = await regularJSON(input);
|
||||
const monitor = await regularJSON(values.monitor ?? '');
|
||||
validate(report);
|
||||
if (monitor.schemaVersion !== 'acceptance-monitor-report/v1' ||
|
||||
monitor.runId !== report.runId || monitor.releaseSha !== report.release.sourceSha ||
|
||||
monitor.secretSafe !== true) {
|
||||
throw new Error('monitor report does not match overall acceptance report');
|
||||
}
|
||||
report.stages.postPromotionMonitor = {
|
||||
status: monitor.passed === true ? 'passed' : 'failed',
|
||||
startedAt: monitor.startedAt,
|
||||
finishedAt: monitor.finishedAt,
|
||||
samples: monitor.samples,
|
||||
failureGateId: monitor.failureGateId
|
||||
};
|
||||
if (monitor.passed !== true) {
|
||||
report.promotion.ready = false;
|
||||
report.promotion.trafficMode = 'validation';
|
||||
}
|
||||
await writeFile(resolve(output), `${JSON.stringify(report, null, 2)}\n`, { mode: 0o600 });
|
||||
process.stdout.write(`acceptance_report_monitor=PASS monitor_passed=${monitor.passed === true}\n`);
|
||||
} else {
|
||||
throw new Error('usage: report.mjs {validate|build-local|build-local-partial|merge-production|mark-promoted|attach-monitor} [options]');
|
||||
}
|
||||
Executable
+536
@@ -0,0 +1,536 @@
|
||||
#!/usr/bin/env bash
|
||||
set -euo pipefail
|
||||
|
||||
script_dir=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)
|
||||
repository_root=$(cd "$script_dir/../.." && pwd)
|
||||
private_root="$repository_root/.local-secrets/acceptance"
|
||||
state_root="$private_root/state"
|
||||
context=k3d-easyai-acceptance-local
|
||||
namespace=easyai
|
||||
runtime_path_file="$state_root/current-runtime-path"
|
||||
snapshot="$state_root/snapshot.json"
|
||||
load_binary="$state_root/easyai-ai-gateway-acceptance-load"
|
||||
active_load_pid=
|
||||
netem_active=false
|
||||
current_phase=initialization
|
||||
stable_profile=
|
||||
|
||||
usage() {
|
||||
cat <<'EOF'
|
||||
Usage:
|
||||
scripts/acceptance/run-local-acceptance.sh quick
|
||||
scripts/acceptance/run-local-acceptance.sh full --release-manifest dist/releases/<SHA>.json
|
||||
scripts/acceptance/run-local-acceptance.sh artifact-smoke --release-manifest dist/releases/<SHA>.json
|
||||
|
||||
`full` executes P24/P28/P32 three times, the fault matrix, autoscaling/drain,
|
||||
80% soak, 120% overload, and exact linux/amd64 artifact smoke. The load process
|
||||
runs outside K3s and splits requests 50/50 across both TLS entrances.
|
||||
EOF
|
||||
}
|
||||
|
||||
private_file() {
|
||||
local path=$1 mode
|
||||
[[ -f $path && ! -L $path ]] || return 1
|
||||
if [[ $(uname -s) == Darwin ]]; then
|
||||
mode=$(stat -f '%Lp' "$path")
|
||||
else
|
||||
mode=$(stat -c '%a' "$path")
|
||||
fi
|
||||
[[ $mode == 600 ]]
|
||||
}
|
||||
|
||||
cleanup() {
|
||||
local status=$?
|
||||
trap - EXIT
|
||||
if [[ -n $active_load_pid ]]; then
|
||||
kill "$active_load_pid" >/dev/null 2>&1 || true
|
||||
wait "$active_load_pid" >/dev/null 2>&1 || true
|
||||
fi
|
||||
if [[ $netem_active == true ]]; then
|
||||
"$script_dir/network-fault.sh" reset >/dev/null 2>&1 || true
|
||||
fi
|
||||
if [[ $status -ne 0 && -n ${runtime:-} && -n ${report_root:-} && -f ${runtime:-} && -f $snapshot ]]; then
|
||||
restore_profile_best_effort "${stable_profile:-P24}"
|
||||
local partial_args=(
|
||||
"$script_dir/report.mjs" build-local-partial
|
||||
--runtime "$runtime" \
|
||||
--snapshot "$snapshot" \
|
||||
--reports "$report_root" \
|
||||
--failure-phase "$current_phase" \
|
||||
--output "$report_root/acceptance-report.partial.json"
|
||||
)
|
||||
if [[ -n $stable_profile ]]; then
|
||||
partial_args+=(--certified-profile "$stable_profile")
|
||||
fi
|
||||
if node "${partial_args[@]}" >/dev/null 2>&1; then
|
||||
echo "local_acceptance=FAILED phase=$current_phase partial_report=$report_root/acceptance-report.partial.json" >&2
|
||||
else
|
||||
echo "local_acceptance=FAILED phase=$current_phase partial_report=unavailable" >&2
|
||||
fi
|
||||
fi
|
||||
exit "$status"
|
||||
}
|
||||
trap cleanup EXIT
|
||||
trap 'current_phase=signal_interrupted; exit 130' HUP INT TERM
|
||||
|
||||
require_local_cluster() {
|
||||
command -v kubectl >/dev/null 2>&1
|
||||
command -v jq >/dev/null 2>&1
|
||||
command -v node >/dev/null 2>&1
|
||||
[[ $(kubectl config get-contexts "$context" -o name) == "$context" ]]
|
||||
private_file "$runtime_path_file" && private_file "$snapshot"
|
||||
runtime=$(<"$runtime_path_file")
|
||||
private_file "$runtime" || {
|
||||
echo 'local acceptance runtime file is missing or not mode 0600' >&2
|
||||
exit 1
|
||||
}
|
||||
jq -e '.schemaVersion == "acceptance-runtime/v1"' "$runtime" >/dev/null
|
||||
run_id=$(jq -r '.runId' "$runtime")
|
||||
release_sha=$(jq -r '.releaseSha' "$runtime")
|
||||
report_root="$repository_root/dist/acceptance/local/$run_id"
|
||||
install -d -m 0700 "$report_root"
|
||||
(
|
||||
cd "$repository_root/apps/api"
|
||||
env -u AI_GATEWAY_TEST_DATABASE_URL go build -trimpath \
|
||||
-o "$load_binary" ./cmd/acceptance-load
|
||||
)
|
||||
}
|
||||
|
||||
database_query() {
|
||||
local sql=$1 primary
|
||||
primary=$(kubectl --context "$context" -n "$namespace" get cluster easyai-postgres \
|
||||
-o 'jsonpath={.status.currentPrimary}')
|
||||
kubectl --context "$context" -n "$namespace" exec "$primary" -c postgres -- \
|
||||
psql -X -v ON_ERROR_STOP=1 -U postgres -d easyai_ai_gateway -At -c "$sql"
|
||||
}
|
||||
|
||||
load_environment() {
|
||||
acceptance_api_keys=$(jq -r '.apiKeys | join(",")' "$runtime")
|
||||
acceptance_run_token=$(jq -r '.runToken' "$runtime")
|
||||
acceptance_gemini_model=$(jq -r '.geminiModel' "$runtime")
|
||||
acceptance_video_model=$(jq -r '.videoModel' "$runtime")
|
||||
acceptance_emulator_url=$(jq -r '.emulatorBaseUrl' "$runtime")
|
||||
}
|
||||
|
||||
run_load() {
|
||||
local profile=$1 report_path=$2
|
||||
shift 2
|
||||
AI_GATEWAY_ACCEPTANCE_GATEWAYS='https://127.0.0.1:18443,https://127.0.0.1:19443' \
|
||||
AI_GATEWAY_ACCEPTANCE_GATEWAY_TLS_SERVER_NAME=gateway.easyai.local \
|
||||
AI_GATEWAY_ACCEPTANCE_GATEWAY_CA_FILE="$state_root/ca.crt" \
|
||||
AI_GATEWAY_ACCEPTANCE_EMULATOR_URL="$acceptance_emulator_url" \
|
||||
AI_GATEWAY_ACCEPTANCE_API_KEYS="$acceptance_api_keys" \
|
||||
AI_GATEWAY_ACCEPTANCE_RUN_ID="$run_id" \
|
||||
AI_GATEWAY_ACCEPTANCE_RUN_TOKEN="$acceptance_run_token" \
|
||||
AI_GATEWAY_ACCEPTANCE_GEMINI_MODEL="$acceptance_gemini_model" \
|
||||
AI_GATEWAY_ACCEPTANCE_VIDEO_MODEL="$acceptance_video_model" \
|
||||
"$load_binary" -profile "$profile" -report "$report_path" "$@" \
|
||||
>"$report_path.stdout"
|
||||
chmod 0600 "$report_path" "$report_path.stdout"
|
||||
jq -e '.schemaVersion == "acceptance-load-report/v1" and .secretSafe == true and .passed == true' \
|
||||
"$report_path" >/dev/null
|
||||
}
|
||||
|
||||
sample_resources() {
|
||||
local output=$1 stop_file=$2
|
||||
printf 'timestamp,scope,name,cpu,memory\n' >"$output"
|
||||
while [[ ! -e $stop_file ]]; do
|
||||
kubectl --context "$context" top nodes --no-headers 2>/dev/null |
|
||||
awk -v timestamp="$(date -u '+%Y-%m-%dT%H:%M:%SZ')" \
|
||||
'{print timestamp ",node," $1 "," $3 "," $5}' >>"$output" || true
|
||||
kubectl --context "$context" -n "$namespace" top pods --no-headers 2>/dev/null |
|
||||
awk -v timestamp="$(date -u '+%Y-%m-%dT%H:%M:%SZ')" \
|
||||
'{print timestamp ",pod," $1 "," $2 "," $3}' >>"$output" || true
|
||||
sleep 5
|
||||
done
|
||||
}
|
||||
|
||||
verify_hard_gates() {
|
||||
local unhealthy ready sync_state connections max_connections queue duplicate_remote duplicate_billing
|
||||
ready=$(kubectl --context "$context" get nodes -o json |
|
||||
jq '[.items[] | select(any(.status.conditions[]; .type=="Ready" and .status=="True"))] | length')
|
||||
[[ $ready == 3 ]]
|
||||
[[ $(kubectl --context "$context" get nodes -o json |
|
||||
jq '[.items[] | select(any(.status.conditions[]; .type=="MemoryPressure" and .status=="True"))] | length') == 0 ]]
|
||||
[[ $(kubectl --context "$context" -n "$namespace" get cluster easyai-postgres \
|
||||
-o 'jsonpath={.status.readyInstances}') == 2 ]]
|
||||
sync_state=$(database_query "SELECT COALESCE(string_agg(sync_state,','),'') FROM pg_stat_replication;")
|
||||
[[ ",$sync_state," == *,sync,* || ",$sync_state," == *,quorum,* ]]
|
||||
unhealthy=$(kubectl --context "$context" -n "$namespace" get pods -o json |
|
||||
jq '[.items[]
|
||||
| select(.metadata.labels["app.kubernetes.io/name"] == "easyai-api" or
|
||||
.metadata.labels["app.kubernetes.io/name"] == "easyai-worker")
|
||||
| .status.containerStatuses[]?
|
||||
| select(.restartCount > 0 or .lastState.terminated.reason == "OOMKilled" or
|
||||
.state.terminated.reason == "OOMKilled")] | length')
|
||||
[[ $unhealthy == 0 ]]
|
||||
while read -r _node _cpu _cpu_percent _memory memory_percent; do
|
||||
memory_percent=${memory_percent%\%}
|
||||
[[ $memory_percent =~ ^[0-9]+$ ]]
|
||||
(( memory_percent < 80 ))
|
||||
done < <(kubectl --context "$context" top nodes --no-headers)
|
||||
while read -r _pod _cpu memory; do
|
||||
case $memory in
|
||||
*Gi) memory=$(awk -v value="${memory%Gi}" 'BEGIN {printf "%.0f", value*1024}') ;;
|
||||
*Mi) memory=${memory%Mi} ;;
|
||||
*Ki) memory=$(awk -v value="${memory%Ki}" 'BEGIN {printf "%.0f", value/1024}') ;;
|
||||
*) return 1 ;;
|
||||
esac
|
||||
(( memory < 1536 ))
|
||||
done < <(kubectl --context "$context" -n "$namespace" top pods \
|
||||
-l 'app.kubernetes.io/part-of=easyai-ai-gateway' --no-headers)
|
||||
connections=$(database_query "SELECT count(*) FROM pg_stat_activity WHERE backend_type='client backend';")
|
||||
max_connections=$(database_query "SELECT setting::int FROM pg_settings WHERE name='max_connections';")
|
||||
(( connections < 150 && connections * 4 < max_connections * 3 ))
|
||||
queue=$(database_query "SELECT count(*) FILTER (WHERE status='queued')||':'||count(*) FILTER (WHERE status='running') FROM gateway_tasks WHERE acceptance_run_id='$run_id'::uuid;")
|
||||
[[ $queue == 0:0 ]]
|
||||
duplicate_remote=$(database_query "SELECT count(*) FROM (SELECT remote_task_id FROM gateway_tasks WHERE acceptance_run_id='$run_id'::uuid AND remote_task_id IS NOT NULL GROUP BY remote_task_id HAVING count(*)>1) d;")
|
||||
duplicate_billing=$(database_query "SELECT count(*) FROM (SELECT reference_id,transaction_type FROM gateway_wallet_transactions WHERE reference_type='gateway_task' AND reference_id IN (SELECT id::text FROM gateway_tasks WHERE acceptance_run_id='$run_id'::uuid) GROUP BY reference_id,transaction_type HAVING count(*)>1) d;")
|
||||
[[ $duplicate_remote == 0 && $duplicate_billing == 0 ]]
|
||||
kubectl --context "$context" -n "$namespace" exec deployment/easyai-acceptance-callback-collector -- \
|
||||
wget -qO- http://127.0.0.1:8091/report |
|
||||
jq -e '.duplicates == 0 and .invalid == 0' >/dev/null
|
||||
}
|
||||
|
||||
apply_profile() {
|
||||
local profile=$1 slots pool
|
||||
case $profile in
|
||||
P24) slots=24; pool=32 ;;
|
||||
P28) slots=28; pool=36 ;;
|
||||
P32) slots=32; pool=40 ;;
|
||||
*) return 64 ;;
|
||||
esac
|
||||
local global=$((slots * 2))
|
||||
kubectl --context "$context" -n "$namespace" patch configmap easyai-ai-gateway-config \
|
||||
--type=merge -p "$(jq -cn \
|
||||
--arg slots "$slots" --arg global "$global" \
|
||||
'{data:{
|
||||
AI_GATEWAY_WORKER_AUTOSCALING_ENABLED:"false",
|
||||
AI_GATEWAY_ASYNC_WORKER_INSTANCE_HARD_LIMIT:$slots,
|
||||
AI_GATEWAY_ASYNC_WORKER_HARD_LIMIT:$slots,
|
||||
AI_GATEWAY_ASYNC_WORKER_GLOBAL_HARD_LIMIT:$global,
|
||||
AI_GATEWAY_WORKER_TARGET_OUTSTANDING_PER_REPLICA:$global
|
||||
}}')" >/dev/null
|
||||
kubectl --context "$context" -n "$namespace" scale \
|
||||
deployment/easyai-worker-ningbo deployment/easyai-worker-hongkong --replicas=1 >/dev/null
|
||||
for deployment in easyai-worker-ningbo easyai-worker-hongkong; do
|
||||
kubectl --context "$context" -n "$namespace" set env deployment/"$deployment" \
|
||||
AI_GATEWAY_ASYNC_WORKER_INSTANCE_HARD_LIMIT="$slots" \
|
||||
AI_GATEWAY_DATABASE_MAX_CONNS="$pool" \
|
||||
AI_GATEWAY_MEDIA_MATERIALIZATION_CONCURRENCY="$slots" \
|
||||
AI_GATEWAY_MEDIA_REQUEST_CONCURRENCY="$slots" >/dev/null
|
||||
kubectl --context "$context" -n "$namespace" rollout status deployment/"$deployment" --timeout=10m
|
||||
done
|
||||
for deployment in easyai-api-ningbo easyai-api-hongkong easyai-capacity-controller; do
|
||||
kubectl --context "$context" -n "$namespace" rollout restart deployment/"$deployment" >/dev/null
|
||||
kubectl --context "$context" -n "$namespace" rollout status deployment/"$deployment" --timeout=10m
|
||||
done
|
||||
}
|
||||
|
||||
restore_profile_best_effort() {
|
||||
local profile=$1 slots pool global
|
||||
case $profile in
|
||||
P24) slots=24; pool=32 ;;
|
||||
P28) slots=28; pool=36 ;;
|
||||
P32) slots=32; pool=40 ;;
|
||||
*) return ;;
|
||||
esac
|
||||
global=$((slots * 2))
|
||||
kubectl --context "$context" -n "$namespace" patch configmap easyai-ai-gateway-config \
|
||||
--type=merge -p "$(jq -cn \
|
||||
--arg slots "$slots" --arg global "$global" \
|
||||
'{data:{
|
||||
AI_GATEWAY_WORKER_AUTOSCALING_ENABLED:"false",
|
||||
AI_GATEWAY_ASYNC_WORKER_INSTANCE_HARD_LIMIT:$slots,
|
||||
AI_GATEWAY_ASYNC_WORKER_HARD_LIMIT:$slots,
|
||||
AI_GATEWAY_ASYNC_WORKER_GLOBAL_HARD_LIMIT:$global,
|
||||
AI_GATEWAY_WORKER_TARGET_OUTSTANDING_PER_REPLICA:$global
|
||||
}}')" >/dev/null 2>&1 || true
|
||||
kubectl --context "$context" -n "$namespace" scale \
|
||||
deployment/easyai-worker-ningbo deployment/easyai-worker-hongkong \
|
||||
--replicas=1 >/dev/null 2>&1 || true
|
||||
for deployment in easyai-worker-ningbo easyai-worker-hongkong; do
|
||||
kubectl --context "$context" -n "$namespace" set env deployment/"$deployment" \
|
||||
AI_GATEWAY_ASYNC_WORKER_INSTANCE_HARD_LIMIT="$slots" \
|
||||
AI_GATEWAY_DATABASE_MAX_CONNS="$pool" \
|
||||
AI_GATEWAY_MEDIA_MATERIALIZATION_CONCURRENCY="$slots" \
|
||||
AI_GATEWAY_MEDIA_REQUEST_CONCURRENCY="$slots" >/dev/null 2>&1 || true
|
||||
done
|
||||
}
|
||||
|
||||
run_profile_round() {
|
||||
local profile=$1 repetition=$2 prefix
|
||||
prefix="$report_root/${profile,,}-$repetition"
|
||||
local stop_file="$prefix.resources.stop"
|
||||
sample_resources "$prefix.resources.csv" "$stop_file" &
|
||||
local sampler=$!
|
||||
run_load gemini-baseline "$prefix-gemini-baseline.json"
|
||||
run_load gemini-large "$prefix-gemini-large.json"
|
||||
run_load gemini-peak "$prefix-gemini-peak.json"
|
||||
run_load video-throughput "$prefix-video-throughput.json"
|
||||
run_recovery "$prefix-video-recovery.json"
|
||||
touch "$stop_file"
|
||||
wait "$sampler"
|
||||
verify_hard_gates
|
||||
}
|
||||
|
||||
wait_for_recovery_owner() {
|
||||
local deadline=$((SECONDS + 120)) owner
|
||||
while (( SECONDS < deadline )); do
|
||||
owner=$(database_query "
|
||||
SELECT task.id::text||','||worker.pod_name||','||worker.site
|
||||
FROM gateway_tasks task
|
||||
JOIN gateway_worker_instances worker ON worker.instance_id=task.locked_by
|
||||
WHERE task.acceptance_run_id='$run_id'::uuid
|
||||
AND task.status='running'
|
||||
AND task.request::text LIKE '%acceptance-long-recovery%'
|
||||
AND worker.status='active'
|
||||
ORDER BY task.updated_at
|
||||
LIMIT 1;")
|
||||
[[ -z $owner ]] || { printf '%s\n' "$owner"; return; }
|
||||
sleep 1
|
||||
done
|
||||
return 1
|
||||
}
|
||||
|
||||
run_recovery() {
|
||||
local report_path=$1 owner task_id pod site
|
||||
run_load video-recovery "$report_path" &
|
||||
active_load_pid=$!
|
||||
owner=$(wait_for_recovery_owner)
|
||||
IFS=',' read -r task_id pod site <<<"$owner"
|
||||
[[ $task_id =~ ^[0-9a-f-]{36}$ && $pod == easyai-worker-"$site"-* ]]
|
||||
kubectl --context "$context" -n "$namespace" delete pod "$pod" \
|
||||
--force --grace-period=0 --wait=false >/dev/null
|
||||
kubectl --context "$context" -n "$namespace" rollout status \
|
||||
deployment/easyai-worker-"$site" --timeout=10m
|
||||
wait "$active_load_pid"
|
||||
active_load_pid=
|
||||
}
|
||||
|
||||
run_fault_matrix() {
|
||||
local duration=${AI_GATEWAY_LOCAL_WEAK_LINK_DURATION:-5m}
|
||||
"$script_dir/network-fault.sh" weak-link
|
||||
netem_active=true
|
||||
run_load mixed-soak "$report_root/fault-weak-link.json" \
|
||||
-duration "$duration" -image-rate 1 -video-rate 1
|
||||
"$script_dir/network-fault.sh" reset
|
||||
netem_active=false
|
||||
|
||||
run_load video-throughput "$report_root/fault-upstream-outage.json" &
|
||||
active_load_pid=$!
|
||||
sleep 5
|
||||
"$script_dir/network-fault.sh" upstream-outage
|
||||
netem_active=true
|
||||
sleep 10
|
||||
"$script_dir/network-fault.sh" reset
|
||||
netem_active=false
|
||||
wait "$active_load_pid"
|
||||
active_load_pid=
|
||||
|
||||
run_load video-recovery "$report_root/fault-database-outage.json" &
|
||||
active_load_pid=$!
|
||||
sleep 10
|
||||
"$script_dir/network-fault.sh" database-outage hongkong
|
||||
netem_active=true
|
||||
sleep 30
|
||||
"$script_dir/network-fault.sh" reset
|
||||
netem_active=false
|
||||
wait "$active_load_pid"
|
||||
active_load_pid=
|
||||
|
||||
local leader
|
||||
leader=$(kubectl --context "$context" -n "$namespace" get pods \
|
||||
-l app.kubernetes.io/name=easyai-capacity-controller -o name |
|
||||
while read -r pod; do
|
||||
status=$(kubectl --context "$context" -n "$namespace" exec "$pod" -- \
|
||||
wget -qO- http://127.0.0.1:8088/status)
|
||||
[[ $(jq -r '.leader' <<<"$status") == true ]] && { printf '%s\n' "${pod#pod/}"; break; }
|
||||
done)
|
||||
[[ -n $leader ]]
|
||||
kubectl --context "$context" -n "$namespace" delete pod "$leader" --wait=false >/dev/null
|
||||
kubectl --context "$context" -n "$namespace" rollout status \
|
||||
deployment/easyai-capacity-controller --timeout=5m
|
||||
verify_hard_gates
|
||||
}
|
||||
|
||||
run_autoscaling() {
|
||||
local profile=$1 slots
|
||||
slots=${profile#P}
|
||||
kubectl --context "$context" -n "$namespace" patch configmap easyai-ai-gateway-config \
|
||||
--type=merge -p "$(jq -cn --arg global "$((slots * 4))" \
|
||||
'{data:{
|
||||
AI_GATEWAY_WORKER_AUTOSCALING_ENABLED:"true",
|
||||
AI_GATEWAY_WORKER_MIN_REPLICAS_NINGBO:"1",
|
||||
AI_GATEWAY_WORKER_MIN_REPLICAS_HONGKONG:"1",
|
||||
AI_GATEWAY_WORKER_MAX_REPLICAS_NINGBO:"2",
|
||||
AI_GATEWAY_WORKER_MAX_REPLICAS_HONGKONG:"2",
|
||||
AI_GATEWAY_ASYNC_WORKER_GLOBAL_HARD_LIMIT:$global
|
||||
}}')" >/dev/null
|
||||
kubectl --context "$context" -n "$namespace" rollout restart \
|
||||
deployment/easyai-capacity-controller >/dev/null
|
||||
kubectl --context "$context" -n "$namespace" rollout status \
|
||||
deployment/easyai-capacity-controller --timeout=5m
|
||||
run_load video-throughput "$report_root/autoscaling-load.json" &
|
||||
active_load_pid=$!
|
||||
local deadline=$((SECONDS + 420)) total=2
|
||||
while (( SECONDS < deadline )); do
|
||||
total=$(kubectl --context "$context" -n "$namespace" get deployment \
|
||||
easyai-worker-ningbo easyai-worker-hongkong \
|
||||
-o json | jq '[.items[].spec.replicas] | add')
|
||||
(( total >= 3 )) && break
|
||||
sleep 5
|
||||
done
|
||||
(( total >= 3 ))
|
||||
wait "$active_load_pid"
|
||||
active_load_pid=
|
||||
deadline=$((SECONDS + 780))
|
||||
while (( SECONDS < deadline )); do
|
||||
total=$(kubectl --context "$context" -n "$namespace" get deployment \
|
||||
easyai-worker-ningbo easyai-worker-hongkong \
|
||||
-o json | jq '[.items[].spec.replicas] | add')
|
||||
(( total == 2 )) && break
|
||||
sleep 10
|
||||
done
|
||||
(( total == 2 ))
|
||||
verify_hard_gates
|
||||
}
|
||||
|
||||
artifact_smoke() {
|
||||
local manifest=$1
|
||||
node "$repository_root/scripts/release-manifest.mjs" validate "$manifest" >/dev/null
|
||||
local manifest_sha api_image web_image api_digest artifact_report
|
||||
manifest_sha=$(node "$repository_root/scripts/release-manifest.mjs" get "$manifest" sourceSha)
|
||||
api_image=$(node "$repository_root/scripts/release-manifest.mjs" get "$manifest" images.api)
|
||||
web_image=$(node "$repository_root/scripts/release-manifest.mjs" get "$manifest" images.web)
|
||||
api_digest=${api_image##*@}
|
||||
[[ $manifest_sha == "$release_sha" && $api_digest =~ ^sha256:[0-9a-f]{64}$ ]]
|
||||
docker pull --platform linux/amd64 "$api_image" >/dev/null
|
||||
docker pull --platform linux/amd64 "$web_image" >/dev/null
|
||||
"$private_root/tools/k3d" image import -c easyai-acceptance-local "$api_image" "$web_image"
|
||||
for deployment in easyai-api-ningbo easyai-api-hongkong easyai-worker-ningbo \
|
||||
easyai-worker-hongkong easyai-capacity-controller easyai-acceptance-emulator \
|
||||
easyai-acceptance-callback-collector; do
|
||||
kubectl --context "$context" -n "$namespace" set image deployment/"$deployment" \
|
||||
"*=$api_image" >/dev/null
|
||||
kubectl --context "$context" -n "$namespace" rollout status deployment/"$deployment" --timeout=15m
|
||||
done
|
||||
for deployment in easyai-acceptance-edge-ningbo easyai-acceptance-edge-hongkong; do
|
||||
kubectl --context "$context" -n "$namespace" set image deployment/"$deployment" \
|
||||
"*=$web_image" >/dev/null
|
||||
kubectl --context "$context" -n "$namespace" rollout status deployment/"$deployment" --timeout=10m
|
||||
done
|
||||
kubectl --context "$context" -n "$namespace" delete job easyai-artifact-migrate \
|
||||
--ignore-not-found --wait=true >/dev/null
|
||||
cat <<EOF | kubectl --context "$context" apply -f - >/dev/null
|
||||
apiVersion: batch/v1
|
||||
kind: Job
|
||||
metadata:
|
||||
name: easyai-artifact-migrate
|
||||
namespace: easyai
|
||||
spec:
|
||||
backoffLimit: 0
|
||||
template:
|
||||
spec:
|
||||
restartPolicy: Never
|
||||
containers:
|
||||
- name: migrate
|
||||
image: $api_image
|
||||
command: ["/bin/sh", "-ec", "cd /app && exec /app/easyai-ai-gateway-migrate"]
|
||||
envFrom:
|
||||
- secretRef:
|
||||
name: easyai-ai-gateway-runtime
|
||||
EOF
|
||||
kubectl --context "$context" -n "$namespace" wait \
|
||||
--for=condition=complete job/easyai-artifact-migrate --timeout=10m >/dev/null
|
||||
run_load simulated-smoke "$report_root/artifact-simulated-smoke.json"
|
||||
verify_hard_gates
|
||||
artifact_report="$report_root/artifact-smoke.json"
|
||||
jq -n \
|
||||
--arg releaseSha "$manifest_sha" \
|
||||
--arg apiImageDigest "$api_digest" \
|
||||
--arg webImageDigest "${web_image##*@}" \
|
||||
'{
|
||||
schemaVersion:"acceptance-artifact-smoke/v1",
|
||||
releaseSha:$releaseSha,
|
||||
apiImageDigest:$apiImageDigest,
|
||||
webImageDigest:$webImageDigest,
|
||||
architecture:"linux/amd64",
|
||||
migrationSmoke:true,
|
||||
startupSmoke:true,
|
||||
mediaSmoke:true,
|
||||
passed:true,
|
||||
secretSafe:true
|
||||
}' >"$artifact_report"
|
||||
chmod 0600 "$artifact_report"
|
||||
}
|
||||
|
||||
quick() {
|
||||
current_phase=quick
|
||||
run_load simulated-smoke "$report_root/quick.json"
|
||||
verify_hard_gates
|
||||
echo "local_acceptance_quick=PASS run_id=$run_id report=$report_root/quick.json"
|
||||
}
|
||||
|
||||
full() {
|
||||
local manifest=$1 profile repetition
|
||||
quick
|
||||
for profile in P24 P28 P32; do
|
||||
current_phase="capacity_${profile,,}_apply"
|
||||
apply_profile "$profile"
|
||||
for repetition in 1 2 3; do
|
||||
current_phase="capacity_${profile,,}_round_$repetition"
|
||||
run_profile_round "$profile" "$repetition"
|
||||
done
|
||||
stable_profile=$profile
|
||||
done
|
||||
current_phase=fault_matrix
|
||||
run_fault_matrix
|
||||
current_phase=autoscaling_and_drain
|
||||
run_autoscaling "$stable_profile"
|
||||
current_phase=certified_soak
|
||||
run_load mixed-soak "$report_root/mixed-soak.json" \
|
||||
-duration "${AI_GATEWAY_LOCAL_SOAK_DURATION:-2h}" \
|
||||
-image-rate "${AI_GATEWAY_LOCAL_CERTIFIED_IMAGE_RATE:-1}" \
|
||||
-video-rate "${AI_GATEWAY_LOCAL_CERTIFIED_VIDEO_RATE:-1}"
|
||||
current_phase=overload_shedding
|
||||
run_load mixed-overload "$report_root/mixed-overload.json" \
|
||||
-duration "${AI_GATEWAY_LOCAL_OVERLOAD_DURATION:-10m}" \
|
||||
-image-rate "${AI_GATEWAY_LOCAL_OVERLOAD_IMAGE_RATE:-2}" \
|
||||
-video-rate "${AI_GATEWAY_LOCAL_OVERLOAD_VIDEO_RATE:-2}"
|
||||
current_phase=amd64_artifact_smoke
|
||||
artifact_smoke "$manifest"
|
||||
current_phase=final_report
|
||||
node "$script_dir/report.mjs" build-local \
|
||||
--runtime "$runtime" \
|
||||
--snapshot "$snapshot" \
|
||||
--reports "$report_root" \
|
||||
--artifact "$report_root/artifact-smoke.json" \
|
||||
--api-digest "$(node "$repository_root/scripts/release-manifest.mjs" get "$manifest" images.api | sed 's/.*@//')" \
|
||||
--certified-profile "$stable_profile" \
|
||||
--output "$report_root/acceptance-report.json"
|
||||
echo "local_acceptance_full=PASS run_id=$run_id certified_profile=$stable_profile report=$report_root/acceptance-report.json"
|
||||
}
|
||||
|
||||
command=${1:-}
|
||||
shift || true
|
||||
require_local_cluster
|
||||
load_environment
|
||||
case $command in
|
||||
quick)
|
||||
[[ $# -eq 0 ]] || { usage >&2; exit 64; }
|
||||
quick
|
||||
;;
|
||||
artifact-smoke)
|
||||
[[ ${1:-} == --release-manifest && $# -eq 2 ]] || { usage >&2; exit 64; }
|
||||
artifact_smoke "$2"
|
||||
;;
|
||||
full)
|
||||
[[ ${1:-} == --release-manifest && $# -eq 2 ]] || { usage >&2; exit 64; }
|
||||
full "$2"
|
||||
;;
|
||||
*)
|
||||
usage >&2
|
||||
exit 64
|
||||
;;
|
||||
esac
|
||||
Reference in New Issue
Block a user