#!/usr/bin/env bash set -euo pipefail script_dir=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd) # shellcheck source=scripts/cluster/common.sh source "$script_dir/common.sh" usage() { cat <<'EOF' Usage: scripts/cluster/run-production-acceptance.sh --execute dist/releases/.json This is a destructive production acceptance workflow. It pauses new live tasks, changes Worker capacity, creates paid real canaries, and deletes one Worker Pod during each recovery phase. It also grants the dedicated acceptance API Key access to the selected production candidates. A failed run deliberately keeps traffic in validation mode and restores only the last stable capacity profile. Required private environment values: AI_GATEWAY_ACCEPTANCE_ADMIN_TOKEN AI_GATEWAY_ACCEPTANCE_API_KEY AI_GATEWAY_ACCEPTANCE_API_KEY_ID AI_GATEWAY_ACCEPTANCE_USER_ID AI_GATEWAY_ACCEPTANCE_GATEWAYS AI_GATEWAY_ACCEPTANCE_REAL_IMAGE_URLS Optional model overrides (otherwise selected from current production candidates): AI_GATEWAY_ACCEPTANCE_GEMINI_MODEL AI_GATEWAY_ACCEPTANCE_VIDEO_MODEL AI_GATEWAY_ACCEPTANCE_GATEWAY_TLS_SERVER_NAME EOF } [[ ${1:-} == --execute && $# -eq 2 ]] || { usage >&2 exit 64 } release_manifest=$2 [[ -f $release_manifest && ! -L $release_manifest ]] || { echo 'release manifest must be a regular file' >&2 exit 1 } load_cluster_env require_commands curl git jq node openssl sed : "${AI_GATEWAY_ACCEPTANCE_ADMIN_TOKEN:?}" : "${AI_GATEWAY_ACCEPTANCE_API_KEY:?}" : "${AI_GATEWAY_ACCEPTANCE_API_KEY_ID:?}" : "${AI_GATEWAY_ACCEPTANCE_USER_ID:?}" : "${AI_GATEWAY_ACCEPTANCE_GEMINI_MODEL:=}" : "${AI_GATEWAY_ACCEPTANCE_VIDEO_MODEL:=}" : "${AI_GATEWAY_ACCEPTANCE_GATEWAY_TLS_SERVER_NAME:=}" : "${AI_GATEWAY_ACCEPTANCE_GATEWAYS:?}" : "${AI_GATEWAY_ACCEPTANCE_REAL_IMAGE_URLS:?}" node "$cluster_root/scripts/release-manifest.mjs" validate "$release_manifest" >/dev/null release_sha=$(node "$cluster_root/scripts/release-manifest.mjs" get "$release_manifest" sourceSha) base_sha=$(node "$cluster_root/scripts/release-manifest.mjs" get "$release_manifest" baseReleaseSha) api_image=$(node "$cluster_root/scripts/release-manifest.mjs" get "$release_manifest" images.api) [[ $release_sha =~ ^[0-9a-f]{40}$ && $api_image =~ @sha256:[0-9a-f]{64}$ ]] || { echo 'release manifest is not full-SHA and digest pinned' >&2 exit 1 } api_digest=${api_image##*@} worker_digest=$api_digest [[ $(git -C "$cluster_root" rev-parse HEAD) == "$release_sha" ]] || { echo 'working copy HEAD must equal the release SHA' >&2 exit 1 } [[ -z $(git -C "$cluster_root" status --short) ]] || { echo 'production acceptance requires a clean release working copy' >&2 exit 1 } namespace=${AI_GATEWAY_K3S_NAMESPACE:-easyai} remote_release_helper=${AI_GATEWAY_REMOTE_RELEASE_HELPER:-/usr/local/sbin/easyai-ai-gateway-cluster-release} [[ $namespace =~ ^[a-z0-9-]+$ && $remote_release_helper =~ ^/[A-Za-z0-9._/-]+$ ]] || { echo 'invalid namespace or remote release helper' >&2 exit 1 } [[ $namespace == easyai ]] || { echo 'the production acceptance manifests currently require namespace easyai' >&2 exit 1 } temporary_root=$(mktemp -d) chmod 0700 "$temporary_root" current_manifest=$temporary_root/current.json run_token=$(openssl rand -hex 32) run_id= report_root= stable_profile=P24 active_profile=P24 failure_reason= failure_recorded=false acceptance_admin_base= cleanup() { local status=$? trap - EXIT if (( status != 0 )) && [[ -n $run_id && $failure_recorded != true ]]; then set +e apply_capacity_profile "$stable_profile" >/dev/null 2>&1 mark_run_failed 'acceptance workflow exited unexpectedly' set -e fi rm -rf -- "$temporary_root" exit "$status" } trap cleanup EXIT remote_kubectl() { local command_text printf -v command_text '%q ' "$@" cluster_ssh "$CLUSTER_NINGBO_HOST" "k3s kubectl $command_text" } database_query() { local sql=$1 local primary primary=$(remote_kubectl get cluster easyai-postgres -n "$namespace" -o 'jsonpath={.status.currentPrimary}') [[ $primary =~ ^easyai-postgres-[0-9]+$ ]] remote_kubectl exec -n "$namespace" "$primary" -c postgres -- \ psql -X -v ON_ERROR_STOP=1 -U postgres -d easyai_ai_gateway -At -c "$sql" } admin_request() { local method=$1 local path=$2 local body=${3:-} local output=$4 local status response payload token_encoded body_encoded has_body=false local service_ip service_port remote_script remote_command if [[ -z $acceptance_admin_base ]]; then service_ip=$(remote_kubectl get service api-ningbo-edge -n "$namespace" \ -o 'jsonpath={.spec.clusterIP}') service_port=$(remote_kubectl get service api-ningbo-edge -n "$namespace" \ -o 'jsonpath={.spec.ports[0].port}') [[ $service_ip =~ ^[0-9.]+$ && $service_port =~ ^[0-9]+$ ]] || { echo 'acceptance admin Service has an invalid cluster address' >&2 return 1 } acceptance_admin_base=http://$service_ip:$service_port fi token_encoded=$(printf '%s' "$AI_GATEWAY_ACCEPTANCE_ADMIN_TOKEN" | openssl base64 -A) body_encoded=$(printf '%s' "$body" | openssl base64 -A) [[ -n $body ]] && has_body=true # shellcheck disable=SC2016 # This script is quoted for execution on the remote node. remote_script=' set -euo pipefail admin_base=$1 method=$2 path=$3 token=$(printf "%s" "$4" | base64 -d) body=$(printf "%s" "$5" | base64 -d) has_body=$6 if [[ $has_body == true ]]; then curl -sS --max-time 30 -w "\n%{http_code}" -X "$method" \ -H "Authorization: Bearer $token" \ -H "Content-Type: application/json" \ --data-binary "$body" "$admin_base$path" else curl -sS --max-time 30 -w "\n%{http_code}" -X "$method" \ -H "Authorization: Bearer $token" "$admin_base$path" fi ' printf -v remote_command 'bash -c %q -- %q %q %q %q %q %q' \ "$remote_script" "$acceptance_admin_base" "$method" "$path" \ "$token_encoded" "$body_encoded" "$has_body" if ! response=$(cluster_ssh "$CLUSTER_NINGBO_HOST" "$remote_command"); then echo "acceptance control API transport failed: method=$method path=$path" >&2 return 1 fi status=${response##*$'\n'} payload=${response%$'\n'*} printf '%s' "$payload" >"$output" [[ $status =~ ^2[0-9][0-9]$ ]] || { echo "acceptance control API failed: method=$method path=$path status=$status" >&2 return 1 } } verify_release_cas() { cluster_ssh "$CLUSTER_NINGBO_HOST" "$remote_release_helper status" >"$current_manifest" node "$cluster_root/scripts/release-manifest.mjs" validate "$current_manifest" >/dev/null local current_sha current_api current_sha=$(node "$cluster_root/scripts/release-manifest.mjs" get "$current_manifest" sourceSha) current_api=$(node "$cluster_root/scripts/release-manifest.mjs" get "$current_manifest" images.api) [[ $current_sha == "$release_sha" && $current_api == "$api_image" ]] || { echo "production release CAS mismatch: expected=$release_sha current=$current_sha" >&2 return 1 } for deployment in easyai-api-ningbo easyai-api-hongkong easyai-worker-ningbo easyai-worker-hongkong; do [[ $(remote_kubectl get deployment "$deployment" -n "$namespace" \ -o 'jsonpath={.spec.template.spec.containers[0].image}') == "$api_image" ]] done [[ $(remote_kubectl get deployment easyai-worker-ningbo -n "$namespace" \ -o 'jsonpath={.spec.replicas}') == 1 ]] [[ $(remote_kubectl get deployment easyai-worker-hongkong -n "$namespace" \ -o 'jsonpath={.spec.replicas}') == 1 ]] } wait_for_existing_tasks_to_drain() { local deadline=$((SECONDS + 900)) local active while (( SECONDS < deadline )); do active=$(database_query "SELECT count(*) FROM gateway_tasks WHERE acceptance_run_id IS NULL AND status IN ('queued','running');") if [[ $active == 0 ]]; then return 0 fi echo "waiting_for_existing_tasks=$active" sleep 5 done echo 'existing production tasks did not drain within 15 minutes' >&2 return 1 } ensure_acceptance_user_group() { local groups_response=$temporary_root/user-groups.json local group_response=$temporary_root/acceptance-user-group.json local users_response=$temporary_root/users.json local user_response=$temporary_root/acceptance-user.json local group_id group_body user_body isolation_state admin_request GET /api/admin/user-groups '' "$groups_response" group_id=$(jq -r ' [.items[] | select(.groupKey == "production-acceptance") | .id] | if length == 1 then .[0] else empty end ' "$groups_response") group_body=$(jq -cn '{ groupKey:"production-acceptance", name:"Production Acceptance", description:"Dedicated production-isomorphic acceptance traffic", source:"gateway", priority:1, rechargeDiscountPolicy:{}, billingDiscountPolicy:{}, rateLimitPolicy:{rules:[]}, quotaPolicy:{}, metadata:{purpose:"production_acceptance",isolated:true}, status:"active" }') if [[ -z $group_id ]]; then [[ $(jq '[.items[] | select(.groupKey == "production-acceptance")] | length' \ "$groups_response") == 0 ]] || { echo 'acceptance user group lookup was ambiguous' >&2 return 1 } admin_request POST /api/admin/user-groups "$group_body" "$group_response" group_id=$(jq -r '.id // empty' "$group_response") else [[ $group_id =~ ^[0-9a-f-]{36}$ ]] || { echo 'acceptance user group API returned an invalid group ID' >&2 return 1 } isolation_state=$(database_query " SELECT (SELECT count(*) FROM gateway_users WHERE default_user_group_id='$group_id'::uuid AND id <> '$AI_GATEWAY_ACCEPTANCE_USER_ID'::uuid) ||':'|| (SELECT count(*) FROM gateway_api_keys WHERE user_group_id='$group_id'::uuid AND gateway_user_id <> '$AI_GATEWAY_ACCEPTANCE_USER_ID'::uuid);") [[ $isolation_state == 0:0 ]] || { echo 'acceptance user group is referenced by non-acceptance identities' >&2 return 1 } admin_request PATCH "/api/admin/user-groups/$group_id" \ "$group_body" "$group_response" fi [[ $group_id =~ ^[0-9a-f-]{36}$ ]] || { echo 'acceptance user group API returned an invalid group ID' >&2 return 1 } admin_request GET /api/admin/users '' "$users_response" user_body=$(jq -c \ --arg userID "$AI_GATEWAY_ACCEPTANCE_USER_ID" \ --arg groupID "$group_id" ' [.items[] | select(.id == $userID)] | if length != 1 then empty else .[0] end | { userKey:(.userKey // ""), source:(.source // "gateway"), externalUserId:(.externalUserId // ""), username:(.username // ""), displayName:(.displayName // ""), email:(.email // ""), phone:(.phone // ""), avatarUrl:(.avatarUrl // ""), gatewayTenantId:(.gatewayTenantId // ""), tenantId:(.tenantId // ""), tenantKey:(.tenantKey // ""), defaultUserGroupId:$groupID, roles:(.roles // []), authProfile:(.authProfile // {}), metadata:(.metadata // {}), status:(.status // "active") } ' "$users_response") [[ -n $user_body && $(jq -r '.username // empty' <<<"$user_body") != '' ]] || { echo 'dedicated acceptance user was not found through the control API' >&2 return 1 } admin_request PATCH "/api/admin/users/$AI_GATEWAY_ACCEPTANCE_USER_ID" \ "$user_body" "$user_response" isolation_state=$(database_query " SELECT (SELECT count(*) FROM gateway_users WHERE id='$AI_GATEWAY_ACCEPTANCE_USER_ID'::uuid AND default_user_group_id='$group_id'::uuid) ||':'|| (SELECT count(*) FROM gateway_api_keys WHERE id='$AI_GATEWAY_ACCEPTANCE_API_KEY_ID'::uuid AND gateway_user_id='$AI_GATEWAY_ACCEPTANCE_USER_ID'::uuid AND COALESCE(user_group_id,'$group_id'::uuid)='$group_id'::uuid AND status='active' AND deleted_at IS NULL) ||':'|| (SELECT count(*) FROM gateway_api_keys WHERE gateway_user_id='$AI_GATEWAY_ACCEPTANCE_USER_ID'::uuid AND status='active' AND deleted_at IS NULL) ||':'|| (SELECT count(*) FROM gateway_user_groups WHERE id='$group_id'::uuid AND status='active' AND NOT EXISTS ( SELECT 1 FROM jsonb_array_elements( COALESCE(rate_limit_policy->'rules','[]'::jsonb) ) rule WHERE rule->>'metric' IN ('concurrent','queue_size') )) ||':'|| (SELECT count(*) FROM gateway_users WHERE default_user_group_id='$group_id'::uuid AND id <> '$AI_GATEWAY_ACCEPTANCE_USER_ID'::uuid) ||':'|| (SELECT count(*) FROM gateway_api_keys WHERE user_group_id='$group_id'::uuid AND gateway_user_id <> '$AI_GATEWAY_ACCEPTANCE_USER_ID'::uuid);") [[ $isolation_state == 1:1:1:1:0:0 ]] || { echo 'dedicated acceptance identity isolation verification failed' >&2 return 1 } echo 'acceptance_user_group=PASS rate_limit=production_candidates active_keys=1 isolated=true' } select_acceptance_models() { if [[ -z $AI_GATEWAY_ACCEPTANCE_GEMINI_MODEL ]]; then AI_GATEWAY_ACCEPTANCE_GEMINI_MODEL=$(database_query " SELECT b.invocation_name FROM platform_models model JOIN integration_platforms platform ON platform.id=model.platform_id JOIN base_model_catalog b ON b.id=model.base_model_id WHERE platform.status='enabled' AND platform.deleted_at IS NULL AND model.enabled=true AND lower(platform.provider) IN ('gemini','google-gemini','gemini-openai') AND model.model_type @> '[\"image_edit\"]'::jsonb ORDER BY CASE WHEN lower(b.invocation_name) LIKE '%gemini%image%' THEN 0 ELSE 1 END, platform.priority, model.created_at LIMIT 1;") fi if [[ -z $AI_GATEWAY_ACCEPTANCE_VIDEO_MODEL ]]; then AI_GATEWAY_ACCEPTANCE_VIDEO_MODEL=$(database_query " SELECT b.invocation_name FROM platform_models model JOIN integration_platforms platform ON platform.id=model.platform_id JOIN base_model_catalog b ON b.id=model.base_model_id WHERE platform.status='enabled' AND platform.deleted_at IS NULL AND model.enabled=true AND model.model_type @> '[\"omni_video\"]'::jsonb AND COALESCE( NULLIF(model.capability_override #>> '{omni_video,max_images}','')::int, NULLIF(model.capabilities #>> '{omni_video,max_images}','')::int, NULLIF(b.capabilities #>> '{omni_video,max_images}','')::int, 0 ) >= 9 ORDER BY CASE WHEN lower(b.invocation_name) LIKE '%seedance%2%fast%' THEN 0 WHEN lower(b.invocation_name) LIKE '%seedance%2%' THEN 1 ELSE 2 END, platform.priority, model.created_at LIMIT 1;") fi [[ -n $AI_GATEWAY_ACCEPTANCE_GEMINI_MODEL && -n $AI_GATEWAY_ACCEPTANCE_VIDEO_MODEL ]] || { echo 'no enabled Gemini image-edit or omni-video max_images>=9 model was found' >&2 return 1 } [[ $AI_GATEWAY_ACCEPTANCE_GEMINI_MODEL =~ ^[A-Za-z0-9._:/+-]+$ && $AI_GATEWAY_ACCEPTANCE_VIDEO_MODEL =~ ^[A-Za-z0-9._:/+-]+$ ]] || { echo 'acceptance model invocation names contain unsupported characters' >&2 return 1 } export AI_GATEWAY_ACCEPTANCE_GEMINI_MODEL AI_GATEWAY_ACCEPTANCE_VIDEO_MODEL } ensure_acceptance_model_access() { local resources response body missing expected response=$temporary_root/model-access.json resources=$(database_query " WITH selected_candidates AS ( SELECT DISTINCT platform.id AS platform_id, model.id AS platform_model_id, model.base_model_id FROM platform_models model JOIN integration_platforms platform ON platform.id=model.platform_id JOIN base_model_catalog base_model ON base_model.id=model.base_model_id WHERE platform.status='enabled' AND platform.deleted_at IS NULL AND model.enabled=true AND base_model.invocation_name IN ( '$AI_GATEWAY_ACCEPTANCE_GEMINI_MODEL', '$AI_GATEWAY_ACCEPTANCE_VIDEO_MODEL' ) ), resources AS ( SELECT 'platform'::text AS resource_type, platform_id AS resource_id FROM selected_candidates UNION SELECT 'platform_model', platform_model_id FROM selected_candidates UNION SELECT 'base_model', base_model_id FROM selected_candidates WHERE base_model_id IS NOT NULL ) SELECT COALESCE( jsonb_agg( jsonb_build_object( 'resourceType', resource_type, 'resourceId', resource_id, 'priority', 100, 'minPermissionLevel', 0, 'status', 'active' ) ORDER BY resource_type, resource_id ), '[]'::jsonb )::text FROM resources;") expected=$(jq 'length' <<<"$resources") (( expected > 0 )) || { echo 'selected acceptance models did not resolve to production candidate resources' >&2 return 1 } body=$(jq -cn \ --arg subjectID "$AI_GATEWAY_ACCEPTANCE_API_KEY_ID" \ --argjson resources "$resources" \ '{ subjectType:"api_key", subjectId:$subjectID, effect:"allow", upsertResources:$resources, deleteResources:[] }') admin_request POST /api/admin/access-rules/batch "$body" "$response" missing=$(database_query " WITH selected_candidates AS ( SELECT DISTINCT platform.id AS platform_id, model.id AS platform_model_id, model.base_model_id FROM platform_models model JOIN integration_platforms platform ON platform.id=model.platform_id JOIN base_model_catalog base_model ON base_model.id=model.base_model_id WHERE platform.status='enabled' AND platform.deleted_at IS NULL AND model.enabled=true AND base_model.invocation_name IN ( '$AI_GATEWAY_ACCEPTANCE_GEMINI_MODEL', '$AI_GATEWAY_ACCEPTANCE_VIDEO_MODEL' ) ), resources AS ( SELECT 'platform'::text AS resource_type, platform_id AS resource_id FROM selected_candidates UNION SELECT 'platform_model', platform_model_id FROM selected_candidates UNION SELECT 'base_model', base_model_id FROM selected_candidates WHERE base_model_id IS NOT NULL ) SELECT count(*) FROM resources resource WHERE NOT EXISTS ( SELECT 1 FROM gateway_access_rules rule WHERE rule.subject_type='api_key' AND rule.subject_id='$AI_GATEWAY_ACCEPTANCE_API_KEY_ID'::uuid AND rule.effect='allow' AND rule.status='active' AND rule.resource_type=resource.resource_type AND rule.resource_id=resource.resource_id );") [[ $missing == 0 ]] || { echo "acceptance API Key is missing $missing selected candidate access rules" >&2 return 1 } echo "acceptance_model_access=PASS resources=$expected" } deploy_protocol_emulator() { sed "s|image: easyai-api|image: $api_image|" \ "$cluster_root/deploy/kubernetes/acceptance/protocol-emulator.yaml" | cluster_ssh "$CLUSTER_NINGBO_HOST" 'k3s kubectl apply -f -' >/dev/null remote_kubectl rollout status deployment/easyai-acceptance-emulator \ -n "$namespace" --timeout=300s } create_and_activate_run() { local create_response=$temporary_root/create-run.json local activate_response=$temporary_root/activate-run.json local traffic_response=$temporary_root/pre-activate-traffic.json local previous_run_response=$temporary_root/previous-run.json local abort_response=$temporary_root/abort-previous-run.json local body body=$(jq -cn \ --arg releaseSha "$release_sha" \ --arg apiDigest "$api_digest" \ --arg workerDigest "$worker_digest" \ --arg apiKeyID "$AI_GATEWAY_ACCEPTANCE_API_KEY_ID" \ --arg userID "$AI_GATEWAY_ACCEPTANCE_USER_ID" \ --arg token "$run_token" \ --arg emulatorURL 'http://easyai-acceptance-emulator.easyai.svc.cluster.local:8090' \ --arg callbackURL 'http://easyai-acceptance-emulator.easyai.svc.cluster.local:8090/callbacks' \ '{ releaseSha: $releaseSha, apiImageDigest: $apiDigest, workerImageDigest: $workerDigest, apiKeyId: $apiKeyID, userId: $userID, token: $token, emulatorBaseUrl: $emulatorURL, callbackUrl: $callbackURL, capacityProfile: "P24", config: {workloads: ["gemini_image_edit", "multi_reference_video"]} }') admin_request POST /api/admin/system/acceptance/runs "$body" "$create_response" run_id=$(jq -r '.id // empty' "$create_response") [[ $run_id =~ ^[0-9a-f-]{36}$ ]] || { echo 'acceptance control API returned an invalid Run ID' >&2 return 1 } report_root=${AI_GATEWAY_ACCEPTANCE_REPORT_DIR:-"$cluster_root/dist/acceptance/$run_id"} mkdir -p "$report_root" chmod 0700 "$report_root" admin_request GET /api/admin/system/acceptance/traffic-mode '' "$traffic_response" local traffic_mode previous_run_id previous_status local traffic_release traffic_api_digest traffic_worker_digest local previous_release previous_api_digest previous_worker_digest traffic_mode=$(jq -r '.mode // empty' "$traffic_response") if [[ $traffic_mode == validation ]]; then previous_run_id=$(jq -r '.runId // empty' "$traffic_response") [[ $previous_run_id =~ ^[0-9a-f-]{36}$ && $previous_run_id != "$run_id" ]] || { echo 'validation mode does not reference a different previous Run' >&2 return 1 } admin_request GET "/api/admin/system/acceptance/runs/$previous_run_id" '' \ "$previous_run_response" previous_status=$(jq -r '.status // empty' "$previous_run_response") [[ $previous_status == failed ]] || { echo "refusing to replace previous acceptance Run in status $previous_status" >&2 return 1 } traffic_release=$(jq -r '.releaseSha // empty' "$traffic_response") traffic_api_digest=$(jq -r '.apiImageDigest // empty' "$traffic_response") traffic_worker_digest=$(jq -r '.workerImageDigest // empty' "$traffic_response") previous_release=$(jq -r '.releaseSha // empty' "$previous_run_response") previous_api_digest=$(jq -r '.apiImageDigest // empty' "$previous_run_response") previous_worker_digest=$(jq -r '.workerImageDigest // empty' "$previous_run_response") [[ $traffic_release == "$previous_release" && $traffic_api_digest == "$previous_api_digest" && $traffic_worker_digest == "$previous_worker_digest" ]] || { echo 'previous failed Run does not match the active validation CAS fields' >&2 return 1 } if [[ $traffic_release == "$release_sha" ]]; then [[ $traffic_api_digest == "$api_digest" && $traffic_worker_digest == "$worker_digest" ]] || { echo 'current-release validation mode has unexpected image digests' >&2 return 1 } elif [[ -z $base_sha || $traffic_release != "$base_sha" ]]; then echo 'refusing to replace validation mode outside the direct release base' >&2 return 1 fi body=$(jq -cn \ --argjson revision "$(jq -r '.revision' "$traffic_response")" \ --arg releaseSha "$traffic_release" \ --arg apiDigest "$traffic_api_digest" \ --arg workerDigest "$traffic_worker_digest" \ '{ revision:$revision, releaseSha:$releaseSha, apiImageDigest:$apiDigest, workerImageDigest:$workerDigest }') admin_request POST \ "/api/admin/system/acceptance/runs/$previous_run_id/abort" \ "$body" "$abort_response" [[ $(jq -r '.mode // empty' "$abort_response") == live ]] || { echo 'previous failed acceptance Run did not return traffic to live' >&2 return 1 } elif [[ $traffic_mode != live ]]; then echo "unsupported production traffic mode before acceptance: $traffic_mode" >&2 return 1 fi admin_request POST "/api/admin/system/acceptance/runs/$run_id/activate" '' "$activate_response" [[ $(jq -r '.mode' "$activate_response") == validation && $(jq -r '.runId' "$activate_response") == "$run_id" ]] } mark_run_failed() { local reason=$1 local response=$temporary_root/failed-run.json local body [[ -n $run_id ]] || return 0 body=$(jq -cn --arg reason "$reason" \ '{passed:false,failureReason:$reason,report:{queueFinal:null}}') if admin_request POST "/api/admin/system/acceptance/runs/$run_id/finish" "$body" "$response"; then failure_recorded=true fi } apply_capacity_profile() { local profile=$1 local slots pool media global case $profile in P24) slots=24; pool=32; media=24; global=48 ;; P28) slots=28; pool=36; media=28; global=56 ;; P32) slots=32; pool=40; media=32; global=64 ;; *) return 1 ;; esac active_profile=$profile for site in ningbo hongkong; do remote_kubectl set env "deployment/easyai-worker-$site" -n "$namespace" \ "AI_GATEWAY_ASYNC_WORKER_INSTANCE_HARD_LIMIT=$slots" \ "AI_GATEWAY_ASYNC_WORKER_HARD_LIMIT=$global" \ "AI_GATEWAY_DATABASE_MAX_CONNS=$pool" \ "AI_GATEWAY_MEDIA_MATERIALIZATION_CONCURRENCY=$media" \ "AI_GATEWAY_MEDIA_REQUEST_CONCURRENCY=$media" >/dev/null remote_kubectl rollout status "deployment/easyai-worker-$site" \ -n "$namespace" --timeout=300s remote_kubectl set env "deployment/easyai-api-$site" -n "$namespace" \ "AI_GATEWAY_ASYNC_WORKER_HARD_LIMIT=$global" >/dev/null remote_kubectl rollout status "deployment/easyai-api-$site" \ -n "$namespace" --timeout=300s done } run_load_profile() { local profile=$1 local report_path=$2 local load_pod remote_script [[ $profile =~ ^[a-z-]+$ ]] || return 1 load_pod=$(remote_kubectl get pods -n "$namespace" \ -l 'app.kubernetes.io/name=easyai-acceptance-emulator' -o json | jq -r '[.items[] | select(any(.status.conditions[]?; .type=="Ready" and .status=="True")) | .metadata.name][0] // empty') [[ $load_pod =~ ^easyai-acceptance-emulator-[a-z0-9-]+$ ]] || { echo 'acceptance load runner Pod is not Ready' >&2 return 1 } # Feed private values through stdin so the API Key and Run Token never enter # the Pod spec, remote command arguments, process list, or report. # shellcheck disable=SC2016 # This script runs inside the release image. remote_script=' set -eu read -r gateways_b64 read -r tls_name_b64 read -r emulator_b64 read -r api_key_b64 read -r run_id_b64 read -r run_token_b64 read -r gemini_model_b64 read -r video_model_b64 read -r image_urls_b64 decode() { printf "%s" "$1" | base64 -d } export AI_GATEWAY_ACCEPTANCE_GATEWAYS=$(decode "$gateways_b64") export AI_GATEWAY_ACCEPTANCE_GATEWAY_TLS_SERVER_NAME=$(decode "$tls_name_b64") export AI_GATEWAY_ACCEPTANCE_EMULATOR_URL=$(decode "$emulator_b64") export AI_GATEWAY_ACCEPTANCE_API_KEY=$(decode "$api_key_b64") export AI_GATEWAY_ACCEPTANCE_RUN_ID=$(decode "$run_id_b64") export AI_GATEWAY_ACCEPTANCE_RUN_TOKEN=$(decode "$run_token_b64") export AI_GATEWAY_ACCEPTANCE_GEMINI_MODEL=$(decode "$gemini_model_b64") export AI_GATEWAY_ACCEPTANCE_VIDEO_MODEL=$(decode "$video_model_b64") export AI_GATEWAY_ACCEPTANCE_REAL_IMAGE_URLS=$(decode "$image_urls_b64") exec /app/easyai-ai-gateway-acceptance-load -profile "$1" ' { printf '%s' "$AI_GATEWAY_ACCEPTANCE_GATEWAYS" | openssl base64 -A printf '\n' printf '%s' "$AI_GATEWAY_ACCEPTANCE_GATEWAY_TLS_SERVER_NAME" | openssl base64 -A printf '\n' printf '%s' 'http://easyai-acceptance-emulator.easyai.svc.cluster.local:8090' | openssl base64 -A printf '\n' printf '%s' "$AI_GATEWAY_ACCEPTANCE_API_KEY" | openssl base64 -A printf '\n' printf '%s' "$run_id" | openssl base64 -A printf '\n' printf '%s' "$run_token" | openssl base64 -A printf '\n' printf '%s' "$AI_GATEWAY_ACCEPTANCE_GEMINI_MODEL" | openssl base64 -A printf '\n' printf '%s' "$AI_GATEWAY_ACCEPTANCE_VIDEO_MODEL" | openssl base64 -A printf '\n' printf '%s' "$AI_GATEWAY_ACCEPTANCE_REAL_IMAGE_URLS" | openssl base64 -A printf '\n' } | remote_kubectl exec -i -n "$namespace" "$load_pod" -- \ sh -c "$remote_script" -- "$profile" >"$report_path" } kill_one_worker_during_recovery() { local deadline=$((SECONDS + 120)) local running pod while (( SECONDS < deadline )); do running=$(database_query "SELECT count(*) FROM gateway_tasks WHERE acceptance_run_id='$run_id'::uuid AND status='running' AND request::text LIKE '%acceptance-long-recovery%';") if (( running > 0 )); then break fi sleep 1 done (( running > 0 )) || { echo 'recovery workload never reached a running Worker' >&2 return 1 } pod=$(remote_kubectl get pods -n "$namespace" \ -l 'app.kubernetes.io/name=easyai-worker,easyai.io/site=hongkong' \ -o 'jsonpath={.items[0].metadata.name}') [[ $pod =~ ^easyai-worker-hongkong-[a-z0-9-]+$ ]] remote_kubectl delete pod "$pod" -n "$namespace" --wait=false >/dev/null remote_kubectl rollout status deployment/easyai-worker-hongkong \ -n "$namespace" --timeout=300s } run_recovery_profile() { local report_path=$1 run_load_profile video-recovery "$report_path" & local load_pid=$! if ! kill_one_worker_during_recovery; then kill "$load_pid" >/dev/null 2>&1 || true wait "$load_pid" >/dev/null 2>&1 || true return 1 fi wait "$load_pid" } verify_wireguard() { local source_host=$1 local target_ip=$2 local max_rtt=$3 local label=$4 local output loss average handshakes now latest age output=$(cluster_ssh "$source_host" "ping -q -c 10 -W 2 $target_ip") loss=$(awk -F',' '/packet loss/ {gsub(/[^0-9.]/, "", $3); print $3}' <<<"$output") average=$(awk -F'=' '/min\\/avg\\/max/ {split($2, values, \"/\"); gsub(/ /, "", values[2]); print values[2]}' <<<"$output") [[ -n $loss && -n $average ]] awk -v loss="$loss" -v average="$average" -v max="$max_rtt" \ 'BEGIN { exit !(loss < 1 && average < max) }' now=$(date +%s) latest=$(cluster_ssh "$source_host" "wg show all latest-handshakes | awk '{print \\$3}' | sort -nr | head -n 2") handshakes=$(wc -l <<<"$latest" | tr -d ' ') (( handshakes >= 2 )) while IFS= read -r timestamp; do [[ $timestamp =~ ^[0-9]+$ && $timestamp -gt 0 ]] age=$((now - timestamp)) (( age < 180 )) done <<<"$latest" echo "wireguard=$label loss_percent=$loss average_rtt_ms=$average recent_handshakes=$handshakes" } verify_cluster_links() { verify_wireguard "$CLUSTER_NINGBO_HOST" 10.77.0.2 80 'ningbo_to_hongkong' verify_wireguard "$CLUSTER_HONGKONG_HOST" 10.77.0.1 80 'hongkong_to_ningbo' verify_wireguard "$CLUSTER_NINGBO_HOST" 10.77.0.3 300 'ningbo_to_los_angeles' verify_wireguard "$CLUSTER_LOS_ANGELES_HOST" 10.77.0.1 300 'los_angeles_to_ningbo' verify_wireguard "$CLUSTER_HONGKONG_HOST" 10.77.0.3 300 'hongkong_to_los_angeles' verify_wireguard "$CLUSTER_LOS_ANGELES_HOST" 10.77.0.2 300 'los_angeles_to_hongkong' } verify_control_plane_and_recent_logs() { local host ready_output pod error_count for host in "$CLUSTER_NINGBO_HOST" "$CLUSTER_HONGKONG_HOST" "$CLUSTER_LOS_ANGELES_HOST"; do ready_output=$(cluster_ssh "$host" "k3s kubectl get --raw='/readyz?verbose'") grep -Fq '[+]etcd ok' <<<"$ready_output" done while read -r pod; do error_count=$(remote_kubectl logs "$pod" -n "$namespace" --since=10m | awk 'BEGIN {IGNORECASE=1} /postgres_unavailable/ || /postgres readiness.*(fail|unavailable|error)/ || /leadership elector.*(error|fail)/ || /concurrency lease.*(renew.*(error|fail)|lost)/ {count++} END {print count+0}') [[ $error_count == 0 ]] done < <(remote_kubectl get pods -n "$namespace" \ -l 'app.kubernetes.io/name=easyai-worker' \ -o jsonpath='{range .items[*]}{.metadata.name}{"\n"}{end}') while read -r pod; do error_count=$(remote_kubectl logs "$pod" -n "$namespace" -c postgres --since=10m | awk 'BEGIN {IGNORECASE=1} /apply request took too long/ || /dial tcp.*6443/ || /synchronous commit.*cancel/ || /canceling statement due to user request/ || /(liveness|readiness).*fail/ {count++} END {print count+0}') [[ $error_count == 0 ]] done < <(remote_kubectl get pods -n "$namespace" \ -l 'cnpg.io/cluster=easyai-postgres' \ -o jsonpath='{range .items[*]}{.metadata.name}{"\n"}{end}') } current_max_pod_memory_mib() { remote_kubectl top pods -n "$namespace" \ -l 'app.kubernetes.io/part-of=easyai-ai-gateway' --no-headers | awk '{ value=$3 if (value ~ /Gi$/) {sub(/Gi$/, "", value); value*=1024} else if (value ~ /Mi$/) {sub(/Mi$/, "", value)} else if (value ~ /Ki$/) {sub(/Ki$/, "", value); value/=1024} if (value>max) max=value } END {printf "%.0f", max+0}' } wait_for_memory_recovery() { local baseline_mib=$1 local deadline=$((SECONDS + 120)) local current_mib while (( SECONDS < deadline )); do current_mib=$(current_max_pod_memory_mib) if (( current_mib <= baseline_mib + 256 )); then return 0 fi sleep 5 done echo "Pod memory did not return near baseline: baseline_mib=$baseline_mib current_mib=$current_mib" >&2 return 1 } verify_runtime_gates() { local expected_slots expected_global expected_pool case $active_profile in P24) expected_slots=24; expected_pool=32; expected_global=48 ;; P28) expected_slots=28; expected_pool=36; expected_global=56 ;; P32) expected_slots=32; expected_pool=40; expected_global=64 ;; esac [[ $(remote_kubectl get nodes -o json | jq '[.items[] | select(any(.status.conditions[]; .type=="Ready" and .status=="True"))] | length') == 3 ]] [[ $(remote_kubectl get nodes -o json | jq '[.items[] | select(any(.status.conditions[]; .type=="MemoryPressure" and .status=="True"))] | length') == 0 ]] [[ $(remote_kubectl get cluster easyai-postgres -n "$namespace" -o 'jsonpath={.status.readyInstances}') == 2 ]] local sync_state sync_state=$(database_query "SELECT COALESCE(string_agg(sync_state,','),'') FROM pg_stat_replication;") [[ ",$sync_state," == *,sync,* || ",$sync_state," == *,quorum,* ]] verify_control_plane_and_recent_logs local unhealthy_containers pod_memory unhealthy_containers=$(remote_kubectl get pods -n "$namespace" -o json | jq '[.items[] | select(.metadata.labels["app.kubernetes.io/name"] == "easyai-api" or .metadata.labels["app.kubernetes.io/name"] == "easyai-worker") | .status.containerStatuses[]? | select(.restartCount > 0 or .lastState.terminated.reason == "OOMKilled" or .state.terminated.reason == "OOMKilled")] | length') [[ $unhealthy_containers == 0 ]] while read -r _node _cpu _cpu_percent _memory memory_percent; do memory_percent=${memory_percent%\%} [[ $memory_percent =~ ^[0-9]+$ ]] (( memory_percent < 80 )) done < <(remote_kubectl top nodes --no-headers) while read -r _pod _cpu pod_memory; do case $pod_memory in *Gi) pod_memory=$(awk -v value="${pod_memory%Gi}" 'BEGIN { printf "%.0f", value * 1024 }') ;; *Mi) pod_memory=${pod_memory%Mi} ;; *Ki) pod_memory=$(awk -v value="${pod_memory%Ki}" 'BEGIN { printf "%.0f", value / 1024 }') ;; *) return 1 ;; esac [[ $pod_memory =~ ^[0-9]+$ ]] (( pod_memory < 1536 )) done < <(remote_kubectl top pods -n "$namespace" \ -l 'app.kubernetes.io/part-of=easyai-ai-gateway' --no-headers) local pod_name pod_anon_mib while read -r pod_name; do pod_anon_mib=$(remote_kubectl exec -n "$namespace" "$pod_name" -- \ cat /sys/fs/cgroup/memory.stat | awk '$1=="anon" {printf "%.0f", $2/1048576}') [[ $pod_anon_mib =~ ^[0-9]+$ ]] (( pod_anon_mib < 1536 )) done < <(remote_kubectl get pods -n "$namespace" -o json | jq -r '.items[] | select(.metadata.labels["app.kubernetes.io/name"] == "easyai-api" or .metadata.labels["app.kubernetes.io/name"] == "easyai-worker") | .metadata.name') local completion_deadline=$((SECONDS + 180)) local incomplete side_effects while (( SECONDS < completion_deadline )); do incomplete=$(database_query " SELECT count(*) FROM gateway_tasks WHERE acceptance_run_id='$run_id'::uuid AND ( status <> 'succeeded' OR billing_status <> 'settled' );") side_effects=$(database_query " SELECT (SELECT count(*) FROM settlement_outbox settlement WHERE settlement.task_id IN ( SELECT id FROM gateway_tasks WHERE acceptance_run_id='$run_id'::uuid ) AND settlement.status <> 'completed') + (SELECT count(*) FROM gateway_tasks task WHERE task.acceptance_run_id='$run_id'::uuid AND NOT EXISTS ( SELECT 1 FROM gateway_task_callback_outbox callback WHERE callback.task_id=task.id AND callback.status='delivered' ));") if [[ $incomplete == 0 && $side_effects == 0 ]]; then break fi sleep 2 done [[ $incomplete == 0 && $side_effects == 0 ]] local queue_state duplicates settlements wallet_duplicates missing_billings callbacks connections max_connections local active_workers stale_workers queue_state=$(database_query "SELECT count(*) FILTER (WHERE status='queued')||':'||count(*) FILTER (WHERE status='running') FROM gateway_tasks WHERE acceptance_run_id='$run_id'::uuid;") [[ $queue_state == 0:0 ]] duplicates=$(database_query "SELECT count(*) FROM (SELECT remote_task_id FROM gateway_tasks WHERE acceptance_run_id='$run_id'::uuid AND remote_task_id IS NOT NULL GROUP BY remote_task_id HAVING count(*)>1) duplicate;") [[ $duplicates == 0 ]] settlements=$(database_query "SELECT count(*) FROM (SELECT task_id,action FROM settlement_outbox WHERE task_id IN (SELECT id FROM gateway_tasks WHERE acceptance_run_id='$run_id'::uuid) GROUP BY task_id,action HAVING count(*)>1) duplicate;") [[ $settlements == 0 ]] wallet_duplicates=$(database_query " SELECT count(*) FROM ( SELECT reference_id,transaction_type FROM gateway_wallet_transactions WHERE reference_type='gateway_task' AND reference_id IN ( SELECT id::text FROM gateway_tasks WHERE acceptance_run_id='$run_id'::uuid ) GROUP BY reference_id,transaction_type HAVING count(*)>1 ) duplicate;") missing_billings=$(database_query " SELECT count(*) FROM gateway_tasks task WHERE task.acceptance_run_id='$run_id'::uuid AND NOT EXISTS ( SELECT 1 FROM gateway_wallet_transactions transaction WHERE transaction.reference_type='gateway_task' AND transaction.reference_id=task.id::text AND transaction.transaction_type='task_billing' );") [[ $wallet_duplicates == 0 && $missing_billings == 0 ]] callbacks=$(database_query " SELECT count(*) FROM gateway_task_callback_outbox WHERE task_id IN ( SELECT id FROM gateway_tasks WHERE acceptance_run_id='$run_id'::uuid ) AND status <> 'delivered';") [[ $callbacks == 0 ]] active_workers=$(database_query " SELECT count(*) FROM gateway_worker_instances WHERE status='active' AND heartbeat_at > now()-interval '30 seconds';") stale_workers=$(database_query " SELECT count(*) FROM gateway_worker_instances WHERE status='active' AND ( heartbeat_at <= now()-interval '30 seconds' OR desired_capacity <> $expected_slots OR capacity_limit <> $expected_slots OR allocated_capacity > $expected_slots );") [[ $active_workers == 2 && $stale_workers == 0 ]] connections=$(database_query "SELECT count(*) FROM pg_stat_activity WHERE backend_type='client backend';") max_connections=$(database_query "SELECT setting::int FROM pg_settings WHERE name='max_connections';") (( connections * 4 < max_connections * 3 )) local pod metrics desired current global allocated active_instances pool_max acquired idle local canceled_acquires lease_failures lease_lost for _ in 1 2 3 4 5 6; do for site in ningbo hongkong; do pod=$(remote_kubectl get pods -n "$namespace" \ -l "app.kubernetes.io/name=easyai-worker,easyai.io/site=$site" \ -o 'jsonpath={.items[0].metadata.name}') metrics=$(remote_kubectl exec -n "$namespace" "$pod" -- wget -qO- http://127.0.0.1:8088/metrics) desired=$(awk '$1=="easyai_gateway_async_worker_desired_capacity" {print $2}' <<<"$metrics") current=$(awk '$1=="easyai_gateway_async_worker_capacity" {print $2}' <<<"$metrics") global=$(awk '$1=="easyai_gateway_worker_global_capacity" {print $2}' <<<"$metrics") allocated=$(awk '$1=="easyai_gateway_worker_allocated_capacity" {print $2}' <<<"$metrics") active_instances=$(awk '$1=="easyai_gateway_worker_active_instances" {print $2}' <<<"$metrics") pool_max=$(awk '$1=="easyai_gateway_postgres_pool_max_connections" {print $2}' <<<"$metrics") acquired=$(awk '$1=="easyai_gateway_postgres_pool_acquired_connections" {print $2}' <<<"$metrics") idle=$(awk '$1=="easyai_gateway_postgres_pool_idle_connections" {print $2}' <<<"$metrics") canceled_acquires=$(awk '$1=="easyai_gateway_postgres_pool_canceled_acquire_total" {print $2}' <<<"$metrics") lease_failures=$(awk 'index($1,"easyai_gateway_concurrency_lease_renewals_total{outcome=\"failure\"}")==1 {print $2}' <<<"$metrics") lease_lost=$(awk 'index($1,"easyai_gateway_concurrency_lease_renewals_total{outcome=\"lost\"}")==1 {print $2}' <<<"$metrics") [[ ${desired%.*} == "$expected_slots" && ${current%.*} == "$expected_slots" ]] [[ ${global%.*} == "$expected_global" && ${allocated%.*} -le "$expected_slots" ]] [[ ${active_instances%.*} == 2 ]] [[ ${pool_max%.*} == "$expected_pool" ]] [[ ${canceled_acquires%.*} == 0 && ${lease_failures%.*} == 0 && ${lease_lost%.*} == 0 ]] if [[ ${acquired%.*} == "$expected_pool" && ${idle%.*} == 0 ]]; then echo "Worker connection pool saturated for site $site" >&2 return 1 fi done sleep 5 done verify_cluster_links curl -fsS --max-time 10 https://ai.51easyai.com/api/v1/healthz >/dev/null curl -fsS --max-time 10 https://ai.51easyai.com/api/v1/readyz >/dev/null } sample_pressure() { local output=$1 local stop_file=$2 local database_state node_memory_percent pod_memory_mib pool_state pod metrics echo 'timestamp,queued,running,oldest_wait_seconds,db_connections,db_max_connections,node_memory_percent,pod_memory_mib,pool_acquired_max,pool_max,pool_idle_min,empty_acquire_total,canceled_acquire_total,lease_failure_total,lease_lost_total,active_instances_min,allocated_capacity_max' >"$output" while [[ ! -f $stop_file ]]; do database_state=$(database_query " SELECT count(*) FILTER (WHERE status='queued')||','|| count(*) FILTER (WHERE status='running')||','|| COALESCE(EXTRACT(EPOCH FROM (now()-(min(created_at) FILTER (WHERE status='queued')))),0)::bigint||','|| (SELECT count(*) FROM pg_stat_activity WHERE backend_type='client backend')||','|| (SELECT setting FROM pg_settings WHERE name='max_connections') FROM gateway_tasks WHERE acceptance_run_id='$run_id'::uuid;") || break node_memory_percent=$(remote_kubectl top nodes --no-headers | awk '{value=$5; sub(/%$/, "", value); if (value>max) max=value} END {print max+0}') || break pod_memory_mib=$(current_max_pod_memory_mib) || break pool_state=$( for site in ningbo hongkong; do pod=$(remote_kubectl get pods -n "$namespace" \ -l "app.kubernetes.io/name=easyai-worker,easyai.io/site=$site" -o json | jq -r '[.items[] | select(any(.status.conditions[]?; .type=="Ready" and .status=="True")) | .metadata.name][0] // empty') || exit [[ -n $pod ]] || continue metrics=$(remote_kubectl exec -n "$namespace" "$pod" -- \ wget -qO- http://127.0.0.1:8088/metrics) || exit awk ' $1=="easyai_gateway_postgres_pool_acquired_connections" {acquired=$2} $1=="easyai_gateway_postgres_pool_max_connections" {pool_max=$2} $1=="easyai_gateway_postgres_pool_idle_connections" {idle=$2} $1=="easyai_gateway_postgres_pool_empty_acquire_total" {empty=$2} $1=="easyai_gateway_postgres_pool_canceled_acquire_total" {canceled=$2} index($1,"easyai_gateway_concurrency_lease_renewals_total{outcome=\"failure\"}")==1 {failure=$2} index($1,"easyai_gateway_concurrency_lease_renewals_total{outcome=\"lost\"}")==1 {lost=$2} $1=="easyai_gateway_worker_active_instances" {active=$2} $1=="easyai_gateway_worker_allocated_capacity" {allocated=$2} END {print acquired "," pool_max "," idle "," empty "," canceled "," failure "," lost "," active "," allocated} ' <<<"$metrics" done | awk -F',' ' NR==1 { acquired=$1; pool_max=$2; idle=$3; empty=$4; canceled=$5 failure=$6; lost=$7; active=$8; allocated=$9 next } { if ($1>acquired) acquired=$1 if ($2>pool_max) pool_max=$2 if ($3allocated) allocated=$9 } END {print acquired "," pool_max "," idle "," empty "," canceled "," failure "," lost "," active "," allocated} ' ) || break printf '%s,%s,%s,%s,%s\n' \ "$(date -u '+%Y-%m-%dT%H:%M:%SZ')" "$database_state" "$node_memory_percent" "$pod_memory_mib" "$pool_state" >>"$output" sleep 5 done } verify_pressure_report() { local report_path=$1 local expected_slots=$2 awk -F',' ' NR == 1 { next } { samples++ if (samples == 1) { first_empty=$12 first_canceled=$13 first_lease_failure=$14 first_lease_lost=$15 } last_empty=$12 last_canceled=$13 last_lease_failure=$14 last_lease_lost=$15 if ($13 > first_canceled || $14 > first_lease_failure || $15 > first_lease_lost) { failed=1 } if ($9 >= $10 && $11 == 0) { saturated++ } else { saturated=0 } if ($4 >= 900 || $6 <= 0 || $5 * 4 >= $6 * 3 || $7 >= 80 || $8 >= 1536 || $10 <= 0 || saturated >= 6 || $16 < 1 || $16 > 2 || $17 > slots) { failed=1 } } END { if (last_canceled > first_canceled || last_lease_failure > first_lease_failure || last_lease_lost > first_lease_lost || (last_empty > first_empty && saturated >= 6)) { failed=1 } exit !(samples > 0 && failed == 0) } ' slots="$expected_slots" "$report_path" } run_capacity_round() { local profile=$1 local repetition=$2 local profile_slug profile_slug=$(printf '%s' "$profile" | tr '[:upper:]' '[:lower:]') local prefix=$report_root/$profile_slug-$repetition local pressure_report=$prefix-pressure.csv local pressure_stop=$temporary_root/pressure-stop local pressure_pid status=0 local baseline_memory_mib local expected_slots case $profile in P24) expected_slots=24 ;; P28) expected_slots=28 ;; P32) expected_slots=32 ;; esac baseline_memory_mib=$(current_max_pod_memory_mib) rm -f -- "$pressure_stop" sample_pressure "$pressure_report" "$pressure_stop" & pressure_pid=$! run_load_profile gemini-baseline "$prefix-gemini-baseline.json" || status=$? if (( status == 0 )); then run_load_profile gemini-large "$prefix-gemini-large.json" || status=$?; fi if (( status == 0 )); then run_load_profile gemini-peak "$prefix-gemini-peak.json" || status=$?; fi if (( status == 0 )); then run_load_profile video-throughput "$prefix-video-throughput.json" || status=$?; fi if (( status == 0 )); then run_recovery_profile "$prefix-video-recovery.json" || status=$?; fi touch "$pressure_stop" wait "$pressure_pid" || true (( status == 0 )) || return "$status" verify_pressure_report "$pressure_report" "$expected_slots" verify_runtime_gates wait_for_memory_recovery "$baseline_memory_mib" } finish_and_promote() { local emulator_report=$report_root/emulator-report.json local finish_response=$temporary_root/finish.json local mode_response=$temporary_root/mode.json remote_kubectl exec -n "$namespace" deployment/easyai-acceptance-emulator -- \ wget -qO- http://127.0.0.1:8090/report >"$emulator_report" local task_count video_task_count gemini_task_count submissions duplicates callbacks local gemini_requests forced_tasks forced_requests verified_conversions compact_payloads task_count=$(database_query "SELECT count(*) FROM gateway_tasks WHERE acceptance_run_id='$run_id'::uuid;") video_task_count=$(database_query "SELECT count(*) FROM gateway_tasks WHERE acceptance_run_id='$run_id'::uuid AND kind='videos.generations' AND run_mode='acceptance';") gemini_task_count=$(database_query "SELECT count(*) FROM gateway_tasks WHERE acceptance_run_id='$run_id'::uuid AND kind='images.edits' AND run_mode='acceptance';") forced_tasks=$(database_query "SELECT count(*) FROM gateway_tasks WHERE acceptance_run_id='$run_id'::uuid AND request::text LIKE '%acceptance-force-conversion%';") compact_payloads=$(database_query "SELECT count(*) FROM gateway_tasks WHERE acceptance_run_id='$run_id'::uuid AND kind='images.edits' AND (pg_column_size(request) >= 1048576 OR pg_column_size(result) >= 1048576);") submissions=$(jq -r '.videoSubmissions' "$emulator_report") duplicates=$(jq -r '.duplicateCallbacks' "$emulator_report") callbacks=$(jq -r '.callbackEvents' "$emulator_report") gemini_requests=$(jq -r '.geminiRequests' "$emulator_report") forced_requests=$(jq -r '.forcedConversionRequests' "$emulator_report") verified_conversions=$(jq -r '.verifiedConversions' "$emulator_report") [[ $submissions == "$video_task_count" && $gemini_requests == "$gemini_task_count" ]] [[ $forced_requests == "$forced_tasks" && $verified_conversions == "$forced_tasks" ]] [[ $duplicates == 0 && $compact_payloads == 0 ]] [[ $callbacks -gt 0 ]] admin_request GET /api/admin/system/acceptance/traffic-mode '' "$mode_response" verify_release_cas local promotion_body promotion_body=$(jq -cn \ --argjson revision "$(jq -r '.revision' "$mode_response")" \ --arg releaseSha "$release_sha" \ --arg apiDigest "$api_digest" \ --arg workerDigest "$worker_digest" \ '{revision:$revision,releaseSha:$releaseSha,apiImageDigest:$apiDigest,workerImageDigest:$workerDigest}') jq -n \ --arg runId "$run_id" \ --arg releaseSha "$release_sha" \ --arg apiImageDigest "$api_digest" \ --arg workerImageDigest "$worker_digest" \ --arg stableCapacityProfile "$stable_profile" \ --argjson tasks "$task_count" \ --argjson geminiTasks "$gemini_task_count" \ --argjson videoTasks "$video_task_count" \ --argjson upstreamSubmissions "$submissions" \ --argjson callbackDeliveries "$callbacks" \ '{ runId:$runId, releaseSha:$releaseSha, apiImageDigest:$apiImageDigest, workerImageDigest:$workerImageDigest, stableCapacityProfile:$stableCapacityProfile, tasks:$tasks, geminiTasks:$geminiTasks, videoTasks:$videoTasks, upstreamSubmissions:$upstreamSubmissions, callbackDeliveries:$callbackDeliveries, promotionRequested:true, passed:true }' >"$report_root/summary.json" local finish_body finish_body=$(jq -cn \ --argjson tasks "$task_count" \ --argjson videos "$video_task_count" \ --argjson submissions "$submissions" \ --arg profile "$stable_profile" \ '{passed:true,report:{tasks:$tasks,videos:$videos,upstreamSubmissions:$submissions,stableCapacityProfile:$profile,queueFinal:0}}') admin_request POST "/api/admin/system/acceptance/runs/$run_id/finish" "$finish_body" "$finish_response" admin_request POST "/api/admin/system/acceptance/runs/$run_id/promote" "$promotion_body" "$finish_response" [[ $(jq -r '.mode' "$finish_response") == live ]] } verify_release_cas ensure_acceptance_user_group select_acceptance_models ensure_acceptance_model_access deploy_protocol_emulator create_and_activate_run wait_for_existing_tasks_to_drain for profile in P24 P28 P32; do if ! apply_capacity_profile "$profile"; then failure_reason="failed to apply $profile" break fi profile_passed=true for repetition in 1 2 3; do if ! run_capacity_round "$profile" "$repetition"; then profile_passed=false failure_reason="$profile repetition $repetition failed" break fi done if [[ $profile_passed != true ]]; then apply_capacity_profile "$stable_profile" || true break fi stable_profile=$profile done if [[ -n $failure_reason ]]; then mark_run_failed "$failure_reason" echo "production_acceptance=FAIL run_id=$run_id reason=$failure_reason traffic_mode=validation restored_profile=$stable_profile" >&2 exit 1 fi if ! run_load_profile real-canary "$report_root/real-canary.json"; then failure_reason='real canary failed' elif ! verify_runtime_gates; then failure_reason='post-canary runtime gates failed' fi if [[ -n $failure_reason ]]; then mark_run_failed "$failure_reason" echo "production_acceptance=FAIL run_id=$run_id reason=$failure_reason traffic_mode=validation restored_profile=$stable_profile" >&2 exit 1 fi finish_and_promote remote_kubectl delete deployment easyai-acceptance-emulator -n "$namespace" --ignore-not-found >/dev/null || echo 'warning: acceptance emulator Deployment cleanup failed' >&2 remote_kubectl delete service easyai-acceptance-emulator -n "$namespace" --ignore-not-found >/dev/null || echo 'warning: acceptance emulator Service cleanup failed' >&2 echo "production_acceptance=PASS run_id=$run_id release=$release_sha stable_profile=$stable_profile traffic_mode=live"